From 06f0f3a2471b225547347ee44cfdf5fb19476130 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Thu, 8 Oct 2026 10:21:00 -0400 Subject: [PATCH 1/6] Anchor UK year-file runs on the survey year policyengine-uk takes the first year of a dataset as observed data: its State Pension formulas split state_pension_reported at that year against the year's legislated rates and scale the share to the simulated year's rates (the triple lock). create_datasets cut year files from the projection, and run() passed a year file back as a single-year dataset, so the projected year became the observed year and the State Pension followed the CPI uprating of the reported amount. 2026 State Pension was GBP 127.50bn through policyengine.py against GBP 133.69bn from policyengine-uk on the same certified data (6.2.1). Year files now record their data year and keep its tables. run() projects those tables forward as policyengine-uk does and puts the year file's own tables at the simulated year, matching records by ID so region scoping carries over. Year files without a recorded data year are regenerated by ensure_datasets and refused by load_datasets. Fixes #556 Co-Authored-By: Claude Opus 5.5 --- changelog.d/556.fixed.md | 1 + docs/microsim.md | 26 + .../tax_benefit_models/uk/datasets.py | 288 ++++++++--- .../tax_benefit_models/uk/model.py | 122 ++++- tests/test_uk_year_file_data_year.py | 488 ++++++++++++++++++ 5 files changed, 837 insertions(+), 88 deletions(-) create mode 100644 changelog.d/556.fixed.md create mode 100644 tests/test_uk_year_file_data_year.py diff --git a/changelog.d/556.fixed.md b/changelog.d/556.fixed.md new file mode 100644 index 00000000..34c48a6d --- /dev/null +++ b/changelog.d/556.fixed.md @@ -0,0 +1 @@ +UK population runs now anchor the State Pension on the survey year their data was projected from, as policyengine-uk does, so a run of a UK year file matches policyengine-uk on the certified dataset. Before, it took the projected year as the survey year and uprated the State Pension by CPI rather than the triple lock. Year files written by earlier releases are regenerated. diff --git a/docs/microsim.md b/docs/microsim.md index 98cdc7bc..9a64f053 100644 --- a/docs/microsim.md +++ b/docs/microsim.md @@ -231,6 +231,32 @@ not need repository-specific download logic. Authentication or authorization failures are reported directly and do not cause a retry against another repository type. +### UK year files and the data year + +`ensure_datasets` and `create_datasets` cut one file per requested year from +the projection policyengine-uk makes of the certified dataset. policyengine-uk +takes the first year of a dataset as observed data. Its State Pension formulas +split each person's reported State Pension against that year's legislated +rates, and scale the share to the simulated year's rates, which follow the +triple lock. The projection uprates the reported amount by CPI, so handing a +projected year's tables to policyengine-uk as observed data would make the +State Pension follow CPI instead. + +Each projected year file therefore keeps its data year, `dataset.data_year` +(2024 for Enhanced FRS 2024–25), and that year's tables, +`dataset.data_year_data`. `Simulation.run()` projects the data year's tables +forward as policyengine-uk does, and uses the file's own tables for the +simulated year. A run of a year file then gives the same result, record by +record, as `policyengine_uk.Microsimulation` on the certified file. Region +scoping applies to both sets of tables, matched by entity ID. A dataset built +in memory without a data year is observed data for its own year, which is how +policyengine-uk treats a single-year dataset. + +Year files written by earlier releases have no recorded data year. +`ensure_datasets` writes them again and `load_datasets` refuses them. A year +file opened directly, as `PolicyEngineUKDataset(filepath=...)`, is not +checked. + ## Simulations A `Simulation` needs a dataset, a tax-benefit model version, and optionally a policy (reform): diff --git a/src/policyengine/tax_benefit_models/uk/datasets.py b/src/policyengine/tax_benefit_models/uk/datasets.py index 105dfdd0..e7d79368 100644 --- a/src/policyengine/tax_benefit_models/uk/datasets.py +++ b/src/policyengine/tax_benefit_models/uk/datasets.py @@ -15,6 +15,19 @@ resolve_dataset_reference, ) +#: HDF5 key holding the data year of a UK year file: the year of the observed +#: data the file's tables were projected from (see +#: ``PolicyEngineUKDataset.data_year``). ``create_datasets`` writes it into +#: every year file. Year files written before policyengine.py kept the data +#: year have none: policyengine-uk would take their projected tables as +#: observed data, so ``ensure_datasets`` regenerates them and +#: ``load_datasets`` refuses them. +DATA_YEAR_KEY = "data_year" + +#: Prefix of the HDF5 keys holding the data year's entity tables in a year +#: file whose year comes after its data year. +DATA_YEAR_TABLE_PREFIX = "data_year_" + class UKYearData(YearData): """Entity-level data for a single year.""" @@ -40,6 +53,27 @@ class PolicyEngineUKDataset(Dataset): data: Optional[UKYearData] = None + data_year: Optional[int] = None + """Year of the observed data that ``data`` was projected from. + + policyengine-uk takes the first year of a dataset as observed data. It + splits each person's reported State Pension against that year's + legislated rates and scales the share to the simulated year's rates, + which follow the triple lock. ``create_datasets`` records the source + dataset's first year here. ``None`` means ``data`` is itself observed + data for ``year``, as it is for a dataset built in memory. + """ + + data_year_data: Optional[UKYearData] = None + """Entity tables of ``data_year``, when it comes before ``year``. + + ``run()`` projects them forward as policyengine-uk does in a direct run, + and uses ``data`` for ``year`` itself. Without them, policyengine-uk + would take the projected ``data`` as observed data for ``year`` and the + State Pension would follow the CPI uprating of its reported amount + instead of the triple lock. + """ + def model_post_init(self, __context): """Called after Pydantic initialization. @@ -53,6 +87,40 @@ def model_post_init(self, __context): """ if self.data is None and self.filepath: self.load() + self._check_data_year() + + def _check_data_year(self) -> None: + """Refuse a data year that cannot be the source of ``year``.""" + if self.data_year is None: + if self.data_year_data is not None: + raise ValueError( + "PolicyEngineUKDataset has data-year tables but no " + "data_year. Set data_year to the year they hold." + ) + return + if self.data_year > self.year: + raise ValueError( + f"PolicyEngineUKDataset data_year {self.data_year} is after " + f"its year {self.year}; data can only be projected forward." + ) + if self.data_year == self.year and self.data_year_data is not None: + raise ValueError( + f"PolicyEngineUKDataset year {self.year} is its data year, so " + "`data` holds the observed tables and data_year_data must be " + "None." + ) + if ( + self.data_year < self.year + and self.data is not None + and self.data_year_data is None + ): + raise ValueError( + f"PolicyEngineUKDataset for {self.year} was projected from " + f"{self.data_year} but has no data_year_data, so " + "policyengine-uk cannot anchor the State Pension on the " + "observed year. Pass the data year's tables, or set " + "data_year=None to treat `data` as observed data." + ) def save(self) -> None: """Save dataset to HDF5 file. @@ -69,41 +137,30 @@ def save(self) -> None: if not filepath.parent.exists(): filepath.parent.mkdir(parents=True, exist_ok=True) - # Convert DataFrames and optimize object columns to categorical - person_df = pd.DataFrame(self.data.person) - benunit_df = pd.DataFrame(self.data.benunit) - household_df = pd.DataFrame(self.data.household) - - # Convert object columns to categorical to avoid pickle serialization - for col in person_df.columns: - if person_df[col].dtype == "object": - person_df[col] = person_df[col].astype("category") - - for col in benunit_df.columns: - if benunit_df[col].dtype == "object": - benunit_df[col] = benunit_df[col].astype("category") - - for col in household_df.columns: - if household_df[col].dtype == "object": - household_df[col] = household_df[col].astype("category") - with pd.HDFStore(filepath, mode="w") as store: - # Use format='table' to support categorical dtypes - store.put("person", person_df, format="table") - store.put("benunit", benunit_df, format="table") - store.put("household", household_df, format="table") + _put_year_data(store, self.data) + if self.data_year is not None: + store.put(DATA_YEAR_KEY, pd.Series([int(self.data_year)])) + if self.data_year_data is not None: + _put_year_data(store, self.data_year_data, DATA_YEAR_TABLE_PREFIX) def load(self) -> None: """Load dataset from HDF5 file into this instance.""" filepath = self.filepath with pd.HDFStore(filepath, mode="r") as store: - self.data = UKYearData( - person=MicroDataFrame(store["person"], weights="person_weight"), - benunit=MicroDataFrame(store["benunit"], weights="benunit_weight"), - household=MicroDataFrame( - store["household"], weights="household_weight" - ), + keys = set(store.keys()) + self.data = _read_year_data(store) + self.data_year = ( + int(store[DATA_YEAR_KEY].iloc[0]) + if f"/{DATA_YEAR_KEY}" in keys + else None + ) + self.data_year_data = ( + _read_year_data(store, DATA_YEAR_TABLE_PREFIX) + if f"/{DATA_YEAR_TABLE_PREFIX}person" in keys + else None ) + self._check_data_year() def __repr__(self) -> str: if self.data is None: @@ -112,7 +169,94 @@ def __repr__(self) -> str: n_people = len(self.data.person) n_benunits = len(self.data.benunit) n_households = len(self.data.household) - return f"" + return f"" + + +def _put_year_data(store: pd.HDFStore, data: UKYearData, prefix: str = "") -> None: + """Write one year's entity tables to an open HDF5 store. + + Object columns are stored as categoricals to avoid slow pickle + serialization; ``format="table"`` supports categorical dtypes. + """ + for name, frame in data.entity_data.items(): + frame = pd.DataFrame(frame) + for column in frame.columns: + if frame[column].dtype == "object": + frame[column] = frame[column].astype("category") + store.put(f"{prefix}{name}", frame, format="table") + + +def _read_year_data(store: pd.HDFStore, prefix: str = "") -> UKYearData: + """Read one year's entity tables from an open HDF5 store.""" + return UKYearData( + person=MicroDataFrame(store[f"{prefix}person"], weights="person_weight"), + benunit=MicroDataFrame(store[f"{prefix}benunit"], weights="benunit_weight"), + household=MicroDataFrame( + store[f"{prefix}household"], weights="household_weight" + ), + ) + + +def _year_data_with_weights(year_dataset) -> UKYearData: + """Return a policyengine-uk year's tables with person and benunit weights. + + policyengine-uk stores household weights only; each person and benefit + unit takes its household's weight. + """ + # Convert to pandas DataFrames and add weight columns + person_df = pd.DataFrame(year_dataset.person) + benunit_df = pd.DataFrame(year_dataset.benunit) + household_df = pd.DataFrame(year_dataset.household) + + # Map household weights to person and benunit levels + person_df = person_df.merge( + household_df[["household_id", "household_weight"]], + left_on="person_household_id", + right_on="household_id", + how="left", + ) + person_df = person_df.rename(columns={"household_weight": "person_weight"}) + person_df = person_df.drop(columns=["household_id"]) + + # Get household_id for each benunit from person table + benunit_household_map = person_df[ + ["person_benunit_id", "person_household_id"] + ].drop_duplicates() + benunit_df = benunit_df.merge( + benunit_household_map, + left_on="benunit_id", + right_on="person_benunit_id", + how="left", + ) + benunit_df = benunit_df.merge( + household_df[["household_id", "household_weight"]], + left_on="person_household_id", + right_on="household_id", + how="left", + ) + benunit_df = benunit_df.rename(columns={"household_weight": "benunit_weight"}) + benunit_df = benunit_df.drop( + columns=[ + "person_benunit_id", + "person_household_id", + "household_id", + ], + errors="ignore", + ) + return UKYearData( + person=MicroDataFrame(person_df, weights="person_weight"), + benunit=MicroDataFrame(benunit_df, weights="benunit_weight"), + household=MicroDataFrame(household_df, weights="household_weight"), + ) + + +def _year_file_records_data_year(path: Path) -> bool: + """Return whether a UK year file records its data year (``DATA_YEAR_KEY``).""" + try: + with pd.HDFStore(path, mode="r") as store: + return f"/{DATA_YEAR_KEY}" in store.keys() + except OSError: + return False def create_datasets( @@ -134,63 +278,22 @@ def create_datasets( from policyengine_uk import Microsimulation sim = Microsimulation(dataset=source.path) + # policyengine-uk takes a dataset's first year as observed data and + # projects the rest from it (see PolicyEngineUKDataset.data_year). + # Each year file keeps that year and its tables, so a run of the + # file anchors on the same observed year as a run of the source. + data_year = int(min(sim.dataset.years)) + data_year_data = _year_data_with_weights(sim.dataset[data_year]) for year in years: - year_dataset = sim.dataset[year] - - # Convert to pandas DataFrames and add weight columns - person_df = pd.DataFrame(year_dataset.person) - benunit_df = pd.DataFrame(year_dataset.benunit) - household_df = pd.DataFrame(year_dataset.household) - - # Map household weights to person and benunit levels - person_df = person_df.merge( - household_df[["household_id", "household_weight"]], - left_on="person_household_id", - right_on="household_id", - how="left", - ) - person_df = person_df.rename(columns={"household_weight": "person_weight"}) - person_df = person_df.drop(columns=["household_id"]) - - # Get household_id for each benunit from person table - benunit_household_map = person_df[ - ["person_benunit_id", "person_household_id"] - ].drop_duplicates() - benunit_df = benunit_df.merge( - benunit_household_map, - left_on="benunit_id", - right_on="person_benunit_id", - how="left", - ) - benunit_df = benunit_df.merge( - household_df[["household_id", "household_weight"]], - left_on="person_household_id", - right_on="household_id", - how="left", - ) - benunit_df = benunit_df.rename( - columns={"household_weight": "benunit_weight"} - ) - benunit_df = benunit_df.drop( - columns=[ - "person_benunit_id", - "person_household_id", - "household_id", - ], - errors="ignore", - ) - uk_dataset = PolicyEngineUKDataset( id=f"{dataset_stem}_year_{year}", name=f"{dataset_stem}-year-{year}", description=f"UK Dataset for year {year} based on {dataset_stem}", filepath=f"{data_folder}/{dataset_stem}_year_{year}.h5", year=int(year), - data=UKYearData( - person=MicroDataFrame(person_df, weights="person_weight"), - benunit=MicroDataFrame(benunit_df, weights="benunit_weight"), - household=MicroDataFrame(household_df, weights="household_weight"), - ), + data=_year_data_with_weights(sim.dataset[year]), + data_year=data_year, + data_year_data=data_year_data if int(year) > data_year else None, ) uk_dataset.save() @@ -205,6 +308,14 @@ def load_datasets( years: list[int] = [2026, 2027, 2028, 2029, 2030], data_folder: str = "./data", ) -> dict[str, PolicyEngineUKDataset]: + """Load UK year files written by ``create_datasets``. + + A year file without a recorded data year (``DATA_YEAR_KEY``) was written + before policyengine.py kept the observed year its tables were projected + from. policyengine-uk would take those projected tables as observed data + and uprate the State Pension by CPI rather than the triple lock, so such + a file is refused. + """ if datasets is None: datasets = [get_release_manifest("uk").default_dataset] result = {} @@ -213,6 +324,17 @@ def load_datasets( dataset_stem = dataset_logical_name(resolved_dataset) for year in years: filepath = f"{data_folder}/{dataset_stem}_year_{year}.h5" + if Path(filepath).exists() and not _year_file_records_data_year( + Path(filepath) + ): + raise ValueError( + f"UK year file {filepath} has no recorded data year, so it " + "was written before policyengine.py kept the observed year " + "its tables were projected from. policyengine-uk would " + "take the projected tables as observed data and uprate the " + "State Pension by CPI rather than the triple lock. " + "Regenerate it with ensure_datasets() or create_datasets()." + ) uk_dataset = PolicyEngineUKDataset( name=f"{dataset_stem}-year-{year}", description=f"UK Dataset for year {year} based on {dataset_stem}", @@ -235,6 +357,10 @@ def ensure_datasets( ) -> dict[str, PolicyEngineUKDataset]: """Ensure datasets exist, loading if available or creating if not. + Year files without a recorded data year were written before + policyengine.py kept the observed year their tables were projected from + (see ``load_datasets``), so they are created again rather than loaded. + Args: datasets: List of HuggingFace dataset paths years: List of years to load/create data for @@ -253,7 +379,9 @@ def ensure_datasets( dataset_stem = dataset_logical_name(resolved_dataset) for year in years: filepath = Path(f"{data_folder}/{dataset_stem}_year_{year}.h5") - if not filepath.exists(): + # A year file without a recorded data year would run as observed + # data, so it is regenerated rather than reused. + if not filepath.exists() or not _year_file_records_data_year(filepath): all_exist = False break if not all_exist: diff --git a/src/policyengine/tax_benefit_models/uk/model.py b/src/policyengine/tax_benefit_models/uk/model.py index b0e22ad1..824935b5 100644 --- a/src/policyengine/tax_benefit_models/uk/model.py +++ b/src/policyengine/tax_benefit_models/uk/model.py @@ -1,6 +1,7 @@ import datetime from typing import TYPE_CHECKING, Optional +import numpy as np import pandas as pd from microdf import MicroDataFrame @@ -22,6 +23,11 @@ from policyengine.core.simulation import Simulation UK_GROUP_ENTITIES = ["benunit", "household"] +UK_ENTITY_ID_COLUMNS = { + "person": "person_id", + "benunit": "benunit_id", + "household": "household_id", +} UK_HOUSEHOLD_PASSTHROUGH_COLUMNS = [ "oa_code", "lsoa_code", @@ -159,7 +165,6 @@ def _dataset_class(self): # --- run ------------------------------------------------------------- def run(self, simulation: "Simulation") -> "Simulation": from policyengine_uk import Microsimulation - from policyengine_uk.data import UKSingleYearDataset from policyengine.utils.parametric_reforms import ( simulation_modifier_from_parameter_values, @@ -197,15 +202,13 @@ def run(self, simulation: "Simulation") -> "Simulation": benunit=scoped_data["benunit"], household=scoped_data["household"], ), + # The data year's tables stay whole; they are matched to the + # scoped records by ID when the input is built. + data_year=dataset.data_year, + data_year_data=dataset.data_year_data, ) - input_data = UKSingleYearDataset( - person=dataset.data.person, - benunit=dataset.data.benunit, - household=dataset.data.household, - fiscal_year=dataset.year, - ) - microsim = Microsimulation(dataset=input_data) + microsim = Microsimulation(dataset=_policyengine_uk_input(dataset)) if simulation.policy and simulation.policy.simulation_modifier is not None: simulation.policy.simulation_modifier(microsim) @@ -264,6 +267,109 @@ def run(self, simulation: "Simulation") -> "Simulation": ) +def _policyengine_uk_input(dataset: PolicyEngineUKDataset): + """Build the policyengine-uk dataset a run of ``dataset`` simulates. + + policyengine-uk takes the first year of its dataset as observed data: + the State Pension formulas split each person's reported State Pension + against that year's legislated rates and scale the share to the + simulated year's rates. A dataset projected from an earlier observed + year (``data_year`` before ``year``) therefore gets the input a direct + policyengine-uk run on its source builds: the data year's tables + projected forward by policyengine-uk, with ``dataset.data`` as the + simulated year. Passing ``dataset.data`` alone would make the simulated + year the observed year, and the State Pension would follow the CPI + uprating of its reported amount instead of the triple lock + (PolicyEngine/policyengine.py#556). + + Any other dataset is passed as observed data for its year, as + policyengine-uk treats a single-year dataset. + """ + from policyengine_uk.data import UKMultiYearDataset, UKSingleYearDataset + + year = int(dataset.year) + # Copies: policyengine-uk encodes enum columns in place on the tables it + # is given, which would change the caller's dataset. + simulated = UKSingleYearDataset( + person=pd.DataFrame(dataset.data.person).copy(), + benunit=pd.DataFrame(dataset.data.benunit).copy(), + household=pd.DataFrame(dataset.data.household).copy(), + fiscal_year=year, + ) + if dataset.data_year is None or int(dataset.data_year) >= year: + return simulated + + from policyengine_uk.data.economic_assumptions import ( + extend_single_year_dataset, + ) + from policyengine_uk.system import system + + data_year = int(dataset.data_year) + observed_tables = _match_records( + dataset.data_year_data.entity_data, + dataset.data.entity_data, + data_year=data_year, + year=year, + ) + observed = UKSingleYearDataset(**observed_tables, fiscal_year=data_year) + projected = extend_single_year_dataset(observed, system.parameters) + if year not in projected.years: + projected = extend_single_year_dataset( + observed, system.parameters, end_year=year + ) + return UKMultiYearDataset( + datasets=[projected[y] for y in projected.years if y != year] + [simulated] + ) + + +def _match_records( + observed: dict[str, pd.DataFrame], + simulated: dict[str, pd.DataFrame], + *, + data_year: int, + year: int, +) -> dict[str, pd.DataFrame]: + """Return the observed tables for the simulated records, in their order. + + policyengine-uk builds its entities from the first year of a dataset and + sets each year's inputs by position, so the observed year must hold the + same records in the same order as the simulated year. Records are + matched by ID, which carries region scoping (a subset of households) + over to the observed year. + """ + matched = {} + for entity, id_column in UK_ENTITY_ID_COLUMNS.items(): + table = pd.DataFrame(observed[entity]) + ids = pd.Index(table[id_column].to_numpy()) + if not ids.is_unique: + raise ValueError( + f"The {data_year} {entity} table repeats {id_column} values, " + "so its records cannot be matched to the simulated year." + ) + positions = ids.get_indexer(pd.DataFrame(simulated[entity])[id_column]) + missing = int((positions < 0).sum()) + if missing: + raise ValueError( + f"{missing} {entity} record(s) of the {year} tables are not in " + f"the {data_year} tables they were projected from, so " + "policyengine-uk has no observed data for them. Keep the data " + "year's tables in step with `data`, or set data_year=None to " + f"treat `data` as observed data for {year}." + ) + matched[entity] = table.iloc[positions].reset_index(drop=True).copy() + simulated_person = pd.DataFrame(simulated["person"]) + for link in ("person_benunit_id", "person_household_id"): + if not np.array_equal( + matched["person"][link].to_numpy(), simulated_person[link].to_numpy() + ): + raise ValueError( + f"People belong to different units ({link}) in the {data_year} " + f"and {year} tables, so the {year} tables were not projected " + f"from the {data_year} ones." + ) + return matched + + def managed_microsimulation( *, dataset: Optional[str] = None, diff --git a/tests/test_uk_year_file_data_year.py b/tests/test_uk_year_file_data_year.py new file mode 100644 index 00000000..abcfae80 --- /dev/null +++ b/tests/test_uk_year_file_data_year.py @@ -0,0 +1,488 @@ +"""UK year files reproduce a direct policyengine-uk run of their source. + +policyengine-uk takes the first year of a dataset as observed data. Its State +Pension formulas split each person's reported State Pension against that +year's legislated rates and scale the share to the simulated year's rates, +which follow the triple lock. ``create_datasets`` cuts year files from the +projection policyengine-uk makes of the source, so each file keeps the +observed year and its tables, and ``run()`` rebuilds that projection. A year +file passed to policyengine-uk on its own would make the simulated year the +observed year, and the State Pension would follow the CPI uprating of its +reported amount instead (PolicyEngine/policyengine.py#556). + +Each test writes a tiny 2024 dataset in policyengine-uk's own file format, +cuts year files from it with ``create_datasets``, and compares a +policyengine.py run of a year file with ``policyengine_uk.Microsimulation`` +on the source file, record by record. + +Invariants exercised: + +- Differential: for any source dataset and any year the source projects to, + every output of a policyengine.py run of the year file equals the direct + policyengine-uk run of the source, record by record (also under region + scoping, for the records kept). +- Round trip: saving and loading a year file preserves its data year and + the data year's tables. +- No aliasing: a run leaves the caller's dataset tables unchanged. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest +from hypothesis import HealthCheck, given, settings +from hypothesis import strategies as st +from microdf import MicroDataFrame + +pytest.importorskip("policyengine_uk") + +import policyengine as pe # noqa: E402 +from policyengine.core import Simulation # noqa: E402 +from policyengine.core.scoping_strategy import RowFilterStrategy # noqa: E402 +from policyengine.tax_benefit_models.uk.datasets import ( # noqa: E402 + DATA_YEAR_KEY, + PolicyEngineUKDataset, + UKYearData, + create_datasets, + ensure_datasets, + load_datasets, +) +from policyengine.tax_benefit_models.uk.model import ( # noqa: E402 + _policyengine_uk_input, +) + +DATA_YEAR = 2024 +STATE_PENSION = [ + "basic_state_pension", + "new_state_pension", + "additional_state_pension", + "state_pension", +] +PERSON_OUTPUTS = STATE_PENSION + ["state_pension_type", "income_tax"] +BENUNIT_OUTPUTS = ["pension_credit", "universal_credit"] +HOUSEHOLD_OUTPUTS = [ + "household_net_income", + "hbai_household_net_income", + "in_poverty_bhc", + "in_poverty_ahc", +] +EXTRA_VARIABLES = { + "person": PERSON_OUTPUTS, + "benunit": BENUNIT_OUTPUTS, + "household": HOUSEHOLD_OUTPUTS, +} + + +def _tables( + ages=(70, 85, 78, 72, 40, 8, 90), + genders=("FEMALE", "MALE", "MALE", "FEMALE", "MALE", "FEMALE", "FEMALE"), + reported=(10_000.0, 9_000.0, 14_000.0, 13_500.0, 0.0, 0.0, 3_000.0), +) -> dict[str, pd.DataFrame]: + """Seven people in four households, observed in 2024. + + The pensioners cover both State Pension types, below and above the + flat-rate maximum (the excess is additional State Pension or a Protected + Payment), and one Pension Credit claimant. + """ + person = pd.DataFrame( + { + "person_id": [101, 102, 201, 202, 301, 302, 401], + "person_benunit_id": [11, 11, 21, 22, 31, 31, 41], + "person_household_id": [1, 1, 2, 2, 3, 3, 4], + "age": list(ages), + "gender": list(genders), + "state_pension_reported": list(reported), + "employment_income": [0.0, 0.0, 0.0, 0.0, 30_000.0, 0.0, 0.0], + "private_pension_income": [2_000.0, 0.0, 5_000.0, 0.0, 0.0, 0.0, 0.0], + } + ) + benunit = pd.DataFrame({"benunit_id": [11, 21, 22, 31, 41]}) + household = pd.DataFrame( + { + "household_id": [1, 2, 3, 4], + "household_weight": [1_000.0, 2_000.0, 1_500.0, 500.0], + "region": ["LONDON", "SCOTLAND", "WALES", "NORTH_EAST"], + "tenure_type": [ + "OWNED_OUTRIGHT", + "RENT_PRIVATELY", + "RENT_FROM_COUNCIL", + "RENT_PRIVATELY", + ], + "council_tax": [2_000.0, 1_500.0, 1_800.0, 1_200.0], + "rent": [0.0, 9_000.0, 6_000.0, 7_000.0], + } + ) + return {"person": person, "benunit": benunit, "household": household} + + +def _write_source(path, tables=None) -> str: + """Write ``tables`` as a policyengine-uk single-year (2024) dataset.""" + from policyengine_uk.data import UKSingleYearDataset + + tables = tables or _tables() + UKSingleYearDataset(**tables, fiscal_year=DATA_YEAR).save(str(path)) + return str(path) + + +def _cut(source: str, directory, years) -> dict[str, PolicyEngineUKDataset]: + return create_datasets( + datasets=[source], + years=list(years), + data_folder=str(directory), + allow_unmanaged=True, + ) + + +def _run(dataset: PolicyEngineUKDataset, **kwargs) -> Simulation: + simulation = Simulation( + dataset=dataset, + tax_benefit_model_version=pe.uk.model, + extra_variables=EXTRA_VARIABLES, + **kwargs, + ) + simulation.run() + return simulation + + +def _direct(source: str): + from policyengine_uk import Microsimulation + + return Microsimulation(dataset=source) + + +def _assert_matches_direct(simulation: Simulation, direct, year: int) -> None: + """Every requested output equals the direct run for the records kept.""" + output = simulation.output_dataset.data + for entity, variables in EXTRA_VARIABLES.items(): + id_column = f"{entity}_id" + frame = pd.DataFrame(getattr(output, entity)) + direct_ids = direct.calculate(id_column, year).values + positions = pd.Index(direct_ids).get_indexer(frame[id_column]) + assert (positions >= 0).all() + for variable in variables: + expected = np.asarray(direct.calculate(variable, year).values)[positions] + actual = frame[variable].to_numpy() + if expected.dtype.kind in "fiub": + np.testing.assert_allclose( + actual.astype(float), + expected.astype(float), + rtol=1e-6, + atol=1e-3, + err_msg=f"{entity}.{variable} in {year}", + ) + else: + assert list(actual.astype(str)) == list(expected.astype(str)), ( + f"{entity}.{variable} in {year}" + ) + + +@pytest.fixture(scope="module") +def source(tmp_path_factory) -> str: + return _write_source(tmp_path_factory.mktemp("source") / "tiny_uk_2024.h5") + + +@pytest.fixture(scope="module") +def year_files(source, tmp_path_factory) -> dict[str, PolicyEngineUKDataset]: + return _cut(source, tmp_path_factory.mktemp("data"), [2024, 2025, 2026, 2028]) + + +@pytest.fixture(scope="module") +def direct(source): + return _direct(source) + + +@pytest.mark.parametrize("year", [2024, 2025, 2026, 2028]) +def test_year_file_run_matches_direct_policyengine_uk_run(year_files, direct, year): + simulation = _run(year_files[f"tiny_uk_2024_{year}"]) + + _assert_matches_direct(simulation, direct, year) + + +def test_year_file_alone_would_not_match(year_files, direct): + """The fixture exercises the fix: the 2026 tables as observed data differ. + + Passed to policyengine-uk on its own, the projected 2026 frame makes 2026 + the observed year, and the State Pension follows the CPI uprating of its + reported amount. This is what the year file ran as before #556. + """ + from policyengine_uk import Microsimulation + from policyengine_uk.data import UKSingleYearDataset + + data = year_files["tiny_uk_2024_2026"].data + alone = Microsimulation( + dataset=UKSingleYearDataset( + person=pd.DataFrame(data.person).copy(), + benunit=pd.DataFrame(data.benunit).copy(), + household=pd.DataFrame(data.household).copy(), + fiscal_year=2026, + ) + ) + direct_pension = direct.calculate("state_pension", 2026).values + alone_pension = alone.calculate("state_pension", 2026).values + receives = direct_pension > 0 + assert receives.sum() == 5 + # Every recipient's State Pension differs, and is lower: CPI grew less + # than the State Pension rates between 2024 and 2026. + assert (alone_pension[receives] < direct_pension[receives] - 1).all() + + +def test_projected_year_files_keep_their_data_year(year_files): + projected = year_files["tiny_uk_2024_2026"] + observed = year_files["tiny_uk_2024_2024"] + + assert projected.data_year == DATA_YEAR + assert projected.data_year_data is not None + assert observed.data_year == DATA_YEAR + # The data year's own file holds its tables as `data`. + assert observed.data_year_data is None + # The data year's tables are the source's, untouched by the projection. + pd.testing.assert_series_equal( + pd.DataFrame(projected.data_year_data.person)["state_pension_reported"], + pd.DataFrame(observed.data.person)["state_pension_reported"], + ) + # The projected year's reported amount is uprated (by CPI). + assert ( + pd.DataFrame(projected.data.person)["state_pension_reported"] + > pd.DataFrame(observed.data.person)["state_pension_reported"] + ).sum() == 5 + + +def test_year_file_round_trips_its_data_year(year_files): + saved = year_files["tiny_uk_2024_2026"] + + loaded = PolicyEngineUKDataset( + name="reloaded", + description="reloaded", + filepath=saved.filepath, + year=2026, + ) + + assert loaded.data_year == DATA_YEAR + for entity in ("person", "benunit", "household"): + pd.testing.assert_frame_equal( + pd.DataFrame(getattr(loaded.data_year_data, entity)), + pd.DataFrame(getattr(saved.data_year_data, entity)), + check_categorical=False, + check_dtype=False, + ) + pd.testing.assert_frame_equal( + pd.DataFrame(getattr(loaded.data, entity)), + pd.DataFrame(getattr(saved.data, entity)), + check_categorical=False, + check_dtype=False, + ) + + +def test_scoped_run_matches_direct_run_for_the_records_kept(year_files, direct): + simulation = _run( + year_files["tiny_uk_2024_2026"], + scoping_strategy=RowFilterStrategy( + variable_name="region", variable_value="SCOTLAND" + ), + ) + + kept = pd.DataFrame(simulation.output_dataset.data.person)["person_id"] + assert sorted(kept) == [201, 202] + _assert_matches_direct(simulation, direct, 2026) + + +def test_run_leaves_the_callers_tables_unchanged(year_files): + dataset = year_files["tiny_uk_2024_2026"] + before = { + name: pd.DataFrame(frame).copy() + for name, frame in dataset.data.entity_data.items() + } + observed_before = { + name: pd.DataFrame(frame).copy() + for name, frame in dataset.data_year_data.entity_data.items() + } + + _run(dataset) + # policyengine-uk encodes enum columns in place on the tables it is given. + _run( + dataset, + scoping_strategy=RowFilterStrategy( + variable_name="region", variable_value="WALES" + ), + ) + + for name, frame in dataset.data.entity_data.items(): + pd.testing.assert_frame_equal(pd.DataFrame(frame), before[name]) + for name, frame in dataset.data_year_data.entity_data.items(): + pd.testing.assert_frame_equal(pd.DataFrame(frame), observed_before[name]) + + +def test_dataset_built_in_memory_runs_as_observed_data(source): + """A dataset without a data year is observed data for its own year. + + That is how policyengine-uk treats a single-year dataset, so the run + equals policyengine-uk on the same tables. + """ + from policyengine_uk import Microsimulation + from policyengine_uk.data import UKSingleYearDataset + + tables = _tables() + weighted = _with_weights(tables) + dataset = PolicyEngineUKDataset( + name="observed-2026", + description="tables taken as observed in 2026", + year=2026, + data=weighted, + ) + simulation = _run(dataset) + direct = Microsimulation( + dataset=UKSingleYearDataset( + **{name: frame.copy() for name, frame in tables.items()}, + fiscal_year=2026, + ) + ) + + _assert_matches_direct(simulation, direct, 2026) + + +def _with_weights(tables: dict[str, pd.DataFrame]) -> UKYearData: + household = tables["household"] + weight = dict(zip(household["household_id"], household["household_weight"])) + person = tables["person"].assign( + person_weight=tables["person"]["person_household_id"].map(weight) + ) + benunit_household = person.drop_duplicates("person_benunit_id").set_index( + "person_benunit_id" + )["person_household_id"] + benunit = tables["benunit"].assign( + benunit_weight=tables["benunit"]["benunit_id"] + .map(benunit_household) + .map(weight) + ) + return UKYearData( + person=MicroDataFrame(person, weights="person_weight"), + benunit=MicroDataFrame(benunit, weights="benunit_weight"), + household=MicroDataFrame(household, weights="household_weight"), + ) + + +def test_projected_dataset_needs_its_data_year_tables(year_files): + projected = year_files["tiny_uk_2024_2026"] + + with pytest.raises(ValueError, match="no data_year_data"): + PolicyEngineUKDataset( + name="x", + description="x", + year=2026, + data=projected.data, + data_year=DATA_YEAR, + ) + with pytest.raises(ValueError, match="after its year"): + PolicyEngineUKDataset( + name="x", + description="x", + year=2023, + data=projected.data, + data_year=DATA_YEAR, + data_year_data=projected.data_year_data, + ) + with pytest.raises(ValueError, match="no data_year"): + PolicyEngineUKDataset( + name="x", + description="x", + year=2026, + data=projected.data, + data_year_data=projected.data_year_data, + ) + + +def test_records_missing_from_the_data_year_are_refused(year_files): + projected = year_files["tiny_uk_2024_2026"] + observed = projected.data_year_data + dropped = UKYearData( + person=MicroDataFrame( + pd.DataFrame(observed.person).iloc[1:], weights="person_weight" + ), + benunit=observed.benunit, + household=observed.household, + ) + dataset = PolicyEngineUKDataset( + name="x", + description="x", + year=2026, + data=projected.data, + data_year=DATA_YEAR, + data_year_data=dropped, + ) + + with pytest.raises(ValueError, match="1 person record"): + _policyengine_uk_input(dataset) + + +def _write_legacy_year_file(dataset: PolicyEngineUKDataset) -> None: + """Rewrite a year file as policyengine.py wrote it before #556.""" + legacy = PolicyEngineUKDataset( + name="legacy", + description="legacy", + filepath=dataset.filepath, + year=dataset.year, + data=dataset.data, + ) + legacy.save() + with pd.HDFStore(dataset.filepath, mode="r") as store: + assert f"/{DATA_YEAR_KEY}" not in store.keys() + + +def test_load_datasets_refuses_a_year_file_without_a_data_year(source, tmp_path): + created = _cut(source, tmp_path, [2026]) + _write_legacy_year_file(created["tiny_uk_2024_2026"]) + + with pytest.raises(ValueError, match="no recorded data year"): + load_datasets(datasets=[source], years=[2026], data_folder=str(tmp_path)) + + +def test_ensure_datasets_regenerates_a_year_file_without_a_data_year( + source, tmp_path, direct +): + created = _cut(source, tmp_path, [2026]) + _write_legacy_year_file(created["tiny_uk_2024_2026"]) + + ensured = ensure_datasets( + datasets=[source], + years=[2026], + data_folder=str(tmp_path), + allow_unmanaged=True, + ) + + dataset = ensured["tiny_uk_2024_2026"] + assert dataset.data_year == DATA_YEAR + reloaded = load_datasets(datasets=[source], years=[2026], data_folder=str(tmp_path)) + assert reloaded["tiny_uk_2024_2026"].data_year == DATA_YEAR + _assert_matches_direct(_run(dataset), direct, 2026) + + +@settings( + max_examples=6, + deadline=None, + suppress_health_check=[HealthCheck.function_scoped_fixture], +) +@given( + ages=st.lists(st.integers(min_value=60, max_value=100), min_size=7, max_size=7), + genders=st.lists(st.sampled_from(["MALE", "FEMALE"]), min_size=7, max_size=7), + reported=st.lists( + st.floats(min_value=0, max_value=25_000, allow_nan=False), + min_size=7, + max_size=7, + ), + year=st.integers(min_value=2025, max_value=2030), +) +def test_any_year_file_matches_the_direct_run( + tmp_path_factory, ages, genders, reported, year +): + """Differential invariant over random pensioners and projection years.""" + directory = tmp_path_factory.mktemp("random") + source = _write_source( + directory / "random_uk_2024.h5", + _tables(ages=ages, genders=genders, reported=reported), + ) + created = _cut(source, directory / "data", [year]) + + simulation = _run(created[f"random_uk_2024_{year}"]) + + _assert_matches_direct(simulation, _direct(source), year) From f384ec243f499ca0d08ac6c3e49dd2f0789d77d2 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Thu, 8 Oct 2026 10:40:46 -0400 Subject: [PATCH 2/6] docs: note the size of UK year files that keep their data year Co-Authored-By: Claude Opus 5.5 --- docs/microsim.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/docs/microsim.md b/docs/microsim.md index 9a64f053..62929cad 100644 --- a/docs/microsim.md +++ b/docs/microsim.md @@ -248,9 +248,11 @@ Each projected year file therefore keeps its data year, `dataset.data_year` forward as policyengine-uk does, and uses the file's own tables for the simulated year. A run of a year file then gives the same result, record by record, as `policyengine_uk.Microsimulation` on the certified file. Region -scoping applies to both sets of tables, matched by entity ID. A dataset built -in memory without a data year is observed data for its own year, which is how -policyengine-uk treats a single-year dataset. +scoping applies to both sets of tables, matched by entity ID. Keeping the data +year's tables doubles a projected file: the Enhanced FRS 2026 file is 226 MB, +against 113 MB for its own tables. A dataset built in memory without a data +year is observed data for its own year, which is how policyengine-uk treats a +single-year dataset. Year files written by earlier releases have no recorded data year. `ensure_datasets` writes them again and `load_datasets` refuses them. A year From 875924cac9caae00b8ef4ee3c886be34009f916e Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Thu, 8 Oct 2026 11:13:03 -0400 Subject: [PATCH 3/6] Skip the data-year read when create_datasets cuts no years create_datasets(years=[]) only materializes and loads the source; reading the data year there is wasted work and broke the runtime tests that mock policyengine-uk's Microsimulation. Co-Authored-By: Claude Opus 5.5 --- src/policyengine/tax_benefit_models/uk/datasets.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/policyengine/tax_benefit_models/uk/datasets.py b/src/policyengine/tax_benefit_models/uk/datasets.py index e7d79368..c8bc7e3b 100644 --- a/src/policyengine/tax_benefit_models/uk/datasets.py +++ b/src/policyengine/tax_benefit_models/uk/datasets.py @@ -278,6 +278,8 @@ def create_datasets( from policyengine_uk import Microsimulation sim = Microsimulation(dataset=source.path) + if not years: + continue # policyengine-uk takes a dataset's first year as observed data and # projects the rest from it (see PolicyEngineUKDataset.data_year). # Each year file keeps that year and its tables, so a run of the From 9e16ad72f9ebd5aa32b4d6a4ee48cd7e855c3ab8 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Thu, 8 Oct 2026 18:13:43 -0400 Subject: [PATCH 4/6] Narrow the scoped-run claim and say which path aliases tables Under row filtering, outputs normalised over the whole dataset (business-rates incidence through shareholding, deciles, relative poverty lines) become region-relative (#567), so the scoped test now checks person and benefit-unit outputs only. policyengine-uk encodes enum columns in place on a multi-year dataset's tables; it copies a single-year dataset when projecting it, so the comment names the path. Co-Authored-By: Claude Opus 5.5 --- .../tax_benefit_models/uk/model.py | 5 ++-- tests/test_uk_year_file_data_year.py | 25 ++++++++++++++----- 2 files changed, 22 insertions(+), 8 deletions(-) diff --git a/src/policyengine/tax_benefit_models/uk/model.py b/src/policyengine/tax_benefit_models/uk/model.py index 824935b5..fd23704b 100644 --- a/src/policyengine/tax_benefit_models/uk/model.py +++ b/src/policyengine/tax_benefit_models/uk/model.py @@ -288,8 +288,9 @@ def _policyengine_uk_input(dataset: PolicyEngineUKDataset): from policyengine_uk.data import UKMultiYearDataset, UKSingleYearDataset year = int(dataset.year) - # Copies: policyengine-uk encodes enum columns in place on the tables it - # is given, which would change the caller's dataset. + # Copies: policyengine-uk encodes enum columns in place on the tables of a + # multi-year dataset it is given (a single-year dataset is copied when + # policyengine-uk projects it), which would change the caller's dataset. simulated = UKSingleYearDataset( person=pd.DataFrame(dataset.data.person).copy(), benunit=pd.DataFrame(dataset.data.benunit).copy(), diff --git a/tests/test_uk_year_file_data_year.py b/tests/test_uk_year_file_data_year.py index abcfae80..7f668210 100644 --- a/tests/test_uk_year_file_data_year.py +++ b/tests/test_uk_year_file_data_year.py @@ -19,8 +19,12 @@ - Differential: for any source dataset and any year the source projects to, every output of a policyengine.py run of the year file equals the direct - policyengine-uk run of the source, record by record (also under region - scoping, for the records kept). + policyengine-uk run of the source, record by record. Under region scoping + this holds, for the records kept, for person and benefit-unit outputs. + Outputs normalised over the whole dataset, such as business-rates + incidence through ``shareholding``, deciles and relative poverty lines, + become region-relative when rows are filtered. That is a separate issue + (PolicyEngine/policyengine.py#567). - Round trip: saving and loading a year file preserves its data year and the data year's tables. - No aliasing: a run leaves the caller's dataset tables unchanged. @@ -151,10 +155,16 @@ def _direct(source: str): return Microsimulation(dataset=source) -def _assert_matches_direct(simulation: Simulation, direct, year: int) -> None: +def _assert_matches_direct( + simulation: Simulation, + direct, + year: int, + entities=("person", "benunit", "household"), +) -> None: """Every requested output equals the direct run for the records kept.""" output = simulation.output_dataset.data - for entity, variables in EXTRA_VARIABLES.items(): + for entity in entities: + variables = EXTRA_VARIABLES[entity] id_column = f"{entity}_id" frame = pd.DataFrame(getattr(output, entity)) direct_ids = direct.calculate(id_column, year).values @@ -284,7 +294,9 @@ def test_scoped_run_matches_direct_run_for_the_records_kept(year_files, direct): kept = pd.DataFrame(simulation.output_dataset.data.person)["person_id"] assert sorted(kept) == [201, 202] - _assert_matches_direct(simulation, direct, 2026) + # Household outputs can depend on dataset-wide normalisation, which row + # filtering changes (PolicyEngine/policyengine.py#567). + _assert_matches_direct(simulation, direct, 2026, entities=("person", "benunit")) def test_run_leaves_the_callers_tables_unchanged(year_files): @@ -299,7 +311,8 @@ def test_run_leaves_the_callers_tables_unchanged(year_files): } _run(dataset) - # policyengine-uk encodes enum columns in place on the tables it is given. + # policyengine-uk encodes enum columns in place on the tables of a + # multi-year dataset it is given, which is what a projected run hands it. _run( dataset, scoping_strategy=RowFilterStrategy( From 1629ee7b1d46334c03a8296ddf5310452f036375 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Thu, 8 Oct 2026 23:31:12 -0400 Subject: [PATCH 5/6] Address review: reforms, disk round trip, saved outputs, scoping wording - Test a State Pension reform against a direct run with the same compiled policy, and a year file loaded back from disk; both fail with run() reverted to the single-year construction. - Saved UK outputs record the data year their run anchored on; load() refuses outputs without it, so ensure() reruns pre-fix outputs. - Drop policyengine.py's person and benunit weight columns from the data year's tables before projecting, so the years in between match a direct run; order the projected years. - Write the data-year key last so it marks a complete file; let unreadable files raise; load each year file once. - State the scoping caveat (#567) in the docs and tests: deciles, the relative poverty median and shareholding are calculated over the simulated households. Co-Authored-By: Claude Opus 5.5 --- changelog.d/556.fixed.md | 2 +- docs/microsim.md | 23 +++- .../common/model_version.py | 15 ++ .../tax_benefit_models/uk/datasets.py | 22 +-- .../tax_benefit_models/uk/model.py | 18 ++- tests/test_uk_year_file_data_year.py | 130 ++++++++++++++++-- 6 files changed, 184 insertions(+), 26 deletions(-) diff --git a/changelog.d/556.fixed.md b/changelog.d/556.fixed.md index 34c48a6d..929e94dd 100644 --- a/changelog.d/556.fixed.md +++ b/changelog.d/556.fixed.md @@ -1 +1 @@ -UK population runs now anchor the State Pension on the survey year their data was projected from, as policyengine-uk does, so a run of a UK year file matches policyengine-uk on the certified dataset. Before, it took the projected year as the survey year and uprated the State Pension by CPI rather than the triple lock. Year files written by earlier releases are regenerated. +UK population runs now anchor the State Pension on the survey year their data was projected from, as policyengine-uk does, so a run of a UK year file matches policyengine-uk on the certified dataset, with or without a reform. Before, it took the projected year as the survey year: the State Pension was uprated by CPI rather than the triple lock, and State Pension rate reforms barely reached it. Projected year files now also store the survey year's tables, which doubles their size. `ensure_datasets` regenerates year files from earlier releases and `load_datasets` refuses them; `Simulation.load()` refuses UK outputs saved by earlier releases, and `Simulation.ensure()` runs them again. diff --git a/docs/microsim.md b/docs/microsim.md index 62929cad..f74272fe 100644 --- a/docs/microsim.md +++ b/docs/microsim.md @@ -247,17 +247,26 @@ Each projected year file therefore keeps its data year, `dataset.data_year` `dataset.data_year_data`. `Simulation.run()` projects the data year's tables forward as policyengine-uk does, and uses the file's own tables for the simulated year. A run of a year file then gives the same result, record by -record, as `policyengine_uk.Microsimulation` on the certified file. Region -scoping applies to both sets of tables, matched by entity ID. Keeping the data -year's tables doubles a projected file: the Enhanced FRS 2026 file is 226 MB, -against 113 MB for its own tables. A dataset built in memory without a data -year is observed data for its own year, which is how policyengine-uk treats a -single-year dataset. +record, as `policyengine_uk.Microsimulation` on the certified file, with or +without a reform. Keeping the data year's tables doubles a projected file: the +Enhanced FRS 2026 file is 226 MB, against 113 MB for its own tables. A dataset +built in memory without a data year is observed data for its own year, which +is how policyengine-uk treats a single-year dataset. + +Region scoping applies to both sets of tables, matched by entity ID. A +row-filtered run is a simulation of the region's households alone, so +variables that policyengine-uk calculates over every household in the +simulation are calculated over the region: income deciles, the relative +poverty median, and `shareholding`, which spreads corporate taxes across +households (PolicyEngine/policyengine.py#567). Variables of a person or +benefit unit that do not depend on those match the national run. Year files written by earlier releases have no recorded data year. `ensure_datasets` writes them again and `load_datasets` refuses them. A year file opened directly, as `PolicyEngineUKDataset(filepath=...)`, is not -checked. +checked. A saved UK simulation output records the data year its run anchored +on; `Simulation.load()` refuses an output saved without one, and +`Simulation.ensure()` runs it again. ## Simulations diff --git a/src/policyengine/tax_benefit_models/common/model_version.py b/src/policyengine/tax_benefit_models/common/model_version.py index 576c0c0f..e10d3330 100644 --- a/src/policyengine/tax_benefit_models/common/model_version.py +++ b/src/policyengine/tax_benefit_models/common/model_version.py @@ -443,6 +443,21 @@ def load(self, simulation: Simulation) -> None: "stored inputs (it has no record of them); run it again" ) + if self.country_code == "uk": + from policyengine.tax_benefit_models.uk.datasets import ( + _year_file_records_data_year, + ) + + # UK outputs saved before runs anchored on the observed data year + # (PolicyEngine/policyengine.py#556) took a projected year as + # observed and uprated the State Pension by CPI, so they are not + # reused. ``Simulation.ensure()`` runs such a simulation again. + if not _year_file_records_data_year(Path(filepath)): + raise ValueError( + "Saved UK simulation predates anchoring on the observed " + "data year (it records none); run it again" + ) + simulation.output_dataset = self._dataset_class( id=simulation.id, name=simulation.dataset.name, diff --git a/src/policyengine/tax_benefit_models/uk/datasets.py b/src/policyengine/tax_benefit_models/uk/datasets.py index c8bc7e3b..dbe31dbc 100644 --- a/src/policyengine/tax_benefit_models/uk/datasets.py +++ b/src/policyengine/tax_benefit_models/uk/datasets.py @@ -109,10 +109,13 @@ def _check_data_year(self) -> None: "`data` holds the observed tables and data_year_data must be " "None." ) + # An output records the data year its run anchored on, without the + # tables. if ( self.data_year < self.year and self.data is not None and self.data_year_data is None + and not self.is_output_dataset ): raise ValueError( f"PolicyEngineUKDataset for {self.year} was projected from " @@ -139,10 +142,11 @@ def save(self) -> None: with pd.HDFStore(filepath, mode="w") as store: _put_year_data(store, self.data) - if self.data_year is not None: - store.put(DATA_YEAR_KEY, pd.Series([int(self.data_year)])) if self.data_year_data is not None: _put_year_data(store, self.data_year_data, DATA_YEAR_TABLE_PREFIX) + # Written last: its presence marks a complete file. + if self.data_year is not None: + store.put(DATA_YEAR_KEY, pd.Series([int(self.data_year)])) def load(self) -> None: """Load dataset from HDF5 file into this instance.""" @@ -251,12 +255,12 @@ def _year_data_with_weights(year_dataset) -> UKYearData: def _year_file_records_data_year(path: Path) -> bool: - """Return whether a UK year file records its data year (``DATA_YEAR_KEY``).""" - try: - with pd.HDFStore(path, mode="r") as store: - return f"/{DATA_YEAR_KEY}" in store.keys() - except OSError: - return False + """Return whether a UK file records its data year (``DATA_YEAR_KEY``). + + A missing or unreadable file raises rather than reading as unrecorded. + """ + with pd.HDFStore(path, mode="r") as store: + return f"/{DATA_YEAR_KEY}" in store.keys() def create_datasets( @@ -337,13 +341,13 @@ def load_datasets( "State Pension by CPI rather than the triple lock. " "Regenerate it with ensure_datasets() or create_datasets()." ) + # Constructing with a filepath loads the file. uk_dataset = PolicyEngineUKDataset( name=f"{dataset_stem}-year-{year}", description=f"UK Dataset for year {year} based on {dataset_stem}", filepath=filepath, year=int(year), ) - uk_dataset.load() dataset_key = f"{dataset_stem}_{year}" result[dataset_key] = uk_dataset diff --git a/src/policyengine/tax_benefit_models/uk/model.py b/src/policyengine/tax_benefit_models/uk/model.py index fd23704b..3ca213e6 100644 --- a/src/policyengine/tax_benefit_models/uk/model.py +++ b/src/policyengine/tax_benefit_models/uk/model.py @@ -259,6 +259,12 @@ def run(self, simulation: "Simulation") -> "Simulation": filepath=str(_output_dataset_filepath(simulation)), year=simulation.dataset.year, is_output_dataset=True, + # The observed year the run anchored on. Saved with the output, + # it also marks the output as calculated with that anchoring + # (see ``load()`` in common/model_version.py). + data_year=( + dataset.data_year if dataset.data_year is not None else dataset.year + ), data=UKYearData( person=data["person"], benunit=data["benunit"], @@ -312,6 +318,14 @@ def _policyengine_uk_input(dataset: PolicyEngineUKDataset): data_year=data_year, year=year, ) + # Person and benefit-unit weights are policyengine.py's additions to the + # tables; a direct run calculates them from household weights. Kept as + # inputs, the projection would carry them unuprated into the years + # between the data year and the simulated year. + for entity, column in (("person", "person_weight"), ("benunit", "benunit_weight")): + observed_tables[entity] = observed_tables[entity].drop( + columns=[column], errors="ignore" + ) observed = UKSingleYearDataset(**observed_tables, fiscal_year=data_year) projected = extend_single_year_dataset(observed, system.parameters) if year not in projected.years: @@ -319,7 +333,9 @@ def _policyengine_uk_input(dataset: PolicyEngineUKDataset): observed, system.parameters, end_year=year ) return UKMultiYearDataset( - datasets=[projected[y] for y in projected.years if y != year] + [simulated] + datasets=[ + simulated if y == year else projected[y] for y in sorted(projected.years) + ] ) diff --git a/tests/test_uk_year_file_data_year.py b/tests/test_uk_year_file_data_year.py index 7f668210..340d78b9 100644 --- a/tests/test_uk_year_file_data_year.py +++ b/tests/test_uk_year_file_data_year.py @@ -19,15 +19,21 @@ - Differential: for any source dataset and any year the source projects to, every output of a policyengine.py run of the year file equals the direct - policyengine-uk run of the source, record by record. Under region scoping - this holds, for the records kept, for person and benefit-unit outputs. - Outputs normalised over the whole dataset, such as business-rates - incidence through ``shareholding``, deciles and relative poverty lines, - become region-relative when rows are filtered. That is a separate issue + policyengine-uk run of the source, record by record. This holds with or + without a reform, applied the same way to both, and for a year file loaded + back from disk. Under row filtering (``RowFilterStrategy``) it holds, for + the records kept, for person and benefit-unit variables that do not depend + on dataset-wide normalisation. Variables normalised over the whole dataset, + such as business-rates incidence through ``shareholding``, deciles, and + relative poverty lines (including household flags mapped to people), + become region-relative under scoping. That is a separate issue (PolicyEngine/policyengine.py#567). - Round trip: saving and loading a year file preserves its data year and the data year's tables. - No aliasing: a run leaves the caller's dataset tables unchanged. +- Saved outputs: a saved UK output records the data year its run anchored + on, and an output saved without one is refused, so ``Simulation.ensure()`` + runs it again. """ from __future__ import annotations @@ -55,6 +61,9 @@ from policyengine.tax_benefit_models.uk.model import ( # noqa: E402 _policyengine_uk_input, ) +from policyengine.utils.parametric_reforms import ( # noqa: E402 + simulation_modifier_from_parameter_values, +) DATA_YEAR = 2024 STATE_PENSION = [ @@ -463,11 +472,116 @@ def test_ensure_datasets_regenerates_a_year_file_without_a_data_year( allow_unmanaged=True, ) - dataset = ensured["tiny_uk_2024_2026"] - assert dataset.data_year == DATA_YEAR + assert ensured["tiny_uk_2024_2026"].data_year == DATA_YEAR reloaded = load_datasets(datasets=[source], years=[2026], data_folder=str(tmp_path)) assert reloaded["tiny_uk_2024_2026"].data_year == DATA_YEAR - _assert_matches_direct(_run(dataset), direct, 2026) + _assert_matches_direct(_run(reloaded["tiny_uk_2024_2026"]), direct, 2026) + + +def test_year_file_loaded_from_disk_matches_direct_run(source, tmp_path, direct): + """The production path: a second ensure_datasets() loads the files. + + Both tables come back through the HDF5 round trip, which stores object + columns as categoricals (depending on the pandas version). + """ + _cut(source, tmp_path, [2025, 2026]) + + loaded = load_datasets( + datasets=[source], years=[2025, 2026], data_folder=str(tmp_path) + ) + + for year in (2025, 2026): + dataset = loaded[f"tiny_uk_2024_{year}"] + assert dataset.data_year == DATA_YEAR + assert dataset.data_year_data is not None + _assert_matches_direct(_run(dataset), direct, year) + + +REFORM = { + "gov.dwp.state_pension.new_state_pension.amount": 260.0, + "gov.dwp.state_pension.basic_state_pension.amount": 200.0, +} + + +def test_reformed_year_file_run_matches_direct_run_with_the_same_reform( + year_files, source, direct +): + """A State Pension reform reaches the year-file run as it does the direct. + + policyengine.py applies a policy from the start of the simulated year. + Before #556 the simulated year was also the data year, so the reform set + the rate both numerator and denominator are taken from. With the data + year anchored on the survey year, the reform scales the State Pension. + """ + simulation = _run(year_files["tiny_uk_2024_2026"], policy=REFORM) + reformed = _direct(source) + simulation_modifier_from_parameter_values(simulation.policy.parameter_values)( + reformed + ) + + _assert_matches_direct(simulation, reformed, 2026) + # The reform raises every recipient's State Pension above the baseline. + baseline = direct.calculate("state_pension", 2026).values + raised = reformed.calculate("state_pension", 2026).values + assert (raised[baseline > 0] > baseline[baseline > 0] + 1).all() + + +def test_projected_input_is_ordered_and_leaves_weights_to_the_model(year_files): + dataset = year_files["tiny_uk_2024_2026"] + + built = _policyengine_uk_input(dataset) + + assert built.years == [2024, 2025, 2026, 2027, 2028, 2029, 2030] + for year in built.years: + if year == 2026: + continue + # A direct run calculates person and benefit-unit weights. + assert "person_weight" not in built[year].person.columns + assert "benunit_weight" not in built[year].benunit.columns + assert "person_weight" in built[2026].person.columns + + +def test_saved_output_records_its_data_year_and_old_outputs_are_rerun(year_files): + import uuid + + from policyengine.tax_benefit_models.common.model_version import ( + output_dataset_filepath, + ) + + dataset = year_files["tiny_uk_2024_2026"] + run_id = f"uk-data-year-{uuid.uuid4().hex}" + first = Simulation( + id=run_id, dataset=dataset, tax_benefit_model_version=pe.uk.model + ) + first.run() + first.save() + + reloaded = Simulation( + id=run_id, dataset=dataset, tax_benefit_model_version=pe.uk.model + ) + reloaded.load() + assert reloaded.output_dataset.data_year == DATA_YEAR + + # An output saved before #556 records no data year. + path = output_dataset_filepath(first) + PolicyEngineUKDataset( + name="legacy-output", + description="legacy-output", + filepath=str(path), + year=2026, + is_output_dataset=True, + data=first.output_dataset.data, + ).save() + legacy = Simulation( + id=run_id, dataset=dataset, tax_benefit_model_version=pe.uk.model + ) + with pytest.raises(ValueError, match="records none"): + legacy.load() + + legacy.ensure() + assert legacy.output_dataset.data_year == DATA_YEAR + with pd.HDFStore(path, mode="r") as store: + assert f"/{DATA_YEAR_KEY}" in store.keys() @settings( From efaeb13a3f9f8f03b4951905bde2d75ba6e7bd8d Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 9 Oct 2026 01:30:07 -0400 Subject: [PATCH 6/6] Address round-2 review nits - Refuse a dataset that records a data year without its tables (a simulation output) with a clear error instead of an AttributeError. - Docs: row filtering applies to both sets of tables; weight replacement changes the simulated year's weights only. - Changelog: reforms had no effect under policyengine-uk 2.102.3 and were understated under 2.123. Co-Authored-By: Claude Opus 5.5 --- changelog.d/556.fixed.md | 2 +- docs/microsim.md | 6 ++++-- src/policyengine/tax_benefit_models/uk/model.py | 7 +++++++ tests/test_uk_year_file_data_year.py | 9 +++++++++ 4 files changed, 21 insertions(+), 3 deletions(-) diff --git a/changelog.d/556.fixed.md b/changelog.d/556.fixed.md index 929e94dd..f26bcd1f 100644 --- a/changelog.d/556.fixed.md +++ b/changelog.d/556.fixed.md @@ -1 +1 @@ -UK population runs now anchor the State Pension on the survey year their data was projected from, as policyengine-uk does, so a run of a UK year file matches policyengine-uk on the certified dataset, with or without a reform. Before, it took the projected year as the survey year: the State Pension was uprated by CPI rather than the triple lock, and State Pension rate reforms barely reached it. Projected year files now also store the survey year's tables, which doubles their size. `ensure_datasets` regenerates year files from earlier releases and `load_datasets` refuses them; `Simulation.load()` refuses UK outputs saved by earlier releases, and `Simulation.ensure()` runs them again. +UK population runs now anchor the State Pension on the survey year their data was projected from, as policyengine-uk does, so a run of a UK year file matches policyengine-uk on the certified dataset, with or without a reform. Before, it took the projected year as the survey year: the State Pension was uprated by CPI rather than the triple lock, and State Pension rate reforms had no effect on it under policyengine-uk 2.102.3 and were understated under 2.123. Projected year files now also store the survey year's tables, which doubles their size. `ensure_datasets` regenerates year files from earlier releases and `load_datasets` refuses them; `Simulation.load()` refuses UK outputs saved by earlier releases, and `Simulation.ensure()` runs them again. diff --git a/docs/microsim.md b/docs/microsim.md index f74272fe..3e1b2da5 100644 --- a/docs/microsim.md +++ b/docs/microsim.md @@ -253,8 +253,10 @@ Enhanced FRS 2026 file is 226 MB, against 113 MB for its own tables. A dataset built in memory without a data year is observed data for its own year, which is how policyengine-uk treats a single-year dataset. -Region scoping applies to both sets of tables, matched by entity ID. A -row-filtered run is a simulation of the region's households alone, so +Row filtering applies to both sets of tables, matched by entity ID. Weight +replacement changes the simulated year's weights only; the data year and the +years between keep national weights. A row-filtered run is a simulation of the +region's households alone, so variables that policyengine-uk calculates over every household in the simulation are calculated over the region: income deciles, the relative poverty median, and `shareholding`, which spreads corporate taxes across diff --git a/src/policyengine/tax_benefit_models/uk/model.py b/src/policyengine/tax_benefit_models/uk/model.py index 3ca213e6..02dfa7c1 100644 --- a/src/policyengine/tax_benefit_models/uk/model.py +++ b/src/policyengine/tax_benefit_models/uk/model.py @@ -312,6 +312,13 @@ def _policyengine_uk_input(dataset: PolicyEngineUKDataset): from policyengine_uk.system import system data_year = int(dataset.data_year) + if dataset.data_year_data is None: + raise ValueError( + f"Dataset {dataset.id} for {year} records data year {data_year} " + "but not its tables, as a simulation output does, so it cannot be " + "simulated. Run the year file it came from, or set data_year=None " + f"to treat its tables as observed data for {year}." + ) observed_tables = _match_records( dataset.data_year_data.entity_data, dataset.data.entity_data, diff --git a/tests/test_uk_year_file_data_year.py b/tests/test_uk_year_file_data_year.py index 340d78b9..522bd9c5 100644 --- a/tests/test_uk_year_file_data_year.py +++ b/tests/test_uk_year_file_data_year.py @@ -437,6 +437,15 @@ def test_records_missing_from_the_data_year_are_refused(year_files): _policyengine_uk_input(dataset) +def test_an_output_dataset_cannot_be_simulated(year_files): + output = _run(year_files["tiny_uk_2024_2026"]).output_dataset + assert output.data_year == DATA_YEAR + assert output.data_year_data is None + + with pytest.raises(ValueError, match="cannot be simulated"): + _policyengine_uk_input(output) + + def _write_legacy_year_file(dataset: PolicyEngineUKDataset) -> None: """Rewrite a year file as policyengine.py wrote it before #556.""" legacy = PolicyEngineUKDataset(