diff --git a/changelog.d/556.fixed.md b/changelog.d/556.fixed.md new file mode 100644 index 00000000..f26bcd1f --- /dev/null +++ b/changelog.d/556.fixed.md @@ -0,0 +1 @@ +UK population runs now anchor the State Pension on the survey year their data was projected from, as policyengine-uk does, so a run of a UK year file matches policyengine-uk on the certified dataset, with or without a reform. Before, it took the projected year as the survey year: the State Pension was uprated by CPI rather than the triple lock, and State Pension rate reforms had no effect on it under policyengine-uk 2.102.3 and were understated under 2.123. Projected year files now also store the survey year's tables, which doubles their size. `ensure_datasets` regenerates year files from earlier releases and `load_datasets` refuses them; `Simulation.load()` refuses UK outputs saved by earlier releases, and `Simulation.ensure()` runs them again. diff --git a/docs/microsim.md b/docs/microsim.md index 98cdc7bc..3e1b2da5 100644 --- a/docs/microsim.md +++ b/docs/microsim.md @@ -231,6 +231,45 @@ not need repository-specific download logic. Authentication or authorization failures are reported directly and do not cause a retry against another repository type. +### UK year files and the data year + +`ensure_datasets` and `create_datasets` cut one file per requested year from +the projection policyengine-uk makes of the certified dataset. policyengine-uk +takes the first year of a dataset as observed data. Its State Pension formulas +split each person's reported State Pension against that year's legislated +rates, and scale the share to the simulated year's rates, which follow the +triple lock. The projection uprates the reported amount by CPI, so handing a +projected year's tables to policyengine-uk as observed data would make the +State Pension follow CPI instead. + +Each projected year file therefore keeps its data year, `dataset.data_year` +(2024 for Enhanced FRS 2024–25), and that year's tables, +`dataset.data_year_data`. `Simulation.run()` projects the data year's tables +forward as policyengine-uk does, and uses the file's own tables for the +simulated year. A run of a year file then gives the same result, record by +record, as `policyengine_uk.Microsimulation` on the certified file, with or +without a reform. Keeping the data year's tables doubles a projected file: the +Enhanced FRS 2026 file is 226 MB, against 113 MB for its own tables. A dataset +built in memory without a data year is observed data for its own year, which +is how policyengine-uk treats a single-year dataset. + +Row filtering applies to both sets of tables, matched by entity ID. Weight +replacement changes the simulated year's weights only; the data year and the +years between keep national weights. A row-filtered run is a simulation of the +region's households alone, so +variables that policyengine-uk calculates over every household in the +simulation are calculated over the region: income deciles, the relative +poverty median, and `shareholding`, which spreads corporate taxes across +households (PolicyEngine/policyengine.py#567). Variables of a person or +benefit unit that do not depend on those match the national run. + +Year files written by earlier releases have no recorded data year. +`ensure_datasets` writes them again and `load_datasets` refuses them. A year +file opened directly, as `PolicyEngineUKDataset(filepath=...)`, is not +checked. A saved UK simulation output records the data year its run anchored +on; `Simulation.load()` refuses an output saved without one, and +`Simulation.ensure()` runs it again. + ## Simulations A `Simulation` needs a dataset, a tax-benefit model version, and optionally a policy (reform): diff --git a/src/policyengine/tax_benefit_models/common/model_version.py b/src/policyengine/tax_benefit_models/common/model_version.py index 576c0c0f..e10d3330 100644 --- a/src/policyengine/tax_benefit_models/common/model_version.py +++ b/src/policyengine/tax_benefit_models/common/model_version.py @@ -443,6 +443,21 @@ def load(self, simulation: Simulation) -> None: "stored inputs (it has no record of them); run it again" ) + if self.country_code == "uk": + from policyengine.tax_benefit_models.uk.datasets import ( + _year_file_records_data_year, + ) + + # UK outputs saved before runs anchored on the observed data year + # (PolicyEngine/policyengine.py#556) took a projected year as + # observed and uprated the State Pension by CPI, so they are not + # reused. ``Simulation.ensure()`` runs such a simulation again. + if not _year_file_records_data_year(Path(filepath)): + raise ValueError( + "Saved UK simulation predates anchoring on the observed " + "data year (it records none); run it again" + ) + simulation.output_dataset = self._dataset_class( id=simulation.id, name=simulation.dataset.name, diff --git a/src/policyengine/tax_benefit_models/uk/datasets.py b/src/policyengine/tax_benefit_models/uk/datasets.py index 105dfdd0..dbe31dbc 100644 --- a/src/policyengine/tax_benefit_models/uk/datasets.py +++ b/src/policyengine/tax_benefit_models/uk/datasets.py @@ -15,6 +15,19 @@ resolve_dataset_reference, ) +#: HDF5 key holding the data year of a UK year file: the year of the observed +#: data the file's tables were projected from (see +#: ``PolicyEngineUKDataset.data_year``). ``create_datasets`` writes it into +#: every year file. Year files written before policyengine.py kept the data +#: year have none: policyengine-uk would take their projected tables as +#: observed data, so ``ensure_datasets`` regenerates them and +#: ``load_datasets`` refuses them. +DATA_YEAR_KEY = "data_year" + +#: Prefix of the HDF5 keys holding the data year's entity tables in a year +#: file whose year comes after its data year. +DATA_YEAR_TABLE_PREFIX = "data_year_" + class UKYearData(YearData): """Entity-level data for a single year.""" @@ -40,6 +53,27 @@ class PolicyEngineUKDataset(Dataset): data: Optional[UKYearData] = None + data_year: Optional[int] = None + """Year of the observed data that ``data`` was projected from. + + policyengine-uk takes the first year of a dataset as observed data. It + splits each person's reported State Pension against that year's + legislated rates and scales the share to the simulated year's rates, + which follow the triple lock. ``create_datasets`` records the source + dataset's first year here. ``None`` means ``data`` is itself observed + data for ``year``, as it is for a dataset built in memory. + """ + + data_year_data: Optional[UKYearData] = None + """Entity tables of ``data_year``, when it comes before ``year``. + + ``run()`` projects them forward as policyengine-uk does in a direct run, + and uses ``data`` for ``year`` itself. Without them, policyengine-uk + would take the projected ``data`` as observed data for ``year`` and the + State Pension would follow the CPI uprating of its reported amount + instead of the triple lock. + """ + def model_post_init(self, __context): """Called after Pydantic initialization. @@ -53,6 +87,43 @@ def model_post_init(self, __context): """ if self.data is None and self.filepath: self.load() + self._check_data_year() + + def _check_data_year(self) -> None: + """Refuse a data year that cannot be the source of ``year``.""" + if self.data_year is None: + if self.data_year_data is not None: + raise ValueError( + "PolicyEngineUKDataset has data-year tables but no " + "data_year. Set data_year to the year they hold." + ) + return + if self.data_year > self.year: + raise ValueError( + f"PolicyEngineUKDataset data_year {self.data_year} is after " + f"its year {self.year}; data can only be projected forward." + ) + if self.data_year == self.year and self.data_year_data is not None: + raise ValueError( + f"PolicyEngineUKDataset year {self.year} is its data year, so " + "`data` holds the observed tables and data_year_data must be " + "None." + ) + # An output records the data year its run anchored on, without the + # tables. + if ( + self.data_year < self.year + and self.data is not None + and self.data_year_data is None + and not self.is_output_dataset + ): + raise ValueError( + f"PolicyEngineUKDataset for {self.year} was projected from " + f"{self.data_year} but has no data_year_data, so " + "policyengine-uk cannot anchor the State Pension on the " + "observed year. Pass the data year's tables, or set " + "data_year=None to treat `data` as observed data." + ) def save(self) -> None: """Save dataset to HDF5 file. @@ -69,41 +140,31 @@ def save(self) -> None: if not filepath.parent.exists(): filepath.parent.mkdir(parents=True, exist_ok=True) - # Convert DataFrames and optimize object columns to categorical - person_df = pd.DataFrame(self.data.person) - benunit_df = pd.DataFrame(self.data.benunit) - household_df = pd.DataFrame(self.data.household) - - # Convert object columns to categorical to avoid pickle serialization - for col in person_df.columns: - if person_df[col].dtype == "object": - person_df[col] = person_df[col].astype("category") - - for col in benunit_df.columns: - if benunit_df[col].dtype == "object": - benunit_df[col] = benunit_df[col].astype("category") - - for col in household_df.columns: - if household_df[col].dtype == "object": - household_df[col] = household_df[col].astype("category") - with pd.HDFStore(filepath, mode="w") as store: - # Use format='table' to support categorical dtypes - store.put("person", person_df, format="table") - store.put("benunit", benunit_df, format="table") - store.put("household", household_df, format="table") + _put_year_data(store, self.data) + if self.data_year_data is not None: + _put_year_data(store, self.data_year_data, DATA_YEAR_TABLE_PREFIX) + # Written last: its presence marks a complete file. + if self.data_year is not None: + store.put(DATA_YEAR_KEY, pd.Series([int(self.data_year)])) def load(self) -> None: """Load dataset from HDF5 file into this instance.""" filepath = self.filepath with pd.HDFStore(filepath, mode="r") as store: - self.data = UKYearData( - person=MicroDataFrame(store["person"], weights="person_weight"), - benunit=MicroDataFrame(store["benunit"], weights="benunit_weight"), - household=MicroDataFrame( - store["household"], weights="household_weight" - ), + keys = set(store.keys()) + self.data = _read_year_data(store) + self.data_year = ( + int(store[DATA_YEAR_KEY].iloc[0]) + if f"/{DATA_YEAR_KEY}" in keys + else None + ) + self.data_year_data = ( + _read_year_data(store, DATA_YEAR_TABLE_PREFIX) + if f"/{DATA_YEAR_TABLE_PREFIX}person" in keys + else None ) + self._check_data_year() def __repr__(self) -> str: if self.data is None: @@ -112,7 +173,94 @@ def __repr__(self) -> str: n_people = len(self.data.person) n_benunits = len(self.data.benunit) n_households = len(self.data.household) - return f"" + return f"" + + +def _put_year_data(store: pd.HDFStore, data: UKYearData, prefix: str = "") -> None: + """Write one year's entity tables to an open HDF5 store. + + Object columns are stored as categoricals to avoid slow pickle + serialization; ``format="table"`` supports categorical dtypes. + """ + for name, frame in data.entity_data.items(): + frame = pd.DataFrame(frame) + for column in frame.columns: + if frame[column].dtype == "object": + frame[column] = frame[column].astype("category") + store.put(f"{prefix}{name}", frame, format="table") + + +def _read_year_data(store: pd.HDFStore, prefix: str = "") -> UKYearData: + """Read one year's entity tables from an open HDF5 store.""" + return UKYearData( + person=MicroDataFrame(store[f"{prefix}person"], weights="person_weight"), + benunit=MicroDataFrame(store[f"{prefix}benunit"], weights="benunit_weight"), + household=MicroDataFrame( + store[f"{prefix}household"], weights="household_weight" + ), + ) + + +def _year_data_with_weights(year_dataset) -> UKYearData: + """Return a policyengine-uk year's tables with person and benunit weights. + + policyengine-uk stores household weights only; each person and benefit + unit takes its household's weight. + """ + # Convert to pandas DataFrames and add weight columns + person_df = pd.DataFrame(year_dataset.person) + benunit_df = pd.DataFrame(year_dataset.benunit) + household_df = pd.DataFrame(year_dataset.household) + + # Map household weights to person and benunit levels + person_df = person_df.merge( + household_df[["household_id", "household_weight"]], + left_on="person_household_id", + right_on="household_id", + how="left", + ) + person_df = person_df.rename(columns={"household_weight": "person_weight"}) + person_df = person_df.drop(columns=["household_id"]) + + # Get household_id for each benunit from person table + benunit_household_map = person_df[ + ["person_benunit_id", "person_household_id"] + ].drop_duplicates() + benunit_df = benunit_df.merge( + benunit_household_map, + left_on="benunit_id", + right_on="person_benunit_id", + how="left", + ) + benunit_df = benunit_df.merge( + household_df[["household_id", "household_weight"]], + left_on="person_household_id", + right_on="household_id", + how="left", + ) + benunit_df = benunit_df.rename(columns={"household_weight": "benunit_weight"}) + benunit_df = benunit_df.drop( + columns=[ + "person_benunit_id", + "person_household_id", + "household_id", + ], + errors="ignore", + ) + return UKYearData( + person=MicroDataFrame(person_df, weights="person_weight"), + benunit=MicroDataFrame(benunit_df, weights="benunit_weight"), + household=MicroDataFrame(household_df, weights="household_weight"), + ) + + +def _year_file_records_data_year(path: Path) -> bool: + """Return whether a UK file records its data year (``DATA_YEAR_KEY``). + + A missing or unreadable file raises rather than reading as unrecorded. + """ + with pd.HDFStore(path, mode="r") as store: + return f"/{DATA_YEAR_KEY}" in store.keys() def create_datasets( @@ -134,63 +282,24 @@ def create_datasets( from policyengine_uk import Microsimulation sim = Microsimulation(dataset=source.path) + if not years: + continue + # policyengine-uk takes a dataset's first year as observed data and + # projects the rest from it (see PolicyEngineUKDataset.data_year). + # Each year file keeps that year and its tables, so a run of the + # file anchors on the same observed year as a run of the source. + data_year = int(min(sim.dataset.years)) + data_year_data = _year_data_with_weights(sim.dataset[data_year]) for year in years: - year_dataset = sim.dataset[year] - - # Convert to pandas DataFrames and add weight columns - person_df = pd.DataFrame(year_dataset.person) - benunit_df = pd.DataFrame(year_dataset.benunit) - household_df = pd.DataFrame(year_dataset.household) - - # Map household weights to person and benunit levels - person_df = person_df.merge( - household_df[["household_id", "household_weight"]], - left_on="person_household_id", - right_on="household_id", - how="left", - ) - person_df = person_df.rename(columns={"household_weight": "person_weight"}) - person_df = person_df.drop(columns=["household_id"]) - - # Get household_id for each benunit from person table - benunit_household_map = person_df[ - ["person_benunit_id", "person_household_id"] - ].drop_duplicates() - benunit_df = benunit_df.merge( - benunit_household_map, - left_on="benunit_id", - right_on="person_benunit_id", - how="left", - ) - benunit_df = benunit_df.merge( - household_df[["household_id", "household_weight"]], - left_on="person_household_id", - right_on="household_id", - how="left", - ) - benunit_df = benunit_df.rename( - columns={"household_weight": "benunit_weight"} - ) - benunit_df = benunit_df.drop( - columns=[ - "person_benunit_id", - "person_household_id", - "household_id", - ], - errors="ignore", - ) - uk_dataset = PolicyEngineUKDataset( id=f"{dataset_stem}_year_{year}", name=f"{dataset_stem}-year-{year}", description=f"UK Dataset for year {year} based on {dataset_stem}", filepath=f"{data_folder}/{dataset_stem}_year_{year}.h5", year=int(year), - data=UKYearData( - person=MicroDataFrame(person_df, weights="person_weight"), - benunit=MicroDataFrame(benunit_df, weights="benunit_weight"), - household=MicroDataFrame(household_df, weights="household_weight"), - ), + data=_year_data_with_weights(sim.dataset[year]), + data_year=data_year, + data_year_data=data_year_data if int(year) > data_year else None, ) uk_dataset.save() @@ -205,6 +314,14 @@ def load_datasets( years: list[int] = [2026, 2027, 2028, 2029, 2030], data_folder: str = "./data", ) -> dict[str, PolicyEngineUKDataset]: + """Load UK year files written by ``create_datasets``. + + A year file without a recorded data year (``DATA_YEAR_KEY``) was written + before policyengine.py kept the observed year its tables were projected + from. policyengine-uk would take those projected tables as observed data + and uprate the State Pension by CPI rather than the triple lock, so such + a file is refused. + """ if datasets is None: datasets = [get_release_manifest("uk").default_dataset] result = {} @@ -213,13 +330,24 @@ def load_datasets( dataset_stem = dataset_logical_name(resolved_dataset) for year in years: filepath = f"{data_folder}/{dataset_stem}_year_{year}.h5" + if Path(filepath).exists() and not _year_file_records_data_year( + Path(filepath) + ): + raise ValueError( + f"UK year file {filepath} has no recorded data year, so it " + "was written before policyengine.py kept the observed year " + "its tables were projected from. policyengine-uk would " + "take the projected tables as observed data and uprate the " + "State Pension by CPI rather than the triple lock. " + "Regenerate it with ensure_datasets() or create_datasets()." + ) + # Constructing with a filepath loads the file. uk_dataset = PolicyEngineUKDataset( name=f"{dataset_stem}-year-{year}", description=f"UK Dataset for year {year} based on {dataset_stem}", filepath=filepath, year=int(year), ) - uk_dataset.load() dataset_key = f"{dataset_stem}_{year}" result[dataset_key] = uk_dataset @@ -235,6 +363,10 @@ def ensure_datasets( ) -> dict[str, PolicyEngineUKDataset]: """Ensure datasets exist, loading if available or creating if not. + Year files without a recorded data year were written before + policyengine.py kept the observed year their tables were projected from + (see ``load_datasets``), so they are created again rather than loaded. + Args: datasets: List of HuggingFace dataset paths years: List of years to load/create data for @@ -253,7 +385,9 @@ def ensure_datasets( dataset_stem = dataset_logical_name(resolved_dataset) for year in years: filepath = Path(f"{data_folder}/{dataset_stem}_year_{year}.h5") - if not filepath.exists(): + # A year file without a recorded data year would run as observed + # data, so it is regenerated rather than reused. + if not filepath.exists() or not _year_file_records_data_year(filepath): all_exist = False break if not all_exist: diff --git a/src/policyengine/tax_benefit_models/uk/model.py b/src/policyengine/tax_benefit_models/uk/model.py index b0e22ad1..02dfa7c1 100644 --- a/src/policyengine/tax_benefit_models/uk/model.py +++ b/src/policyengine/tax_benefit_models/uk/model.py @@ -1,6 +1,7 @@ import datetime from typing import TYPE_CHECKING, Optional +import numpy as np import pandas as pd from microdf import MicroDataFrame @@ -22,6 +23,11 @@ from policyengine.core.simulation import Simulation UK_GROUP_ENTITIES = ["benunit", "household"] +UK_ENTITY_ID_COLUMNS = { + "person": "person_id", + "benunit": "benunit_id", + "household": "household_id", +} UK_HOUSEHOLD_PASSTHROUGH_COLUMNS = [ "oa_code", "lsoa_code", @@ -159,7 +165,6 @@ def _dataset_class(self): # --- run ------------------------------------------------------------- def run(self, simulation: "Simulation") -> "Simulation": from policyengine_uk import Microsimulation - from policyengine_uk.data import UKSingleYearDataset from policyengine.utils.parametric_reforms import ( simulation_modifier_from_parameter_values, @@ -197,15 +202,13 @@ def run(self, simulation: "Simulation") -> "Simulation": benunit=scoped_data["benunit"], household=scoped_data["household"], ), + # The data year's tables stay whole; they are matched to the + # scoped records by ID when the input is built. + data_year=dataset.data_year, + data_year_data=dataset.data_year_data, ) - input_data = UKSingleYearDataset( - person=dataset.data.person, - benunit=dataset.data.benunit, - household=dataset.data.household, - fiscal_year=dataset.year, - ) - microsim = Microsimulation(dataset=input_data) + microsim = Microsimulation(dataset=_policyengine_uk_input(dataset)) if simulation.policy and simulation.policy.simulation_modifier is not None: simulation.policy.simulation_modifier(microsim) @@ -256,6 +259,12 @@ def run(self, simulation: "Simulation") -> "Simulation": filepath=str(_output_dataset_filepath(simulation)), year=simulation.dataset.year, is_output_dataset=True, + # The observed year the run anchored on. Saved with the output, + # it also marks the output as calculated with that anchoring + # (see ``load()`` in common/model_version.py). + data_year=( + dataset.data_year if dataset.data_year is not None else dataset.year + ), data=UKYearData( person=data["person"], benunit=data["benunit"], @@ -264,6 +273,127 @@ def run(self, simulation: "Simulation") -> "Simulation": ) +def _policyengine_uk_input(dataset: PolicyEngineUKDataset): + """Build the policyengine-uk dataset a run of ``dataset`` simulates. + + policyengine-uk takes the first year of its dataset as observed data: + the State Pension formulas split each person's reported State Pension + against that year's legislated rates and scale the share to the + simulated year's rates. A dataset projected from an earlier observed + year (``data_year`` before ``year``) therefore gets the input a direct + policyengine-uk run on its source builds: the data year's tables + projected forward by policyengine-uk, with ``dataset.data`` as the + simulated year. Passing ``dataset.data`` alone would make the simulated + year the observed year, and the State Pension would follow the CPI + uprating of its reported amount instead of the triple lock + (PolicyEngine/policyengine.py#556). + + Any other dataset is passed as observed data for its year, as + policyengine-uk treats a single-year dataset. + """ + from policyengine_uk.data import UKMultiYearDataset, UKSingleYearDataset + + year = int(dataset.year) + # Copies: policyengine-uk encodes enum columns in place on the tables of a + # multi-year dataset it is given (a single-year dataset is copied when + # policyengine-uk projects it), which would change the caller's dataset. + simulated = UKSingleYearDataset( + person=pd.DataFrame(dataset.data.person).copy(), + benunit=pd.DataFrame(dataset.data.benunit).copy(), + household=pd.DataFrame(dataset.data.household).copy(), + fiscal_year=year, + ) + if dataset.data_year is None or int(dataset.data_year) >= year: + return simulated + + from policyengine_uk.data.economic_assumptions import ( + extend_single_year_dataset, + ) + from policyengine_uk.system import system + + data_year = int(dataset.data_year) + if dataset.data_year_data is None: + raise ValueError( + f"Dataset {dataset.id} for {year} records data year {data_year} " + "but not its tables, as a simulation output does, so it cannot be " + "simulated. Run the year file it came from, or set data_year=None " + f"to treat its tables as observed data for {year}." + ) + observed_tables = _match_records( + dataset.data_year_data.entity_data, + dataset.data.entity_data, + data_year=data_year, + year=year, + ) + # Person and benefit-unit weights are policyengine.py's additions to the + # tables; a direct run calculates them from household weights. Kept as + # inputs, the projection would carry them unuprated into the years + # between the data year and the simulated year. + for entity, column in (("person", "person_weight"), ("benunit", "benunit_weight")): + observed_tables[entity] = observed_tables[entity].drop( + columns=[column], errors="ignore" + ) + observed = UKSingleYearDataset(**observed_tables, fiscal_year=data_year) + projected = extend_single_year_dataset(observed, system.parameters) + if year not in projected.years: + projected = extend_single_year_dataset( + observed, system.parameters, end_year=year + ) + return UKMultiYearDataset( + datasets=[ + simulated if y == year else projected[y] for y in sorted(projected.years) + ] + ) + + +def _match_records( + observed: dict[str, pd.DataFrame], + simulated: dict[str, pd.DataFrame], + *, + data_year: int, + year: int, +) -> dict[str, pd.DataFrame]: + """Return the observed tables for the simulated records, in their order. + + policyengine-uk builds its entities from the first year of a dataset and + sets each year's inputs by position, so the observed year must hold the + same records in the same order as the simulated year. Records are + matched by ID, which carries region scoping (a subset of households) + over to the observed year. + """ + matched = {} + for entity, id_column in UK_ENTITY_ID_COLUMNS.items(): + table = pd.DataFrame(observed[entity]) + ids = pd.Index(table[id_column].to_numpy()) + if not ids.is_unique: + raise ValueError( + f"The {data_year} {entity} table repeats {id_column} values, " + "so its records cannot be matched to the simulated year." + ) + positions = ids.get_indexer(pd.DataFrame(simulated[entity])[id_column]) + missing = int((positions < 0).sum()) + if missing: + raise ValueError( + f"{missing} {entity} record(s) of the {year} tables are not in " + f"the {data_year} tables they were projected from, so " + "policyengine-uk has no observed data for them. Keep the data " + "year's tables in step with `data`, or set data_year=None to " + f"treat `data` as observed data for {year}." + ) + matched[entity] = table.iloc[positions].reset_index(drop=True).copy() + simulated_person = pd.DataFrame(simulated["person"]) + for link in ("person_benunit_id", "person_household_id"): + if not np.array_equal( + matched["person"][link].to_numpy(), simulated_person[link].to_numpy() + ): + raise ValueError( + f"People belong to different units ({link}) in the {data_year} " + f"and {year} tables, so the {year} tables were not projected " + f"from the {data_year} ones." + ) + return matched + + def managed_microsimulation( *, dataset: Optional[str] = None, diff --git a/tests/test_uk_year_file_data_year.py b/tests/test_uk_year_file_data_year.py new file mode 100644 index 00000000..522bd9c5 --- /dev/null +++ b/tests/test_uk_year_file_data_year.py @@ -0,0 +1,624 @@ +"""UK year files reproduce a direct policyengine-uk run of their source. + +policyengine-uk takes the first year of a dataset as observed data. Its State +Pension formulas split each person's reported State Pension against that +year's legislated rates and scale the share to the simulated year's rates, +which follow the triple lock. ``create_datasets`` cuts year files from the +projection policyengine-uk makes of the source, so each file keeps the +observed year and its tables, and ``run()`` rebuilds that projection. A year +file passed to policyengine-uk on its own would make the simulated year the +observed year, and the State Pension would follow the CPI uprating of its +reported amount instead (PolicyEngine/policyengine.py#556). + +Each test writes a tiny 2024 dataset in policyengine-uk's own file format, +cuts year files from it with ``create_datasets``, and compares a +policyengine.py run of a year file with ``policyengine_uk.Microsimulation`` +on the source file, record by record. + +Invariants exercised: + +- Differential: for any source dataset and any year the source projects to, + every output of a policyengine.py run of the year file equals the direct + policyengine-uk run of the source, record by record. This holds with or + without a reform, applied the same way to both, and for a year file loaded + back from disk. Under row filtering (``RowFilterStrategy``) it holds, for + the records kept, for person and benefit-unit variables that do not depend + on dataset-wide normalisation. Variables normalised over the whole dataset, + such as business-rates incidence through ``shareholding``, deciles, and + relative poverty lines (including household flags mapped to people), + become region-relative under scoping. That is a separate issue + (PolicyEngine/policyengine.py#567). +- Round trip: saving and loading a year file preserves its data year and + the data year's tables. +- No aliasing: a run leaves the caller's dataset tables unchanged. +- Saved outputs: a saved UK output records the data year its run anchored + on, and an output saved without one is refused, so ``Simulation.ensure()`` + runs it again. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest +from hypothesis import HealthCheck, given, settings +from hypothesis import strategies as st +from microdf import MicroDataFrame + +pytest.importorskip("policyengine_uk") + +import policyengine as pe # noqa: E402 +from policyengine.core import Simulation # noqa: E402 +from policyengine.core.scoping_strategy import RowFilterStrategy # noqa: E402 +from policyengine.tax_benefit_models.uk.datasets import ( # noqa: E402 + DATA_YEAR_KEY, + PolicyEngineUKDataset, + UKYearData, + create_datasets, + ensure_datasets, + load_datasets, +) +from policyengine.tax_benefit_models.uk.model import ( # noqa: E402 + _policyengine_uk_input, +) +from policyengine.utils.parametric_reforms import ( # noqa: E402 + simulation_modifier_from_parameter_values, +) + +DATA_YEAR = 2024 +STATE_PENSION = [ + "basic_state_pension", + "new_state_pension", + "additional_state_pension", + "state_pension", +] +PERSON_OUTPUTS = STATE_PENSION + ["state_pension_type", "income_tax"] +BENUNIT_OUTPUTS = ["pension_credit", "universal_credit"] +HOUSEHOLD_OUTPUTS = [ + "household_net_income", + "hbai_household_net_income", + "in_poverty_bhc", + "in_poverty_ahc", +] +EXTRA_VARIABLES = { + "person": PERSON_OUTPUTS, + "benunit": BENUNIT_OUTPUTS, + "household": HOUSEHOLD_OUTPUTS, +} + + +def _tables( + ages=(70, 85, 78, 72, 40, 8, 90), + genders=("FEMALE", "MALE", "MALE", "FEMALE", "MALE", "FEMALE", "FEMALE"), + reported=(10_000.0, 9_000.0, 14_000.0, 13_500.0, 0.0, 0.0, 3_000.0), +) -> dict[str, pd.DataFrame]: + """Seven people in four households, observed in 2024. + + The pensioners cover both State Pension types, below and above the + flat-rate maximum (the excess is additional State Pension or a Protected + Payment), and one Pension Credit claimant. + """ + person = pd.DataFrame( + { + "person_id": [101, 102, 201, 202, 301, 302, 401], + "person_benunit_id": [11, 11, 21, 22, 31, 31, 41], + "person_household_id": [1, 1, 2, 2, 3, 3, 4], + "age": list(ages), + "gender": list(genders), + "state_pension_reported": list(reported), + "employment_income": [0.0, 0.0, 0.0, 0.0, 30_000.0, 0.0, 0.0], + "private_pension_income": [2_000.0, 0.0, 5_000.0, 0.0, 0.0, 0.0, 0.0], + } + ) + benunit = pd.DataFrame({"benunit_id": [11, 21, 22, 31, 41]}) + household = pd.DataFrame( + { + "household_id": [1, 2, 3, 4], + "household_weight": [1_000.0, 2_000.0, 1_500.0, 500.0], + "region": ["LONDON", "SCOTLAND", "WALES", "NORTH_EAST"], + "tenure_type": [ + "OWNED_OUTRIGHT", + "RENT_PRIVATELY", + "RENT_FROM_COUNCIL", + "RENT_PRIVATELY", + ], + "council_tax": [2_000.0, 1_500.0, 1_800.0, 1_200.0], + "rent": [0.0, 9_000.0, 6_000.0, 7_000.0], + } + ) + return {"person": person, "benunit": benunit, "household": household} + + +def _write_source(path, tables=None) -> str: + """Write ``tables`` as a policyengine-uk single-year (2024) dataset.""" + from policyengine_uk.data import UKSingleYearDataset + + tables = tables or _tables() + UKSingleYearDataset(**tables, fiscal_year=DATA_YEAR).save(str(path)) + return str(path) + + +def _cut(source: str, directory, years) -> dict[str, PolicyEngineUKDataset]: + return create_datasets( + datasets=[source], + years=list(years), + data_folder=str(directory), + allow_unmanaged=True, + ) + + +def _run(dataset: PolicyEngineUKDataset, **kwargs) -> Simulation: + simulation = Simulation( + dataset=dataset, + tax_benefit_model_version=pe.uk.model, + extra_variables=EXTRA_VARIABLES, + **kwargs, + ) + simulation.run() + return simulation + + +def _direct(source: str): + from policyengine_uk import Microsimulation + + return Microsimulation(dataset=source) + + +def _assert_matches_direct( + simulation: Simulation, + direct, + year: int, + entities=("person", "benunit", "household"), +) -> None: + """Every requested output equals the direct run for the records kept.""" + output = simulation.output_dataset.data + for entity in entities: + variables = EXTRA_VARIABLES[entity] + id_column = f"{entity}_id" + frame = pd.DataFrame(getattr(output, entity)) + direct_ids = direct.calculate(id_column, year).values + positions = pd.Index(direct_ids).get_indexer(frame[id_column]) + assert (positions >= 0).all() + for variable in variables: + expected = np.asarray(direct.calculate(variable, year).values)[positions] + actual = frame[variable].to_numpy() + if expected.dtype.kind in "fiub": + np.testing.assert_allclose( + actual.astype(float), + expected.astype(float), + rtol=1e-6, + atol=1e-3, + err_msg=f"{entity}.{variable} in {year}", + ) + else: + assert list(actual.astype(str)) == list(expected.astype(str)), ( + f"{entity}.{variable} in {year}" + ) + + +@pytest.fixture(scope="module") +def source(tmp_path_factory) -> str: + return _write_source(tmp_path_factory.mktemp("source") / "tiny_uk_2024.h5") + + +@pytest.fixture(scope="module") +def year_files(source, tmp_path_factory) -> dict[str, PolicyEngineUKDataset]: + return _cut(source, tmp_path_factory.mktemp("data"), [2024, 2025, 2026, 2028]) + + +@pytest.fixture(scope="module") +def direct(source): + return _direct(source) + + +@pytest.mark.parametrize("year", [2024, 2025, 2026, 2028]) +def test_year_file_run_matches_direct_policyengine_uk_run(year_files, direct, year): + simulation = _run(year_files[f"tiny_uk_2024_{year}"]) + + _assert_matches_direct(simulation, direct, year) + + +def test_year_file_alone_would_not_match(year_files, direct): + """The fixture exercises the fix: the 2026 tables as observed data differ. + + Passed to policyengine-uk on its own, the projected 2026 frame makes 2026 + the observed year, and the State Pension follows the CPI uprating of its + reported amount. This is what the year file ran as before #556. + """ + from policyengine_uk import Microsimulation + from policyengine_uk.data import UKSingleYearDataset + + data = year_files["tiny_uk_2024_2026"].data + alone = Microsimulation( + dataset=UKSingleYearDataset( + person=pd.DataFrame(data.person).copy(), + benunit=pd.DataFrame(data.benunit).copy(), + household=pd.DataFrame(data.household).copy(), + fiscal_year=2026, + ) + ) + direct_pension = direct.calculate("state_pension", 2026).values + alone_pension = alone.calculate("state_pension", 2026).values + receives = direct_pension > 0 + assert receives.sum() == 5 + # Every recipient's State Pension differs, and is lower: CPI grew less + # than the State Pension rates between 2024 and 2026. + assert (alone_pension[receives] < direct_pension[receives] - 1).all() + + +def test_projected_year_files_keep_their_data_year(year_files): + projected = year_files["tiny_uk_2024_2026"] + observed = year_files["tiny_uk_2024_2024"] + + assert projected.data_year == DATA_YEAR + assert projected.data_year_data is not None + assert observed.data_year == DATA_YEAR + # The data year's own file holds its tables as `data`. + assert observed.data_year_data is None + # The data year's tables are the source's, untouched by the projection. + pd.testing.assert_series_equal( + pd.DataFrame(projected.data_year_data.person)["state_pension_reported"], + pd.DataFrame(observed.data.person)["state_pension_reported"], + ) + # The projected year's reported amount is uprated (by CPI). + assert ( + pd.DataFrame(projected.data.person)["state_pension_reported"] + > pd.DataFrame(observed.data.person)["state_pension_reported"] + ).sum() == 5 + + +def test_year_file_round_trips_its_data_year(year_files): + saved = year_files["tiny_uk_2024_2026"] + + loaded = PolicyEngineUKDataset( + name="reloaded", + description="reloaded", + filepath=saved.filepath, + year=2026, + ) + + assert loaded.data_year == DATA_YEAR + for entity in ("person", "benunit", "household"): + pd.testing.assert_frame_equal( + pd.DataFrame(getattr(loaded.data_year_data, entity)), + pd.DataFrame(getattr(saved.data_year_data, entity)), + check_categorical=False, + check_dtype=False, + ) + pd.testing.assert_frame_equal( + pd.DataFrame(getattr(loaded.data, entity)), + pd.DataFrame(getattr(saved.data, entity)), + check_categorical=False, + check_dtype=False, + ) + + +def test_scoped_run_matches_direct_run_for_the_records_kept(year_files, direct): + simulation = _run( + year_files["tiny_uk_2024_2026"], + scoping_strategy=RowFilterStrategy( + variable_name="region", variable_value="SCOTLAND" + ), + ) + + kept = pd.DataFrame(simulation.output_dataset.data.person)["person_id"] + assert sorted(kept) == [201, 202] + # Household outputs can depend on dataset-wide normalisation, which row + # filtering changes (PolicyEngine/policyengine.py#567). + _assert_matches_direct(simulation, direct, 2026, entities=("person", "benunit")) + + +def test_run_leaves_the_callers_tables_unchanged(year_files): + dataset = year_files["tiny_uk_2024_2026"] + before = { + name: pd.DataFrame(frame).copy() + for name, frame in dataset.data.entity_data.items() + } + observed_before = { + name: pd.DataFrame(frame).copy() + for name, frame in dataset.data_year_data.entity_data.items() + } + + _run(dataset) + # policyengine-uk encodes enum columns in place on the tables of a + # multi-year dataset it is given, which is what a projected run hands it. + _run( + dataset, + scoping_strategy=RowFilterStrategy( + variable_name="region", variable_value="WALES" + ), + ) + + for name, frame in dataset.data.entity_data.items(): + pd.testing.assert_frame_equal(pd.DataFrame(frame), before[name]) + for name, frame in dataset.data_year_data.entity_data.items(): + pd.testing.assert_frame_equal(pd.DataFrame(frame), observed_before[name]) + + +def test_dataset_built_in_memory_runs_as_observed_data(source): + """A dataset without a data year is observed data for its own year. + + That is how policyengine-uk treats a single-year dataset, so the run + equals policyengine-uk on the same tables. + """ + from policyengine_uk import Microsimulation + from policyengine_uk.data import UKSingleYearDataset + + tables = _tables() + weighted = _with_weights(tables) + dataset = PolicyEngineUKDataset( + name="observed-2026", + description="tables taken as observed in 2026", + year=2026, + data=weighted, + ) + simulation = _run(dataset) + direct = Microsimulation( + dataset=UKSingleYearDataset( + **{name: frame.copy() for name, frame in tables.items()}, + fiscal_year=2026, + ) + ) + + _assert_matches_direct(simulation, direct, 2026) + + +def _with_weights(tables: dict[str, pd.DataFrame]) -> UKYearData: + household = tables["household"] + weight = dict(zip(household["household_id"], household["household_weight"])) + person = tables["person"].assign( + person_weight=tables["person"]["person_household_id"].map(weight) + ) + benunit_household = person.drop_duplicates("person_benunit_id").set_index( + "person_benunit_id" + )["person_household_id"] + benunit = tables["benunit"].assign( + benunit_weight=tables["benunit"]["benunit_id"] + .map(benunit_household) + .map(weight) + ) + return UKYearData( + person=MicroDataFrame(person, weights="person_weight"), + benunit=MicroDataFrame(benunit, weights="benunit_weight"), + household=MicroDataFrame(household, weights="household_weight"), + ) + + +def test_projected_dataset_needs_its_data_year_tables(year_files): + projected = year_files["tiny_uk_2024_2026"] + + with pytest.raises(ValueError, match="no data_year_data"): + PolicyEngineUKDataset( + name="x", + description="x", + year=2026, + data=projected.data, + data_year=DATA_YEAR, + ) + with pytest.raises(ValueError, match="after its year"): + PolicyEngineUKDataset( + name="x", + description="x", + year=2023, + data=projected.data, + data_year=DATA_YEAR, + data_year_data=projected.data_year_data, + ) + with pytest.raises(ValueError, match="no data_year"): + PolicyEngineUKDataset( + name="x", + description="x", + year=2026, + data=projected.data, + data_year_data=projected.data_year_data, + ) + + +def test_records_missing_from_the_data_year_are_refused(year_files): + projected = year_files["tiny_uk_2024_2026"] + observed = projected.data_year_data + dropped = UKYearData( + person=MicroDataFrame( + pd.DataFrame(observed.person).iloc[1:], weights="person_weight" + ), + benunit=observed.benunit, + household=observed.household, + ) + dataset = PolicyEngineUKDataset( + name="x", + description="x", + year=2026, + data=projected.data, + data_year=DATA_YEAR, + data_year_data=dropped, + ) + + with pytest.raises(ValueError, match="1 person record"): + _policyengine_uk_input(dataset) + + +def test_an_output_dataset_cannot_be_simulated(year_files): + output = _run(year_files["tiny_uk_2024_2026"]).output_dataset + assert output.data_year == DATA_YEAR + assert output.data_year_data is None + + with pytest.raises(ValueError, match="cannot be simulated"): + _policyengine_uk_input(output) + + +def _write_legacy_year_file(dataset: PolicyEngineUKDataset) -> None: + """Rewrite a year file as policyengine.py wrote it before #556.""" + legacy = PolicyEngineUKDataset( + name="legacy", + description="legacy", + filepath=dataset.filepath, + year=dataset.year, + data=dataset.data, + ) + legacy.save() + with pd.HDFStore(dataset.filepath, mode="r") as store: + assert f"/{DATA_YEAR_KEY}" not in store.keys() + + +def test_load_datasets_refuses_a_year_file_without_a_data_year(source, tmp_path): + created = _cut(source, tmp_path, [2026]) + _write_legacy_year_file(created["tiny_uk_2024_2026"]) + + with pytest.raises(ValueError, match="no recorded data year"): + load_datasets(datasets=[source], years=[2026], data_folder=str(tmp_path)) + + +def test_ensure_datasets_regenerates_a_year_file_without_a_data_year( + source, tmp_path, direct +): + created = _cut(source, tmp_path, [2026]) + _write_legacy_year_file(created["tiny_uk_2024_2026"]) + + ensured = ensure_datasets( + datasets=[source], + years=[2026], + data_folder=str(tmp_path), + allow_unmanaged=True, + ) + + assert ensured["tiny_uk_2024_2026"].data_year == DATA_YEAR + reloaded = load_datasets(datasets=[source], years=[2026], data_folder=str(tmp_path)) + assert reloaded["tiny_uk_2024_2026"].data_year == DATA_YEAR + _assert_matches_direct(_run(reloaded["tiny_uk_2024_2026"]), direct, 2026) + + +def test_year_file_loaded_from_disk_matches_direct_run(source, tmp_path, direct): + """The production path: a second ensure_datasets() loads the files. + + Both tables come back through the HDF5 round trip, which stores object + columns as categoricals (depending on the pandas version). + """ + _cut(source, tmp_path, [2025, 2026]) + + loaded = load_datasets( + datasets=[source], years=[2025, 2026], data_folder=str(tmp_path) + ) + + for year in (2025, 2026): + dataset = loaded[f"tiny_uk_2024_{year}"] + assert dataset.data_year == DATA_YEAR + assert dataset.data_year_data is not None + _assert_matches_direct(_run(dataset), direct, year) + + +REFORM = { + "gov.dwp.state_pension.new_state_pension.amount": 260.0, + "gov.dwp.state_pension.basic_state_pension.amount": 200.0, +} + + +def test_reformed_year_file_run_matches_direct_run_with_the_same_reform( + year_files, source, direct +): + """A State Pension reform reaches the year-file run as it does the direct. + + policyengine.py applies a policy from the start of the simulated year. + Before #556 the simulated year was also the data year, so the reform set + the rate both numerator and denominator are taken from. With the data + year anchored on the survey year, the reform scales the State Pension. + """ + simulation = _run(year_files["tiny_uk_2024_2026"], policy=REFORM) + reformed = _direct(source) + simulation_modifier_from_parameter_values(simulation.policy.parameter_values)( + reformed + ) + + _assert_matches_direct(simulation, reformed, 2026) + # The reform raises every recipient's State Pension above the baseline. + baseline = direct.calculate("state_pension", 2026).values + raised = reformed.calculate("state_pension", 2026).values + assert (raised[baseline > 0] > baseline[baseline > 0] + 1).all() + + +def test_projected_input_is_ordered_and_leaves_weights_to_the_model(year_files): + dataset = year_files["tiny_uk_2024_2026"] + + built = _policyengine_uk_input(dataset) + + assert built.years == [2024, 2025, 2026, 2027, 2028, 2029, 2030] + for year in built.years: + if year == 2026: + continue + # A direct run calculates person and benefit-unit weights. + assert "person_weight" not in built[year].person.columns + assert "benunit_weight" not in built[year].benunit.columns + assert "person_weight" in built[2026].person.columns + + +def test_saved_output_records_its_data_year_and_old_outputs_are_rerun(year_files): + import uuid + + from policyengine.tax_benefit_models.common.model_version import ( + output_dataset_filepath, + ) + + dataset = year_files["tiny_uk_2024_2026"] + run_id = f"uk-data-year-{uuid.uuid4().hex}" + first = Simulation( + id=run_id, dataset=dataset, tax_benefit_model_version=pe.uk.model + ) + first.run() + first.save() + + reloaded = Simulation( + id=run_id, dataset=dataset, tax_benefit_model_version=pe.uk.model + ) + reloaded.load() + assert reloaded.output_dataset.data_year == DATA_YEAR + + # An output saved before #556 records no data year. + path = output_dataset_filepath(first) + PolicyEngineUKDataset( + name="legacy-output", + description="legacy-output", + filepath=str(path), + year=2026, + is_output_dataset=True, + data=first.output_dataset.data, + ).save() + legacy = Simulation( + id=run_id, dataset=dataset, tax_benefit_model_version=pe.uk.model + ) + with pytest.raises(ValueError, match="records none"): + legacy.load() + + legacy.ensure() + assert legacy.output_dataset.data_year == DATA_YEAR + with pd.HDFStore(path, mode="r") as store: + assert f"/{DATA_YEAR_KEY}" in store.keys() + + +@settings( + max_examples=6, + deadline=None, + suppress_health_check=[HealthCheck.function_scoped_fixture], +) +@given( + ages=st.lists(st.integers(min_value=60, max_value=100), min_size=7, max_size=7), + genders=st.lists(st.sampled_from(["MALE", "FEMALE"]), min_size=7, max_size=7), + reported=st.lists( + st.floats(min_value=0, max_value=25_000, allow_nan=False), + min_size=7, + max_size=7, + ), + year=st.integers(min_value=2025, max_value=2030), +) +def test_any_year_file_matches_the_direct_run( + tmp_path_factory, ages, genders, reported, year +): + """Differential invariant over random pensioners and projection years.""" + directory = tmp_path_factory.mktemp("random") + source = _write_source( + directory / "random_uk_2024.h5", + _tables(ages=ages, genders=genders, reported=reported), + ) + created = _cut(source, directory / "data", [year]) + + simulation = _run(created[f"random_uk_2024_{year}"]) + + _assert_matches_direct(simulation, _direct(source), year)