diff --git a/changelog.d/spi-unknown-region.fixed.md b/changelog.d/spi-unknown-region.fixed.md new file mode 100644 index 00000000..f5111f98 --- /dev/null +++ b/changelog.d/spi-unknown-region.fixed.md @@ -0,0 +1 @@ +Take Scottish taxpayer status in the SPI dataset from HMRC's `SCOT_TXP` flag rather than the region code. `GORCODE` 13, 14 and -1 (address abroad, address unknown, composite records) stay `UNKNOWN`, which policyengine-uk 2.104.5 and later can simulate; `load_spi_dataset` relabels them `SOUTH_EAST` only when the imported model cannot simulate them, and rebuilds a cached SPI H5 that predates the flag. diff --git a/policyengine_uk_data/datasets/spi.py b/policyengine_uk_data/datasets/spi.py index 41de9641..0f7b01a2 100644 --- a/policyengine_uk_data/datasets/spi.py +++ b/policyengine_uk_data/datasets/spi.py @@ -1,3 +1,5 @@ +from functools import cache + from policyengine_uk_data.storage import STORAGE_FOLDER import pandas as pd import numpy as np @@ -23,10 +25,11 @@ 7: (74, 90), } -# SPI GORCODE → policyengine-uk region enum. -# NB the SPI codebook does not include a "region unknown" code; we surface -# unknown codes explicitly rather than silently mapping them to SOUTH_EAST -# (which the previous implementation did, distorting regional income totals). +# SPI GORCODE → policyengine-uk region enum. GORCODE also takes 13 ("Address +# abroad"), 14 ("Address unknown or not available") and -1 (composite +# records). None of those is a UK region, so they become "UNKNOWN" rather +# than SOUTH_EAST (which the previous implementation used, distorting +# regional income totals). REGION_MAP = { 1: "NORTH_EAST", 2: "NORTH_WEST", @@ -43,6 +46,57 @@ } +def _spi_shaped_household(region: str) -> UKSingleYearDataset: + # The fixture policyengine-uk's own test_rent_uprating.py simulates with + # Region.UNKNOWN, so the model keeps supporting this shape. + ids = [1] + return UKSingleYearDataset( + person=pd.DataFrame( + { + "person_id": ids, + "person_benunit_id": ids, + "person_household_id": ids, + "age": [40], + "employment_income": [60_000.0], + } + ), + benunit=pd.DataFrame({"benunit_id": ids}), + household=pd.DataFrame( + { + "household_id": ids, + "household_weight": [1.0], + "region": [region], + "rent": [0.0], + "tenure_type": ["OWNED_OUTRIGHT"], + "council_tax": [0.0], + } + ), + fiscal_year=SPI_FISCAL_YEAR, + ) + + +@cache +def model_simulates_unknown_region() -> bool: + """Whether the imported policyengine-uk can build a simulation of an + SPI-shaped household in Region.UNKNOWN. + + Releases before 2.104.5 have no rent index for it + (PolicyEngine/policyengine-uk#1985). This builds one household rather than + reading the installed version, which need not be the imported code + (``make data-local`` puts a checkout on PYTHONPATH). A failure counts as + "no" only if the same household labelled SOUTH_EAST builds; otherwise + that error is raised, since relabelling would not help. + """ + from policyengine_uk import Microsimulation + + try: + Microsimulation(dataset=_spi_shaped_household("UNKNOWN")) + except Exception: + Microsimulation(dataset=_spi_shaped_household("SOUTH_EAST")) + return False + return True + + def _get_marriage_allowance(fiscal_year: int) -> float: """Return the maximum Marriage Allowance transfer for the given UK fiscal year in £. This equals ``max`` × ``personal_allowance`` at the start of @@ -98,10 +152,10 @@ def create_spi( existing call sites don't break. seed: Seed for the random age imputation. Fixed by default so builds are deterministic. - unknown_region: Fallback region label for SPI GORCODE values outside - the documented 1-12 range. Defaults to ``"UNKNOWN"`` so regional - totals are not silently distorted; pass ``"SOUTH_EAST"`` to - reproduce legacy behaviour if needed. + unknown_region: Region label for SPI GORCODE values outside 1-12 + (address abroad, address unknown, composite records). Defaults to + ``"UNKNOWN"`` so regional totals are not silently distorted; pass + ``"SOUTH_EAST"`` to reproduce legacy behaviour if needed. """ df = pd.read_csv(spi_data_file_path, delimiter="\t") rng = np.random.default_rng(seed) @@ -121,6 +175,16 @@ def create_spi( person["dividend_income"] = df.DIVIDENDS person["gift_aid"] = df.GIFTAID household["region"] = df.GORCODE.map(REGION_MAP).fillna(unknown_region) + # GORCODE is the address at the end of the tax year; SCOT_TXP marks + # records HMRC taxed under the Scottish system for the year. The two + # disagree for some records and SCOT_TXP is also set on records with no + # UK region, so SCOT_TXP decides the rates. HMRC documents it as "." or 1; + # the 2022-23 tape holds 0 or 1. WELSH_TXP is not used: policyengine-uk + # has no separate Welsh rates. policyengine-uk carries dataset inputs to + # 2030; from 2031 it derives the status from the region again. + person["pays_scottish_income_tax"] = ( + pd.to_numeric(df.SCOT_TXP, errors="coerce") == 1 + ) household["rent"] = 0 household["tenure_type"] = "OWNED_OUTRIGHT" household["council_tax"] = 0 diff --git a/policyengine_uk_data/tests/test_spi_allowance_deductions.py b/policyengine_uk_data/tests/test_spi_allowance_deductions.py index f635bb1f..f4089290 100644 --- a/policyengine_uk_data/tests/test_spi_allowance_deductions.py +++ b/policyengine_uk_data/tests/test_spi_allowance_deductions.py @@ -10,6 +10,7 @@ def test_spi_overrides_allowance_deductions_not_policy_parameters(tmp_path): "DIVIDENDS": 0, "GIFTAID": 0, "GORCODE": 7, + "SCOT_TXP": 0, "INCBBS": 0, "INCPROP": 1_000, "PAY": 0, diff --git a/policyengine_uk_data/tests/test_spi_build.py b/policyengine_uk_data/tests/test_spi_build.py index eff98ef7..298f9585 100644 --- a/policyengine_uk_data/tests/test_spi_build.py +++ b/policyengine_uk_data/tests/test_spi_build.py @@ -30,6 +30,10 @@ allow_module_level=True, ) +from policyengine_core.errors import ParameterNotFoundError + +from policyengine_uk_data.datasets.spi import model_simulates_unknown_region + SPI_COLUMNS = [ "SEX", @@ -38,6 +42,7 @@ "DIVIDENDS", "GIFTAID", "GORCODE", + "SCOT_TXP", "INCBBS", "INCPROP", "PAY", @@ -357,7 +362,17 @@ def save(self, path): assert dataset_path.read_text() == "rebuilt h5" -def test_income_projection_loads_local_h5_dataset(monkeypatch): +@pytest.mark.parametrize( + "model_simulates, expected", + [ + (False, ["SOUTH_EAST", "LONDON", "SOUTH_EAST"]), + (True, ["UNKNOWN", "LONDON", "SOUTH_EAST"]), + ], +) +def test_income_projection_loads_local_h5_dataset( + monkeypatch, model_simulates, expected +): + """UNKNOWN becomes SOUTH_EAST only for a model that cannot simulate it.""" from policyengine_uk_data.utils import incomes_projection calls = {} @@ -375,16 +390,55 @@ def __init__(self, path): lambda: "/tmp/spi_2022_23.h5", ) monkeypatch.setattr(incomes_projection, "UKSingleYearDataset", FakeDataset) + monkeypatch.setattr( + incomes_projection, + "model_simulates_unknown_region", + lambda: model_simulates, + ) dataset = incomes_projection.load_spi_dataset() assert isinstance(dataset, FakeDataset) assert calls == {"path": "/tmp/spi_2022_23.h5"} - assert dataset.household["region"].tolist() == [ - "SOUTH_EAST", - "LONDON", - "SOUTH_EAST", - ] + assert dataset.household["region"].tolist() == expected + + +def test_income_projection_rebuilds_spi_dataset_without_scottish_flag( + tmp_path, monkeypatch +): + """A cached H5 from before create_spi read SCOT_TXP is rebuilt; a current + one is reused.""" + from policyengine_uk_data.datasets.spi import create_spi + from policyengine_uk_data.utils import incomes_projection + + tab_dir = tmp_path / "spi_2022_23" + tab_dir.mkdir() + tab = tab_dir / "put2223uk.tab" + _write_fake_spi(tab, gor_values=(11, 13), maind_values=(0, 0)) + dataset_path = tmp_path / "spi_2022_23.h5" + stale = create_spi(tab, 2022) + stale.person = stale.person.drop(columns="pays_scottish_income_tax") + stale.save(dataset_path) + + builds = [] + + def counting_create_spi(path, fiscal_year): + builds.append(fiscal_year) + return create_spi(path, fiscal_year) + + monkeypatch.setattr(incomes_projection, "STORAGE_FOLDER", tmp_path) + monkeypatch.setattr(incomes_projection, "SPI_RELEASE_NAME", "spi_2022_23") + monkeypatch.setattr(incomes_projection, "SPI_TAB_FILENAME", "put2223uk.tab") + monkeypatch.setattr(incomes_projection, "SPI_H5_FILENAME", "spi_2022_23.h5") + monkeypatch.setattr(incomes_projection, "SPI_FISCAL_YEAR", 2022) + monkeypatch.setattr(incomes_projection, "create_spi", counting_create_spi) + + assert not incomes_projection._has_scottish_taxpayer_flag(dataset_path) + assert incomes_projection.ensure_spi_dataset() == str(dataset_path) + assert builds == [2022] + assert incomes_projection._has_scottish_taxpayer_flag(dataset_path) + assert incomes_projection.ensure_spi_dataset() == str(dataset_path) + assert builds == [2022] def _write_income_model_cache(path, income_module, metadata): @@ -458,6 +512,263 @@ def test_income_model_cache_accepts_current_spi_release(tmp_path, monkeypatch): assert income_module.create_income_model().metadata == current_metadata +def _set_spi_columns(path, **columns): + df = pd.read_csv(path, sep="\t") + for col, values in columns.items(): + df[col] = list(values) + df.to_csv(path, sep="\t", index=False) + + +# HMRC's GORCODE codes 1-12 (SN 9422, Annex A) as policyengine-uk regions, +# written out here rather than read from REGION_MAP so a wrong mapping fails. +HMRC_REGIONS = { + 1: "NORTH_EAST", + 2: "NORTH_WEST", + 3: "YORKSHIRE", # Yorkshire and the Humber + 4: "EAST_MIDLANDS", + 5: "WEST_MIDLANDS", + 6: "EAST_OF_ENGLAND", + 7: "LONDON", + 8: "SOUTH_EAST", + 9: "SOUTH_WEST", + 10: "WALES", + 11: "SCOTLAND", + 12: "NORTHERN_IRELAND", +} + + +@pytest.mark.parametrize("not_scottish", [0, ".", ""]) +def test_create_spi_region_and_scottish_taxpayer_invariants(tmp_path, not_scottish): + """For every GORCODE (documented 1-14, composite -1, undocumented 99) and + SCOT_TXP value, the region follows GORCODE alone and is always a Region + member, and Scottish taxpayer status follows SCOT_TXP alone. HMRC + documents "not a Scottish taxpayer" as "."; the 2022-23 tape writes 0. + """ + from itertools import product + + from policyengine_uk.variables.household.demographic.geography import Region + + from policyengine_uk_data.datasets.spi import create_spi + + cases = list(product([-1, *range(1, 15), 99], (not_scottish, 1))) + gor = [g for g, _ in cases] + scot = [s for _, s in cases] + tab = tmp_path / "spi.tab" + _write_fake_spi(tab, gor_values=gor, maind_values=[0] * len(cases)) + _set_spi_columns(tab, SCOT_TXP=scot) + + ds = create_spi(tab, 2022) + + regions = ds.household["region"].tolist() + assert regions == [HMRC_REGIONS.get(g, "UNKNOWN") for g in gor] + assert set(regions) <= {region.name for region in Region} + assert ds.person["pays_scottish_income_tax"].tolist() == [s == 1 for s in scot] + # Address abroad, address unknown and composite records stay UNKNOWN even + # when they are Scottish taxpayers. + assert {r for r, g in zip(regions, gor) if g in (-1, 13, 14)} == {"UNKNOWN"} + + +def test_create_spi_scottish_taxpayer_status_survives_h5_round_trip(tmp_path): + from policyengine_uk.data import UKSingleYearDataset + + from policyengine_uk_data.datasets.spi import create_spi + + tab = tmp_path / "spi.tab" + _write_fake_spi(tab, gor_values=(13, 11, 7), maind_values=(0, 0, 0)) + _set_spi_columns(tab, SCOT_TXP=(1, 0, 1)) + ds = create_spi(tab, 2022) + ds.save(tmp_path / "spi.h5") + + loaded = UKSingleYearDataset(str(tmp_path / "spi.h5")) + + assert loaded.person["pays_scottish_income_tax"].tolist() == [True, False, True] + assert loaded.household["region"].tolist() == ["UNKNOWN", "SCOTLAND", "LONDON"] + + +def _missing(name): + return ParameterNotFoundError(name, "2023-01-01", "rent") + + +UNKNOWN_RENT_INDEX = _missing( + "gov.economic_assumptions.yoy_growth.ons.private_rental_prices.UNKNOWN" +) + + +@pytest.mark.parametrize( + "errors, expected", + [ + ({}, True), + ({"UNKNOWN": UNKNOWN_RENT_INDEX}, False), + # Any failure that relabelling to SOUTH_EAST cures. + ({"UNKNOWN": KeyError("UNKNOWN")}, False), + ], +) +def test_unknown_region_probe_reads_the_simulation(monkeypatch, errors, expected): + import policyengine_uk + + regions = [] + + def simulate(dataset): + (region,) = dataset.household["region"] + regions.append(region) + if region in errors: + raise errors[region] + + monkeypatch.setattr(policyengine_uk, "Microsimulation", simulate) + + assert model_simulates_unknown_region.__wrapped__() is expected + assert regions == (["UNKNOWN"] if expected else ["UNKNOWN", "SOUTH_EAST"]) + + +@pytest.mark.parametrize( + "error", + [ + _missing("gov.hmrc.unrelated.UNKNOWN"), + _missing("gov.hmrc.income_tax.rates.uk"), + KeyError("employment_income"), + ], +) +def test_unknown_region_probe_raises_failures_relabelling_would_not_cure( + monkeypatch, error +): + """When the SOUTH_EAST household fails too, its error is raised, not read + as a missing UNKNOWN index.""" + import policyengine_uk + + original = RuntimeError("the UNKNOWN household failed") + + def simulate(dataset): + (region,) = dataset.household["region"] + raise original if region == "UNKNOWN" else error + + monkeypatch.setattr(policyengine_uk, "Microsimulation", simulate) + + with pytest.raises(type(error)) as raised: + model_simulates_unknown_region.__wrapped__() + assert raised.value is error + assert raised.value.__context__ is original + + +def test_unknown_region_probe_agrees_with_policyengine_uk_release(): + """2.104.5 is the first policyengine-uk release with + PolicyEngine/policyengine-uk#1985. For a final release not installed from + a direct URL, whose imported economic_assumptions.py is the installed file + and matches its RECORD, the probe agrees with the release number.""" + import base64 + import hashlib + from importlib.metadata import PackageNotFoundError, distribution + from pathlib import Path + + from packaging.version import Version + from policyengine_uk.data import economic_assumptions + + try: + release = distribution("policyengine-uk") + except PackageNotFoundError: + pytest.skip("policyengine-uk is not installed as a distribution") + version = Version(release.version) + if version.is_prerelease or version.is_devrelease or version.local: + pytest.skip(f"policyengine-uk {version} is not a final release") + if release.read_text("direct_url.json") is not None: + pytest.skip("policyengine-uk was installed from a direct URL") + # #1985 changed this module, so it must be the installed file, unmodified. + path = "policyengine_uk/data/economic_assumptions.py" + installed = Path(release.locate_file(path)) + if not installed.exists() or not installed.samefile(economic_assumptions.__file__): + pytest.skip("the imported policyengine-uk is not the installed release") + record = {file.as_posix(): file.hash for file in release.files or []}.get(path) + if record is None: + pytest.skip("policyengine-uk's RECORD does not list the module") + digest = hashlib.new(record.mode, installed.read_bytes()).digest() + if base64.urlsafe_b64encode(digest).rstrip(b"=").decode() != record.value: + pytest.skip("the installed economic_assumptions.py has been modified") + + assert model_simulates_unknown_region() == (version >= Version("2.104.5")) + + +# GORCODE, SCOT_TXP: abroad, unknown, composite, London, Scotland, abroad and +# Scottish, Scotland but not Scottish. +SIMULATED_RECORDS = ((13, 0), (14, 0), (-1, 0), (7, 0), (11, 1), (13, 1), (11, 0)) +# The data year, an uprated year and 2030, the last year policyengine-uk +# carries dataset inputs to. +YEARS = [2022, 2026, 2030] + + +def _spi_income_tax(tmp_path, region_rule=False, **kwargs): + """Income tax on SIMULATED_RECORDS, each paid £60,000, in YEARS. With + region_rule, the Scottish taxpayer flag is dropped, so policyengine-uk + derives it from the region as it did before create_spi read SCOT_TXP.""" + from policyengine_uk import Microsimulation + + from policyengine_uk_data.datasets.spi import create_spi + + tab = tmp_path / "spi.tab" + gor, scot = zip(*SIMULATED_RECORDS) + _write_fake_spi(tab, gor_values=gor, maind_values=[0] * len(gor)) + _set_spi_columns( + tab, SCOT_TXP=scot, PAY=[60_000] * len(gor), AGERANGE=[3] * len(gor) + ) + dataset = create_spi(tab, 2022, **kwargs) + if region_rule: + dataset.person = dataset.person.drop(columns="pays_scottish_income_tax") + sim = Microsimulation(dataset=dataset) + return {year: sim.calculate("income_tax", year).values for year in YEARS} + + +def test_spi_income_tax_follows_scottish_taxpayer_flag(tmp_path): + """Equal pay, so income tax depends only on SCOT_TXP, in the data year + and after uprating. Uses the legacy SOUTH_EAST label so it runs on any + policyengine-uk release.""" + for year, tax in _spi_income_tax(tmp_path, unknown_region="SOUTH_EAST").items(): + abroad, unknown, composite, london, scotland, abroad_scot, scotland_ruk = tax + + assert london > 0, year + assert abroad == unknown == composite == scotland_ruk == london, year + assert abroad_scot == scotland != london, year + + +def test_spi_income_tax_changes_only_where_scottish_flag_and_region_disagree( + tmp_path, +): + """Records whose SCOT_TXP agrees with their region get exactly the income + tax they got when the region decided Scottish status. The two records + where they disagree change.""" + label = "UNKNOWN" if model_simulates_unknown_region() else "SOUTH_EAST" + flag = _spi_income_tax(tmp_path, unknown_region=label) + region = _spi_income_tax(tmp_path, region_rule=True, unknown_region=label) + agrees = np.array([(g == 11) == (s == 1) for g, s in SIMULATED_RECORDS]) + + for year in YEARS: + assert (flag[year][agrees] == region[year][agrees]).all(), year + assert (flag[year][~agrees] != region[year][~agrees]).all(), year + + +def test_create_spi_output_with_unknown_region_can_be_simulated(tmp_path): + """SPI records with an address abroad (13), an unknown address (14) or a + composite record (-1) keep region UNKNOWN and still run through + policyengine-uk, giving record for record the income tax of the legacy + SOUTH_EAST relabelling: the region label does not move income tax. + Models without PolicyEngine/policyengine-uk#1985 must fail on the missing + rent index, and on nothing else. + """ + if not model_simulates_unknown_region(): + with pytest.raises( + ParameterNotFoundError, match=r"private_rental_prices\.UNKNOWN'" + ): + _spi_income_tax(tmp_path) + pytest.xfail( + "policyengine-uk before 2.104.5 (PolicyEngine/policyengine-uk#1985) " + "has no rent index for Region.UNKNOWN" + ) + + tax = _spi_income_tax(tmp_path) + legacy = _spi_income_tax(tmp_path, unknown_region="SOUTH_EAST") + + for year in YEARS: + assert (tax[year] > 0).all(), year + assert (tax[year] == legacy[year]).all(), year + + def test_create_spi_marks_every_taxpayer_as_their_benefit_units_claimant(tmp_path): from policyengine_uk_data.datasets.spi import create_spi diff --git a/policyengine_uk_data/utils/incomes_projection.py b/policyengine_uk_data/utils/incomes_projection.py index 144e3f83..7e0fe5e8 100644 --- a/policyengine_uk_data/utils/incomes_projection.py +++ b/policyengine_uk_data/utils/incomes_projection.py @@ -7,6 +7,7 @@ SPI_RELEASE_NAME, SPI_TAB_FILENAME, create_spi, + model_simulates_unknown_region, ) from policyengine_uk_data.storage import STORAGE_FOLDER from policyengine_uk_data.targets.sources.hmrc_spi import ( @@ -31,6 +32,12 @@ def _read_spi_dataset_year(dataset_path) -> int: return int(store["time_period"].iloc[0]) +def _has_scottish_taxpayer_flag(dataset_path) -> bool: + # SPI H5s built before create_spi read SCOT_TXP lack this column. + with pd.HDFStore(dataset_path, mode="r") as store: + return "pays_scottish_income_tax" in store.select("person", stop=0) + + def ensure_spi_dataset() -> str: """Create the SPI H5 projection input from the current TAB release if needed. @@ -42,6 +49,7 @@ def ensure_spi_dataset() -> str: if ( dataset_path.exists() and _read_spi_dataset_year(dataset_path) == SPI_FISCAL_YEAR + and _has_scottish_taxpayer_flag(dataset_path) ): return str(dataset_path) @@ -64,9 +72,13 @@ def ensure_spi_dataset() -> str: def load_spi_dataset() -> UKSingleYearDataset: dataset = UKSingleYearDataset(ensure_spi_dataset()) - dataset.household["region"] = dataset.household["region"].replace( - {"UNKNOWN": "SOUTH_EAST"} - ) + # Older policyengine-uk releases cannot simulate an unknown region. The + # stand-in label does not move income tax, which follows the Scottish + # taxpayer flag. + if not model_simulates_unknown_region(): + dataset.household["region"] = dataset.household["region"].replace( + {"UNKNOWN": "SOUTH_EAST"} + ) return dataset diff --git a/uv.lock b/uv.lock index 5ae82135..dce9c828 100644 --- a/uv.lock +++ b/uv.lock @@ -1433,7 +1433,7 @@ wheels = [ [[package]] name = "policyengine-uk-data" -version = "1.57.4" +version = "1.58.0" source = { editable = "." } dependencies = [ { name = "google-auth" },