Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/spi-unknown-region.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Take Scottish taxpayer status in the SPI dataset from HMRC's `SCOT_TXP` flag rather than the region code. `GORCODE` 13, 14 and -1 (address abroad, address unknown, composite records) stay `UNKNOWN`, which policyengine-uk 2.104.5 and later can simulate; `load_spi_dataset` relabels them `SOUTH_EAST` only when the imported model cannot simulate them, and rebuilds a cached SPI H5 that predates the flag.
80 changes: 72 additions & 8 deletions policyengine_uk_data/datasets/spi.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
from functools import cache

from policyengine_uk_data.storage import STORAGE_FOLDER
import pandas as pd
import numpy as np
Expand All @@ -23,10 +25,11 @@
7: (74, 90),
}

# SPI GORCODE → policyengine-uk region enum.
# NB the SPI codebook does not include a "region unknown" code; we surface
# unknown codes explicitly rather than silently mapping them to SOUTH_EAST
# (which the previous implementation did, distorting regional income totals).
# SPI GORCODE → policyengine-uk region enum. GORCODE also takes 13 ("Address
# abroad"), 14 ("Address unknown or not available") and -1 (composite
# records). None of those is a UK region, so they become "UNKNOWN" rather
# than SOUTH_EAST (which the previous implementation used, distorting
# regional income totals).
REGION_MAP = {
1: "NORTH_EAST",
2: "NORTH_WEST",
Expand All @@ -43,6 +46,57 @@
}


def _spi_shaped_household(region: str) -> UKSingleYearDataset:
# The fixture policyengine-uk's own test_rent_uprating.py simulates with
# Region.UNKNOWN, so the model keeps supporting this shape.
ids = [1]
return UKSingleYearDataset(
person=pd.DataFrame(
{
"person_id": ids,
"person_benunit_id": ids,
"person_household_id": ids,
"age": [40],
"employment_income": [60_000.0],
}
),
benunit=pd.DataFrame({"benunit_id": ids}),
household=pd.DataFrame(
{
"household_id": ids,
"household_weight": [1.0],
"region": [region],
"rent": [0.0],
"tenure_type": ["OWNED_OUTRIGHT"],
"council_tax": [0.0],
}
),
fiscal_year=SPI_FISCAL_YEAR,
)


@cache
def model_simulates_unknown_region() -> bool:
"""Whether the imported policyengine-uk can build a simulation of an
SPI-shaped household in Region.UNKNOWN.

Releases before 2.104.5 have no rent index for it
(PolicyEngine/policyengine-uk#1985). This builds one household rather than
reading the installed version, which need not be the imported code
(``make data-local`` puts a checkout on PYTHONPATH). A failure counts as
"no" only if the same household labelled SOUTH_EAST builds; otherwise
that error is raised, since relabelling would not help.
"""
from policyengine_uk import Microsimulation

try:
Microsimulation(dataset=_spi_shaped_household("UNKNOWN"))
except Exception:
Microsimulation(dataset=_spi_shaped_household("SOUTH_EAST"))
return False
return True


def _get_marriage_allowance(fiscal_year: int) -> float:
"""Return the maximum Marriage Allowance transfer for the given UK fiscal
year in £. This equals ``max`` × ``personal_allowance`` at the start of
Expand Down Expand Up @@ -98,10 +152,10 @@ def create_spi(
existing call sites don't break.
seed: Seed for the random age imputation. Fixed by default so builds
are deterministic.
unknown_region: Fallback region label for SPI GORCODE values outside
the documented 1-12 range. Defaults to ``"UNKNOWN"`` so regional
totals are not silently distorted; pass ``"SOUTH_EAST"`` to
reproduce legacy behaviour if needed.
unknown_region: Region label for SPI GORCODE values outside 1-12
(address abroad, address unknown, composite records). Defaults to
``"UNKNOWN"`` so regional totals are not silently distorted; pass
``"SOUTH_EAST"`` to reproduce legacy behaviour if needed.
"""
df = pd.read_csv(spi_data_file_path, delimiter="\t")
rng = np.random.default_rng(seed)
Expand All @@ -121,6 +175,16 @@ def create_spi(
person["dividend_income"] = df.DIVIDENDS
person["gift_aid"] = df.GIFTAID
household["region"] = df.GORCODE.map(REGION_MAP).fillna(unknown_region)
# GORCODE is the address at the end of the tax year; SCOT_TXP marks
# records HMRC taxed under the Scottish system for the year. The two
# disagree for some records and SCOT_TXP is also set on records with no
# UK region, so SCOT_TXP decides the rates. HMRC documents it as "." or 1;
# the 2022-23 tape holds 0 or 1. WELSH_TXP is not used: policyengine-uk
# has no separate Welsh rates. policyengine-uk carries dataset inputs to
# 2030; from 2031 it derives the status from the region again.
person["pays_scottish_income_tax"] = (
pd.to_numeric(df.SCOT_TXP, errors="coerce") == 1
)
household["rent"] = 0
household["tenure_type"] = "OWNED_OUTRIGHT"
household["council_tax"] = 0
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ def test_spi_overrides_allowance_deductions_not_policy_parameters(tmp_path):
"DIVIDENDS": 0,
"GIFTAID": 0,
"GORCODE": 7,
"SCOT_TXP": 0,
"INCBBS": 0,
"INCPROP": 1_000,
"PAY": 0,
Expand Down
Loading
Loading