Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -19,3 +19,6 @@
**/_build
!policyengine_uk_data/storage/*.csv
**/version.json

# Hypothesis example database (property-based tests)
.hypothesis/
1 change: 1 addition & 0 deletions changelog.d/spi-synthetic-reported-benefits.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
SPI-synthetic rows no longer report income-related (except council tax reduction, which keeps its imputed value for now), out-of-work or Child Benefit receipt, take their industrial injuries, armed forces compensation and bereavement support from the FRS donor, and get UC, Pension Credit and `receives_benefits_in_own_right` flags from their own reports instead of the donor's.
70 changes: 46 additions & 24 deletions policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@
STORAGE_FOLDER,
)
from policyengine_uk_data.parameters import load_take_up_rate, load_parameter
from policyengine_uk_data.utils.takeup import assign_takeup_with_reported_anchors
from policyengine_uk_data.datasets.childcare.assumptions import (
EXTENDED_HOURS_MEAN,
EXTENDED_HOURS_SD,
Expand Down Expand Up @@ -74,6 +75,14 @@
"esa_contrib_reported",
"esa_income_reported",
)
# Take-up flags anchored on reported receipt: flag -> (take-up rate
# parameter, person-level report column). A benefit unit with any member
# reporting receipt claims with certainty; the rest are filled at random.
REPORTED_TAKEUP_ANCHORS = {
"would_claim_child_benefit": ("child_benefit", "child_benefit_reported"),
"would_claim_pc": ("pension_credit", "pension_credit_reported"),
"would_claim_uc": ("universal_credit", "universal_credit_reported"),
}
NON_ADVANCED_EDUCATION_LEVELS = (
"PRE_PRIMARY",
"PRIMARY",
Expand Down Expand Up @@ -211,6 +220,34 @@ def derive_esa_support_group_proxy(
)


def reported_benunit_mask(
person: pd.DataFrame, benunit: pd.DataFrame, person_column: str
) -> np.ndarray:
"""Benefit units with any member reporting a positive ``person_column``."""
reporter_benunits = set(
person.loc[person[person_column] > 0, "person_benunit_id"].values
)
return benunit["benunit_id"].isin(reporter_benunits).values


def assign_reported_takeup(
person: pd.DataFrame,
benunit: pd.DataFrame,
flag: str,
year: int,
draws: np.ndarray,
) -> np.ndarray:
"""Take-up for one ``REPORTED_TAKEUP_ANCHORS`` flag: benefit units
reporting receipt claim, and the rest are filled at random so the share
claiming matches the take-up rate (see ``utils/takeup.py``)."""
rate_name, report_column = REPORTED_TAKEUP_ANCHORS[flag]
return assign_takeup_with_reported_anchors(
draws,
load_take_up_rate(rate_name, year),
reported_mask=reported_benunit_mask(person, benunit, report_column),
)


def derive_receives_benefits_in_own_right(pe_person: pd.DataFrame) -> pd.Series:
"""Identify people reporting adult benefits that end QYP status."""

Expand Down Expand Up @@ -1487,9 +1524,6 @@ def determine_education_level(fted_val, typeed2_val, age_val):
generator = np.random.default_rng(seed=100)

# Load take-up rates from parameter files
child_benefit_rate = load_take_up_rate("child_benefit", year)
pension_credit_rate = load_take_up_rate("pension_credit", year)
universal_credit_rate = load_take_up_rate("universal_credit", year)
marriage_allowance_rate = load_take_up_rate("marriage_allowance", year)
child_benefit_opts_out_rate = load_take_up_rate("child_benefit_opts_out_rate", year)
tfc_rate = load_take_up_rate("tax_free_childcare", year)
Expand All @@ -1507,40 +1541,28 @@ def determine_education_level(fted_val, typeed2_val, age_val):
# who report positive receipt of a benefit are assigned takeup=True with
# certainty; the remaining non-reporters are filled probabilistically to
# hit the aggregate target rate. See policyengine_uk_data/utils/takeup.py.
from policyengine_uk_data.utils.takeup import (
assign_takeup_with_reported_anchors,
)

def _reported_benunit_mask(person_column: str) -> np.ndarray:
reporter_benunits = set(
pe_person.loc[pe_person[person_column] > 0, "person_benunit_id"].values
)
return pe_benunit["benunit_id"].isin(reporter_benunits).values

# Person-level
pe_person["would_claim_marriage_allowance"] = (
generator.random(len(pe_person)) < marriage_allowance_rate
)

# Benefit unit-level — anchor on any adult in the benefit unit having
# reported positive receipt in the FRS benefits table.
pe_benunit["would_claim_child_benefit"] = assign_takeup_with_reported_anchors(
pe_benunit["would_claim_child_benefit"] = assign_reported_takeup(
pe_person,
pe_benunit,
"would_claim_child_benefit",
year,
generator.random(len(pe_benunit)),
child_benefit_rate,
reported_mask=_reported_benunit_mask("child_benefit_reported"),
)
pe_benunit["child_benefit_opts_out"] = (
generator.random(len(pe_benunit)) < child_benefit_opts_out_rate
)
pe_benunit["would_claim_pc"] = assign_takeup_with_reported_anchors(
generator.random(len(pe_benunit)),
pension_credit_rate,
reported_mask=_reported_benunit_mask("pension_credit_reported"),
pe_benunit["would_claim_pc"] = assign_reported_takeup(
pe_person, pe_benunit, "would_claim_pc", year, generator.random(len(pe_benunit))
)
pe_benunit["would_claim_uc"] = assign_takeup_with_reported_anchors(
generator.random(len(pe_benunit)),
universal_credit_rate,
reported_mask=_reported_benunit_mask("universal_credit_reported"),
pe_benunit["would_claim_uc"] = assign_reported_takeup(
pe_person, pe_benunit, "would_claim_uc", year, generator.random(len(pe_benunit))
)
pe_benunit["would_claim_tfc"] = generator.random(len(pe_benunit)) < tfc_rate

Expand Down
156 changes: 140 additions & 16 deletions policyengine_uk_data/datasets/imputations/frs_only.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,12 @@
add_disability_benefit_categories_from_reported_amounts,
add_disability_benefit_flags_from_reported_amounts,
)
from policyengine_uk_data.datasets.frs import (
BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS,
REPORTED_TAKEUP_ANCHORS,
assign_reported_takeup,
derive_receives_benefits_in_own_right,
)

logger = logging.getLogger(__name__)

Expand Down Expand Up @@ -102,6 +108,113 @@
"esa_income_reported",
]

# The QRF draws each person's benefit reports from their age, gender, region
# and incomes. It sees nothing of their benefit unit (partner, children,
# rent, capital), their health or their history, and policyengine-uk reads a
# positive report as an existing claim. On SPI-donor rows, after the draw:
#
# Zeroed (SPI_DONOR_ZEROED_PERSON_VARIABLES):
# - Income-related awards. Entitlement turns on the unit's joint means and
# make-up, which here come from the imputed incomes. UC and Pension Credit
# keep a route: their take-up flags are redrawn below. In
# policyengine-uk 2.93.0 the others (housing benefit, income support, tax
# credits, income-related ESA and JSA) can only be claimed with a report,
# so these rows no longer receive them. Sure Start Maternity Grant needs
# one of these awards. Council tax reduction is kept (below).
# - Benefits paid only to people out of work or incapable of it: ESA and JSA
# (contributory), incapacity benefit and severe disablement allowance. On
# the 2024-25 build, 41% of SPI-row ESA (contributory) reporters by weight
# earned more than ESA's permitted-work limit.
# - Child Benefit, which the model reads only through the take-up flag. The
# draw ignores the children: 34% of SPI-row reports by weight were in
# benefit units with no child or qualifying young person (FRS rows: 0%).
#
# Restored to the donor's own value (SPI_DONOR_RESTORED_PERSON_VARIABLES):
# industrial injuries, armed forces compensation and bereavement support.
# These follow from an injury, service or a death, not income, and the QRF
# drew them at 6.2, 2.4 and 4.6 times the FRS rate by weight.
#
# Kept as drawn: state pension (paid as reported once over pension age),
# winter fuel payment (not read by the model), the disability benefits and
# carer's allowance, whose drawn rates sit below the FRS rates as the income
# gradient implies. Council tax reduction is also kept as drawn for now. In
# 2.93.0 it too can only be claimed with a report, and zeroing it cut 2025
# CTR by 21% on the 2024-25 build. It will be zeroed together with a
# household-level CTR imputation (#499).
#
# Every column stays in the QRF chain above, so the values kept do not
# change. They were drawn alongside the values later zeroed or restored.
SPI_DONOR_ZEROED_PERSON_VARIABLES = [
"universal_credit_reported",
"pension_credit_reported",
"housing_benefit_reported",
"income_support_reported",
"working_tax_credit_reported",
"child_tax_credit_reported",
"jsa_income_reported",
"esa_income_reported",
"ssmg_reported",
"jsa_contrib_reported",
"esa_contrib_reported",
"incapacity_benefit_reported",
"sda_reported",
"child_benefit_reported",
]
SPI_DONOR_RESTORED_PERSON_VARIABLES = [
"iidb_reported",
"afcs_reported",
"bsp_reported",
]

# Take-up flags redrawn on SPI-donor rows. Whether a synthetic family claims
# a means-tested benefit at its imputed income is unobserved, so these units
# draw at the take-up rate. The Child Benefit flag keeps the donor's value:
# the award doesn't depend on the replaced incomes, and the donor's claim is
# for the same children.
SPI_DONOR_REDRAWN_TAKEUP_FLAGS = ("would_claim_uc", "would_claim_pc")
# Seed for those draws; create_frs uses 100.
SPI_DONOR_TAKEUP_SEED = 101


def apply_spi_donor_benefit_rules(
dataset: UKSingleYearDataset,
donor_person: pd.DataFrame | None = None,
) -> UKSingleYearDataset:
"""Apply the SPI-donor benefit rules above to ``dataset``.

``donor_person`` is the person table before the QRF draw, holding the
donor's own reports; without it the restored columns are left alone.
``receives_benefits_in_own_right`` and the redrawn take-up flags, which
``create_frs`` built from the donor's reports, are rebuilt from the rows'
own reports.
"""
dataset = dataset.copy()
person, benunit = dataset.person, dataset.benunit
for column in SPI_DONOR_ZEROED_PERSON_VARIABLES:
if column in person.columns:
person[column] = 0.0
if donor_person is not None:
for column in SPI_DONOR_RESTORED_PERSON_VARIABLES:
if column in person.columns and column in donor_person.columns:
person[column] = donor_person[column].values

if "receives_benefits_in_own_right" in person.columns:
own_right = person.reindex(
columns=list(BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS), fill_value=0.0
)
person["receives_benefits_in_own_right"] = (
derive_receives_benefits_in_own_right(own_right).values
)

year = int(str(dataset.time_period)[:4])
generator = np.random.default_rng(seed=SPI_DONOR_TAKEUP_SEED)
for flag in SPI_DONOR_REDRAWN_TAKEUP_FLAGS:
draws = generator.random(len(benunit))
report_column = REPORTED_TAKEUP_ANCHORS[flag][1]
if flag in benunit.columns and report_column in person.columns:
benunit[flag] = assign_reported_takeup(person, benunit, flag, year, draws)
return dataset


def _one_hot_encode(df: pd.DataFrame, columns: list[str]) -> pd.DataFrame:
"""Return ``df`` with object-typed ``columns`` one-hot encoded.
Expand Down Expand Up @@ -177,11 +290,13 @@ def impute_frs_only_variables(
to predict values for every row of ``target_dataset``; predictions
replace the existing (donor-leaked) values in
``FRS_ONLY_PERSON_VARIABLES`` only. Variables absent from either
frame are skipped silently.
frame are skipped silently. ``apply_spi_donor_benefit_rules`` then
zeroes or restores some reports and rebuilds the flags derived from
them, before the disability categories and flags are derived from the
final reports.
"""
from policyengine_uk_data.utils.qrf import QRF

target_dataset = target_dataset.copy()
donor_person = target_dataset.person.copy()

train_person = train_dataset.person
target_person = target_dataset.person
Expand All @@ -200,13 +315,32 @@ def impute_frs_only_variables(
len(missing),
sorted(missing),
)
if not outputs:
if outputs:
target_dataset = _impute_outputs(train_dataset, target_dataset, outputs)
else:
logger.warning(
"Stage-2 FRS-only imputation: no output variables available; "
"returning target_dataset unchanged."
"applying only the SPI-donor benefit rules."
)
return target_dataset

target_dataset = apply_spi_donor_benefit_rules(target_dataset, donor_person)
target_dataset.person = add_disability_benefit_categories_from_reported_amounts(
target_dataset.person,
int(str(target_dataset.time_period)[:4]),
)
target_dataset.person = add_disability_benefit_flags_from_reported_amounts(
target_dataset.person,
int(str(target_dataset.time_period)[:4]),
)

return target_dataset


def _impute_outputs(train_dataset, target_dataset, outputs):
"""Fit the stage-2 QRF and write its draws of ``outputs`` to the target."""
from policyengine_uk_data.utils.qrf import QRF

train_person = train_dataset.person
train_inputs_raw = _build_predictor_frame(train_dataset)
target_inputs_raw = _build_predictor_frame(target_dataset)

Expand Down Expand Up @@ -240,14 +374,4 @@ def impute_frs_only_variables(
# amounts or contributions and are non-negative by construction.
values = np.maximum(predictions[column].values, 0.0)
target_dataset.person[column] = values

target_dataset.person = add_disability_benefit_categories_from_reported_amounts(
target_dataset.person,
int(str(target_dataset.time_period)[:4]),
)
target_dataset.person = add_disability_benefit_flags_from_reported_amounts(
target_dataset.person,
int(str(target_dataset.time_period)[:4]),
)

return target_dataset
25 changes: 13 additions & 12 deletions policyengine_uk_data/tests/test_frs_only_imputation.py
Original file line number Diff line number Diff line change
Expand Up @@ -178,25 +178,26 @@ def test_frs_only_skips_missing_output_columns():


def test_frs_only_reported_values_correlate_with_training_pattern():
"""UC ``_reported`` predictions should respect the training-data pattern.
"""Drawn ``_reported`` values should respect the training-data pattern.

The stage-1 QRF-imputed income on the SPI-donor side gets fed back
as a stage-2 predictor. If the training data only has non-zero UC
for low-income respondents, the QRF should preferentially draw
near-zero values when predicting for high-income target rows,
compared with low-income target rows.
as a stage-2 predictor. If the training data only has non-zero
carer's allowance for low earners (it has an earnings limit), the QRF
should preferentially draw near-zero values when predicting for
high-income target rows, compared with low-income target rows. (UC
would be the obvious case, but SPI-donor rows have it zeroed.)
"""
from policyengine_uk_data.datasets.imputations.frs_only import (
impute_frs_only_variables,
)

# Train set with a clean employment-income → UC relationship:
# low-income respondents sometimes claim UC, high-income never do.
# Train set with a clean employment-income → carer's allowance relationship:
# low earners sometimes receive it, high earners never do.
rng = np.random.default_rng(42)
train = _fake_dataset(person_rows=2_000, seed=0)
low_income_mask = train.person["employment_income"] < 20_000
train.person["universal_credit_reported"] = 0.0
train.person.loc[low_income_mask, "universal_credit_reported"] = rng.gamma(
train.person["carers_allowance_reported"] = 0.0
train.person.loc[low_income_mask, "carers_allowance_reported"] = rng.gamma(
2, 4_000, size=int(low_income_mask.sum())
)

Expand All @@ -215,10 +216,10 @@ def test_frs_only_reported_values_correlate_with_training_pattern():
target_dataset=low_target,
)

high_mean = high_result.person["universal_credit_reported"].mean()
low_mean = low_result.person["universal_credit_reported"].mean()
high_mean = high_result.person["carers_allowance_reported"].mean()
low_mean = low_result.person["carers_allowance_reported"].mean()
assert high_mean < low_mean, (
"Stage-2 QRF should produce lower UC-receipt predictions for high-"
"Stage-2 QRF should produce lower carer's allowance predictions for high-"
f"income target rows (got high={high_mean:.2f} vs low={low_mean:.2f})."
)

Expand Down
Loading
Loading