Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/uc-managed-migration-takeup.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Draw `would_claim_uc_at_legacy_closure` for benefit units that report a legacy means-tested benefit but not Universal Credit, at DWP's Move to Universal Credit claim rates for their combination of legacy benefits (Stat-Xplore, notices to end of March 2026). policyengine-uk reads it once one of those benefits closes, so a legacy reporter whose `would_claim_uc` draw was false now moves to Universal Credit at the observed rate instead of losing its legacy award with nothing in its place. Units reporting Universal Credit, and units with no legacy benefit, are true. The draw has its own seeded generator, so every other take-up flag is unchanged, and SPI-synthetic rows are redrawn after stage-2 imputation rewrites their benefit receipt (#492).
22 changes: 21 additions & 1 deletion policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,11 @@
fill_with_mean,
STORAGE_FOLDER,
)
from policyengine_uk_data.parameters import load_take_up_rate, load_parameter
from policyengine_uk_data.parameters import (
load_parameter,
load_take_up_rate,
load_uc_managed_migration_claim_rates,
)
from policyengine_uk_data.datasets.childcare.assumptions import (
EXTENDED_HOURS_MEAN,
EXTENDED_HOURS_SD,
Expand Down Expand Up @@ -1508,7 +1512,9 @@ def determine_education_level(fted_val, typeed2_val, age_val):
# certainty; the remaining non-reporters are filled probabilistically to
# hit the aggregate target rate. See policyengine_uk_data/utils/takeup.py.
from policyengine_uk_data.utils.takeup import (
UC_MANAGED_MIGRATION_SEED,
assign_takeup_with_reported_anchors,
assign_uc_claim_at_legacy_closure,
)

def _reported_benunit_mask(person_column: str) -> np.ndarray:
Expand Down Expand Up @@ -1542,6 +1548,20 @@ def _reported_benunit_mask(person_column: str) -> np.ndarray:
universal_credit_rate,
reported_mask=_reported_benunit_mask("universal_credit_reported"),
)
# Whether a legacy-benefit family claims Universal Credit once its legacy
# benefits close, at DWP's Move to Universal Credit claim rates. Its own
# generator keeps every other draw on the seed=100 sequence. Checked
# because the dataset loader drops columns the model does not define.
require_variable(
"would_claim_uc_at_legacy_closure",
"Move to Universal Credit claim flag",
)
pe_benunit["would_claim_uc_at_legacy_closure"] = assign_uc_claim_at_legacy_closure(
pe_person,
pe_benunit,
load_uc_managed_migration_claim_rates(),
seed=UC_MANAGED_MIGRATION_SEED,
)
pe_benunit["would_claim_tfc"] = generator.random(len(pe_benunit)) < tfc_rate

pe_benunit["would_claim_extended_childcare"] = (
Expand Down
20 changes: 20 additions & 0 deletions policyengine_uk_data/datasets/imputations/income.py
Original file line number Diff line number Diff line change
Expand Up @@ -291,6 +291,26 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset:
train_dataset=dataset,
target_dataset=zero_weight_copy,
)
# Stage 2 rewrote these rows' legacy benefit and Universal Credit
# receipt, so redraw the Move to Universal Credit claim flag that
# create_frs drew from their donors' receipt.
if "would_claim_uc_at_legacy_closure" in zero_weight_copy.benunit.columns:
from policyengine_uk_data.parameters import (
load_uc_managed_migration_claim_rates,
)
from policyengine_uk_data.utils.takeup import (
UC_MANAGED_MIGRATION_SPI_SEED,
assign_uc_claim_at_legacy_closure,
)

zero_weight_copy.benunit["would_claim_uc_at_legacy_closure"] = (
assign_uc_claim_at_legacy_closure(
zero_weight_copy.person,
zero_weight_copy.benunit,
load_uc_managed_migration_claim_rates(),
seed=UC_MANAGED_MIGRATION_SPI_SEED,
)
)

dataset = impute_over_incomes(
dataset,
Expand Down
16 changes: 16 additions & 0 deletions policyengine_uk_data/parameters/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,3 +62,19 @@ def load_take_up_rate(variable_name: str, year: int = 2015) -> float:
Take-up rate as a float between 0 and 1
"""
return load_parameter("take_up", variable_name, year)


def load_uc_managed_migration_claim_rates() -> dict[str, float]:
"""Load Move to Universal Credit claim rates by legacy benefit combination.

Returns:
Claim rate for each combination in ``take_up/uc_managed_migration.yaml``
(benefit names joined by "+", plus "all"): households that claimed
Universal Credit over those that claimed or did not claim.
"""
with open(PARAMETERS_DIR / "take_up" / "uc_managed_migration.yaml") as f:
households = yaml.safe_load(f)["households"]
return {
combination: counts["claimed"] / (counts["claimed"] + counts["did_not_claim"])
for combination, counts in households.items()
}
42 changes: 42 additions & 0 deletions policyengine_uk_data/parameters/take_up/uc_managed_migration.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
description: >-
Households sent a Move to Universal Credit migration notice in Great Britain
(July 2022 to end of March 2026), by the legacy benefits they received and
whether they claimed Universal Credit. The claim rate for a combination is
claimed / (claimed + did_not_claim): the 538 households still in progress
are left out. Combinations DWP does not report use the all-households row.
"esa_income" is Stat-Xplore's "Employment Support Allowance"; notices go only
to claimants of an existing benefit, which is income-related ESA.
metadata:
unit: /1
label: Move to Universal Credit claim rate by legacy benefit combination
reference:
- title: >-
DWP Stat-Xplore, "Households invited to Move to Universal Credit",
table "MtUC Households 3 - Migration notices by legacy benefit and
move status" (retrieved 2026-10-03)
href: https://stat-xplore.dwp.gov.uk
- title: "DWP: Move to Universal Credit, July 2022 to end March 2026 (12 May 2026)"
href: https://www.gov.uk/government/statistics/move-to-universal-credit-july-2022-to-end-march-2026
# Benefit names follow the order child_tax_credit, working_tax_credit,
# housing_benefit, esa_income, income_support, jsa_income, joined by "+".
households:
child_tax_credit: {claimed: 36_723, did_not_claim: 17_733}
working_tax_credit: {claimed: 53_993, did_not_claim: 51_805}
child_tax_credit+working_tax_credit: {claimed: 363_984, did_not_claim: 126_397}
housing_benefit: {claimed: 30_900, did_not_claim: 10_121}
esa_income: {claimed: 321_075, did_not_claim: 13_944}
income_support: {claimed: 17_719, did_not_claim: 1_351}
jsa_income: {claimed: 5_867, did_not_claim: 329}
child_tax_credit+housing_benefit: {claimed: 15_646, did_not_claim: 1_042}
child_tax_credit+esa_income: {claimed: 11_171, did_not_claim: 381}
child_tax_credit+income_support: {claimed: 8_254, did_not_claim: 276}
child_tax_credit+jsa_income: {claimed: 381, did_not_claim: 8}
working_tax_credit+housing_benefit: {claimed: 7_820, did_not_claim: 1_042}
housing_benefit+esa_income: {claimed: 448_253, did_not_claim: 9_117}
housing_benefit+income_support: {claimed: 31_087, did_not_claim: 750}
housing_benefit+jsa_income: {claimed: 11_019, did_not_claim: 333}
child_tax_credit+working_tax_credit+housing_benefit: {claimed: 81_962, did_not_claim: 5_352}
child_tax_credit+housing_benefit+esa_income: {claimed: 73_859, did_not_claim: 949}
child_tax_credit+housing_benefit+income_support: {claimed: 58_764, did_not_claim: 691}
child_tax_credit+housing_benefit+jsa_income: {claimed: 2_286, did_not_claim: 30}
all: {claimed: 1_580_761, did_not_claim: 241_662}
181 changes: 181 additions & 0 deletions policyengine_uk_data/tests/test_uc_managed_migration_takeup.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,181 @@
"""Tests for the Move to Universal Credit claim flag.

``would_claim_uc_at_legacy_closure`` says whether a benefit unit claims
Universal Credit once a legacy benefit it reports closes (policyengine-uk
``legacy_benefits_closed``). Invariants:

1. Source: every combination is a "+"-join of LEGACY_BENEFITS in order, each
rate is claimed / (claimed + did_not_claim) and lies in (0, 1), and the
combinations add up to DWP's all-households row.
2. Every benefit unit reporting Universal Credit is True.
3. Every benefit unit reporting no legacy benefit is True, the model's
default, so the column changes nothing outside the legacy cohorts.
4. A legacy reporter not on Universal Credit is True exactly when its draw is
below its combination's rate ("all" for combinations DWP does not report).
5. Each cohort's share lands within sampling error of its rate.
6. The draw is deterministic, has its own generator and leaves NumPy's global
random state alone.
"""

from __future__ import annotations

import numpy as np
import pandas as pd
import pytest
import yaml
from hypothesis import given, settings
from hypothesis import strategies as st

from policyengine_uk_data.parameters import (
PARAMETERS_DIR,
load_uc_managed_migration_claim_rates,
)
from policyengine_uk_data.utils.takeup import (
LEGACY_BENEFITS,
UC_MANAGED_MIGRATION_SEED,
assign_uc_claim_at_legacy_closure,
legacy_benefit_combination,
)

RATES = load_uc_managed_migration_claim_rates()
REPORTED = [f"{benefit}_reported" for benefit in LEGACY_BENEFITS] + [
"universal_credit_reported"
]


def frames(units: list[list[dict]]) -> tuple[pd.DataFrame, pd.DataFrame]:
"""Person and benefit unit tables from a list of units of people."""
rows = [
{"person_benunit_id": 10 + i, **{c: person.get(c, 0.0) for c in REPORTED}}
for i, people in enumerate(units)
for person in people
]
person = pd.DataFrame(rows, columns=["person_benunit_id", *REPORTED])
# Benefit unit ids out of order, to catch positional mix-ups.
benunit = pd.DataFrame({"benunit_id": [10 + i for i in range(len(units))]})
return person, benunit.iloc[::-1].reset_index(drop=True)


people = st.lists(
st.fixed_dictionaries(
{c: st.sampled_from([0.0, 0.0, 0.0, 50.0, 3_000.0]) for c in REPORTED}
),
min_size=1,
max_size=3,
)


def test_source_combinations_and_rates():
with open(PARAMETERS_DIR / "take_up" / "uc_managed_migration.yaml") as f:
households = yaml.safe_load(f)["households"]
assert set(households) == set(RATES)
totals = {"claimed": 0, "did_not_claim": 0}
for combination, counts in households.items():
rate = counts["claimed"] / (counts["claimed"] + counts["did_not_claim"])
assert RATES[combination] == rate and 0 < rate < 1
if combination == "all":
continue
benefits = combination.split("+")
assert benefits == [b for b in LEGACY_BENEFITS if b in benefits]
for key in totals:
totals[key] += counts[key]
# Stat-Xplore applies disclosure control, so rows need not add exactly.
for key, total in totals.items():
assert abs(total - households["all"][key]) <= 20


def test_published_rates():
# Spot checks against Stat-Xplore table MtUC Households 3.
assert RATES["housing_benefit"] == pytest.approx(30_900 / 41_021)
assert RATES["housing_benefit+esa_income"] == pytest.approx(448_253 / 457_370)
assert RATES["working_tax_credit"] == pytest.approx(53_993 / 105_798)
assert RATES["all"] == pytest.approx(0.8674, abs=1e-4)


@settings(max_examples=200, deadline=None, derandomize=True)
@given(st.lists(people, min_size=1, max_size=40), st.integers(0, 2**32 - 1))
def test_anchors_and_draws(units, seed):
person, benunit = frames(units)
flags = assign_uc_claim_at_legacy_closure(person, benunit, RATES, seed=seed)
combination = legacy_benefit_combination(person, benunit)
draws = np.random.default_rng(seed).random(len(benunit))
assert flags.dtype == bool and len(flags) == len(benunit)
for i, unit_id in enumerate(benunit["benunit_id"]):
members = units[unit_id - 10]
reports = {c for c in REPORTED if any(p[c] > 0 for p in members)}
expected_combination = "+".join(
b for b in LEGACY_BENEFITS if f"{b}_reported" in reports
)
assert combination[i] == expected_combination
if "universal_credit_reported" in reports or not expected_combination:
assert flags[i]
else:
rate = RATES.get(expected_combination, RATES["all"])
assert flags[i] == (draws[i] < rate)


@pytest.mark.parametrize("combination", sorted(RATES))
def test_cohort_shares_within_sampling_error(combination):
benefits = [] if combination == "all" else combination.split("+")
if combination == "all":
# A combination DWP does not report (income-related ESA with Income
# Support) takes the all-households rate.
benefits = ["esa_income", "income_support"]
n = 20_000
units = [[{f"{b}_reported": 100.0 for b in benefits}] for _ in range(n)]
person, benunit = frames(units)
flags = assign_uc_claim_at_legacy_closure(
person, benunit, RATES, seed=UC_MANAGED_MIGRATION_SEED
)
rate = RATES[combination]
assert abs(flags.mean() - rate) <= 4 * np.sqrt(rate * (1 - rate) / n)


def test_deterministic_and_isolated():
units = [[{"housing_benefit_reported": 1.0}] for _ in range(500)]
person, benunit = frames(units)
np.random.seed(7)
before = np.random.get_state()[1].copy()
first = assign_uc_claim_at_legacy_closure(person, benunit, RATES, seed=1)
after = np.random.get_state()[1]
assert (before == after).all()
assert (first == assign_uc_claim_at_legacy_closure(person, benunit, RATES, 1)).all()
assert (first != assign_uc_claim_at_legacy_closure(person, benunit, RATES, 2)).any()


def _built_flags(dataset):
benunit = dataset.benunit
if "would_claim_uc_at_legacy_closure" not in benunit.columns:
pytest.skip("Dataset predates the Move to Universal Credit claim flag")
combination = legacy_benefit_combination(dataset.person, benunit)
on_uc = (
benunit["benunit_id"]
.isin(
dataset.person.loc[
dataset.person["universal_credit_reported"] > 0, "person_benunit_id"
]
)
.values
)
return benunit["would_claim_uc_at_legacy_closure"].values, combination, on_uc


@pytest.mark.parametrize("fixture", ["frs", "enhanced_frs"])
def test_built_dataset_anchors(fixture, request):
flags, combination, on_uc = _built_flags(request.getfixturevalue(fixture))
assert flags[on_uc].all()
assert flags[combination == ""].all()


def test_built_frs_cohort_shares(frs):
# The FRS build, before SPI rows and geography clones, has one row per
# surveyed benefit unit, so its draws are independent.
flags, combination, on_uc = _built_flags(frs)
drawn = ~on_uc & (combination != "")
for c in set(combination[drawn]):
in_cohort = drawn & (combination == c)
n = in_cohort.sum()
rate = RATES.get(c, RATES["all"])
if n >= 30:
share = flags[in_cohort].mean()
assert abs(share - rate) <= 4 * np.sqrt(rate * (1 - rate) / n), c
Loading
Loading