From aa604525af6e781b23ceb634c448673086f0691e Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 2 Oct 2026 22:13:19 -0400 Subject: [PATCH 1/3] Target GB universal credit as one OBR total; reach the UC payment top band MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit OBR EFO table 4.9 splits UC between spending inside the welfare cap and outside it. The outside row is DWP's UC equivalent of JSA (the Intensive Work Search group), not UC for households the benefit cap leaves alone, so computing it as UC where benefit_cap_reduction is zero asked for £12.9bn of nearly all UC. policyengine-uk cannot identify the Intensive Work Search group, so the two rows become one target, obr/universal_credit, restricted to Great Britain (the rows sit under DWP social security; Northern Ireland has its own). The GB restriction uses #490's countries field, copied verbatim. The DWP UC payment distribution's open top band ("£2500.01 or over") parsed to NaN bounds, so its four targets had a zero column. Parse it as (30,000, inf), make bands (lower, upper] so they meet without gaps, and reject overlapping summary bands. Co-Authored-By: Claude Opus 5.5 --- .../obr-uc-welfare-cap-targets.fixed.md | 1 + .../targets/build_loss_matrix.py | 19 +- .../targets/compute/__init__.py | 2 - .../targets/compute/benefits.py | 20 +-- policyengine_uk_data/targets/schema.py | 9 + policyengine_uk_data/targets/sources/obr.py | 88 +++++++--- .../tests/test_obr_universal_credit_target.py | 166 ++++++++++++++++++ .../test_uc_payment_distribution_targets.py | 150 ++++++++++++++++ policyengine_uk_data/utils/uc_data.py | 54 ++++-- pyproject.toml | 1 + uv.lock | 73 +++++++- 11 files changed, 522 insertions(+), 61 deletions(-) create mode 100644 changelog.d/obr-uc-welfare-cap-targets.fixed.md create mode 100644 policyengine_uk_data/tests/test_obr_universal_credit_target.py create mode 100644 policyengine_uk_data/tests/test_uc_payment_distribution_targets.py diff --git a/changelog.d/obr-uc-welfare-cap-targets.fixed.md b/changelog.d/obr-uc-welfare-cap-targets.fixed.md new file mode 100644 index 000000000..d2e0643ee --- /dev/null +++ b/changelog.d/obr-uc-welfare-cap-targets.fixed.md @@ -0,0 +1 @@ +Calibrate universal credit to one OBR total for Great Britain, the sum of EFO table 4.9's rows inside and outside the welfare cap. The outside-the-cap row is DWP's UC equivalent of JSA (the Intensive Work Search group), not UC for households the benefit cap leaves alone: computed that way, it asked for £12.9bn of nearly all UC (estimate £72bn, +460%), and the inside-the-cap row was compared with all UK UC. Also count the open top band of the DWP UC payment distribution ("£2,500.01 or over" a month): it parsed to missing bounds, so its four targets (1.4k to 83k households) could never be met, and the bands now meet without gaps. diff --git a/policyengine_uk_data/targets/build_loss_matrix.py b/policyengine_uk_data/targets/build_loss_matrix.py index 9552aeea2..010501a05 100644 --- a/policyengine_uk_data/targets/build_loss_matrix.py +++ b/policyengine_uk_data/targets/build_loss_matrix.py @@ -52,7 +52,6 @@ compute_two_child_limit, compute_uc_by_children, compute_uc_by_family_type, - compute_uc_outside_cap, compute_uc_payment_dist, compute_uk_population, compute_vehicles, @@ -113,7 +112,7 @@ def create_target_matrix( col = _compute_column(target, ctx, year) if col is None: continue - df[target.name] = col + df[target.name] = restrict_to_countries(col, ctx.country, target.countries) target_names.append(target.name) target_values.append(val) except Exception as e: @@ -122,6 +121,18 @@ def create_target_matrix( return df, pd.Series(target_values, index=target_names) +def restrict_to_countries(column, household_country, countries): + """Zero a household column outside the countries a target covers. + + ``countries`` of None means the target covers the whole UK, and the + column is returned unchanged. + """ + if countries is None: + return column + in_scope = np.isin(np.asarray(household_country), countries) + return np.asarray(column, dtype=float) * in_scope + + def _resolve_value(target: Target, year: int) -> float | None: """Get the target value for a year, falling back to nearest year. @@ -392,10 +403,6 @@ def _compute_column(target: Target, ctx: _SimContext, year: int) -> np.ndarray | ): return compute_ss_ni_relief(target, ctx) - # UC outside benefit cap - if name == "obr/universal_credit_outside_cap": - return compute_uc_outside_cap(target, ctx) - # Two-child limit if "two_child_limit" in name: return compute_two_child_limit(target, ctx) diff --git a/policyengine_uk_data/targets/compute/__init__.py b/policyengine_uk_data/targets/compute/__init__.py index c7000eb41..877dd8e16 100644 --- a/policyengine_uk_data/targets/compute/__init__.py +++ b/policyengine_uk_data/targets/compute/__init__.py @@ -7,7 +7,6 @@ compute_two_child_limit, compute_uc_by_children, compute_uc_by_family_type, - compute_uc_outside_cap, compute_uc_payment_dist, ) from policyengine_uk_data.targets.compute.council_tax import ( @@ -76,7 +75,6 @@ "compute_two_child_limit", "compute_uc_by_children", "compute_uc_by_family_type", - "compute_uc_outside_cap", "compute_uc_payment_dist", "compute_uk_population", "compute_vehicles", diff --git a/policyengine_uk_data/targets/compute/benefits.py b/policyengine_uk_data/targets/compute/benefits.py index 0daca8507..47126b14b 100644 --- a/policyengine_uk_data/targets/compute/benefits.py +++ b/policyengine_uk_data/targets/compute/benefits.py @@ -82,7 +82,12 @@ def ft_hh(value): def compute_uc_payment_dist(target, ctx) -> np.ndarray: - """Compute UC payment distribution band x family type.""" + """Compute UC payment distribution band x family type. + + Bands are (lower, upper] annual amounts (see + utils.uc_data.parse_monthly_award_band), so consecutive bands meet and + the open top band has an infinite upper bound. + """ name = target.name.removeprefix("dwp/uc_payment_dist/") idx = name.index("_annual_payment_") family_type = name[:idx] @@ -93,22 +98,11 @@ def compute_uc_payment_dist(target, ctx) -> np.ndarray: uc_family_type = ctx.sim.calculate("family_type", map_to="benunit").values in_band = ( - (uc_payments >= lower) & (uc_payments < upper) & (uc_family_type == family_type) + (uc_payments > lower) & (uc_payments <= upper) & (uc_family_type == family_type) ) return ctx.household_from_family(in_band) -def compute_uc_outside_cap(target, ctx) -> np.ndarray: - """Compute OBR UC outside benefit cap.""" - uc = ctx.sim.calculate("universal_credit") - uc_hh = ctx.household_from_family(uc) - cap_reduction = ctx.sim.calculate( - "benefit_cap_reduction", map_to="household" - ).values - not_capped = cap_reduction == 0 - return uc_hh * not_capped - - def compute_two_child_limit(target, ctx) -> np.ndarray | None: """Compute two-child limit targets.""" name = target.name diff --git a/policyengine_uk_data/targets/schema.py b/policyengine_uk_data/targets/schema.py index 97b814678..0a678a228 100644 --- a/policyengine_uk_data/targets/schema.py +++ b/policyengine_uk_data/targets/schema.py @@ -19,6 +19,11 @@ class Unit(str, Enum): RATE = "rate" +# DWP statistics cover Great Britain: benefits for Northern Ireland residents +# are the Northern Ireland Executive's responsibility. +GREAT_BRITAIN = ("ENGLAND", "SCOTLAND", "WALES") + + class Target(BaseModel): """A single calibration target from an official statistical source. @@ -41,6 +46,10 @@ class Target(BaseModel): is_count: bool = False reference_url: str | None = None forecast_vintage: str | None = None + # Countries a national source covers when that is less than the UK, as + # values of the model's `country` variable. The loss matrix column only + # counts households in these countries. None means the whole UK. + countries: tuple[str, ...] | None = None # For targets needing custom simulation logic (UC splits, # counterfactuals). Excluded from serialisation. diff --git a/policyengine_uk_data/targets/sources/obr.py b/policyengine_uk_data/targets/sources/obr.py index ad4e3cf17..1207aa435 100644 --- a/policyengine_uk_data/targets/sources/obr.py +++ b/policyengine_uk_data/targets/sources/obr.py @@ -17,7 +17,7 @@ import openpyxl import requests -from policyengine_uk_data.targets.schema import Target, Unit +from policyengine_uk_data.targets.schema import GREAT_BRITAIN, Target, Unit from policyengine_uk_data.targets.sources._common import ( HEADERS, load_config, @@ -455,6 +455,31 @@ def _parse_nics(wb: openpyxl.Workbook) -> list[Target]: return targets +def _universal_credit_rows(ws, max_row: int = 55) -> tuple[int, int]: + """Table 4.9's universal credit rows inside and outside the welfare cap. + + Each section has exactly one row starting "Universal credit"; the + outside-the-cap section starts at the row headed "Welfare spending + outside the welfare cap". + """ + boundary = _find_row( + ws, "Welfare spending outside the welfare cap", max_row=max_row + ) + rows = [ + row + for row in range(1, max_row + 1) + if str(ws[f"B{row}"].value or "").strip().startswith("Universal credit") + ] + inside = [row for row in rows if row < boundary] + outside = [row for row in rows if row > boundary] + if len(inside) != 1 or len(outside) != 1: + raise ValueError( + f"expected one universal credit row each side of row {boundary}, " + f"found rows {rows}" + ) + return inside[0], outside[0] + + def _parse_welfare(wb: openpyxl.Workbook) -> list[Target]: """Parse Table 4.9 (welfare spending) from expenditure xlsx.""" config = load_config() @@ -506,10 +531,6 @@ def read_49(row_num: int) -> dict[int, float]: "Winter fuel payment", "winter_fuel_allowance", ), - "universal_credit_in_cap": ( - "Universal credit", - "universal_credit", - ), "child_benefit": ("Child benefit", "child_benefit"), "state_pension": ("State pension", "state_pension"), "jobseekers_allowance": ( @@ -539,30 +560,41 @@ def read_49(row_num: int) -> dict[int, float]: except ValueError: logger.warning("OBR welfare: row '%s' not found", label) - # Universal credit outside cap (row 43) is jobseekers UC + # Universal credit has two rows: spending inside the welfare cap (row 18 + # of the March 2026 table) and outside it (row 43). The welfare cap is the + # Charter for Budget Responsibility's limit on welfare spending, not the + # household benefit cap. It excludes the State Pension and the payments + # most sensitive to the economic cycle: JSA, associated Housing Benefit + # and their UC equivalent. DWP's benefit expenditure and caseload tables + # (Spring Forecast 2026) give the same £12.876bn for 2025-26 and define + # it (Notes) as "the total expenditure ... directed to those in the + # Intensive Work Search group"; EFO March 2026 para 4.24 n.20 uses the + # same regime. policyengine-uk has no UC conditionality regime. A proxy + # built from the legal tests and FRS employment status went from 5% under + # to 27% over DWP's 2025-26 figure when one employment-status code was + # mapped correctly, too fragile to calibrate to, so the two rows are + # targeted as one total. Both sit under "DWP social security", + # which covers Great Britain: Northern Ireland's UC is in the "NI social + # security" rows. try: - # UC outside cap = predominantly JSA-conditionality UC - uc_outside_row = _find_row(ws, "Universal credit", col="B", max_row=55) - # Find the second UC row (outside cap section) - for row in range(uc_outside_row + 1, 55): - cell_val = ws[f"B{row}"].value - if cell_val and str(cell_val).strip().startswith("Universal credit"): - values = read_49(row) - if values: - targets.append( - Target( - name="obr/universal_credit_outside_cap", - variable="universal_credit", - source="obr", - unit=Unit.GBP, - values=values, - reference_url=ref, - forecast_vintage=vintage, - ) - ) - break - except ValueError: - logger.warning("OBR welfare: UC outside cap not found") + inside_row, outside_row = _universal_credit_rows(ws) + inside, outside = read_49(inside_row), read_49(outside_row) + if not inside or inside.keys() != outside.keys(): + raise ValueError("the two universal credit rows cover different years") + targets.append( + Target( + name="obr/universal_credit", + variable="universal_credit", + source="obr", + unit=Unit.GBP, + values={year: inside[year] + outside[year] for year in inside}, + countries=GREAT_BRITAIN, + reference_url=ref, + forecast_vintage=vintage, + ) + ) + except ValueError as e: + logger.warning("OBR welfare: universal credit not parsed: %s", e) return targets diff --git a/policyengine_uk_data/tests/test_obr_universal_credit_target.py b/policyengine_uk_data/tests/test_obr_universal_credit_target.py new file mode 100644 index 000000000..d09f36353 --- /dev/null +++ b/policyengine_uk_data/tests/test_obr_universal_credit_target.py @@ -0,0 +1,166 @@ +"""Tests for the OBR universal credit target. + +OBR EFO table 4.9 splits universal credit between spending inside the welfare +cap and outside it (the Intensive Work Search group, DWP's UC equivalent of +JSA). The split is not the household benefit cap, and policyengine-uk cannot +identify the Intensive Work Search group, so the two rows are one target: +GB universal credit, the sum of both rows. +""" + +from types import SimpleNamespace + +import numpy as np +import openpyxl +import pytest +from hypothesis import given, settings +from hypothesis import strategies as st + +from policyengine_uk_data.storage import STORAGE_FOLDER +from policyengine_uk_data.targets.build_loss_matrix import ( + _compute_column, + restrict_to_countries, +) +from policyengine_uk_data.targets.schema import GREAT_BRITAIN +from policyengine_uk_data.targets.sources import obr + +COUNTRIES = ("ENGLAND", "SCOTLAND", "WALES", "NORTHERN_IRELAND") +YEAR_COLUMNS = dict(zip("CDEFGHI", range(2024, 2031))) + + +def _committed_table() -> openpyxl.Workbook: + return openpyxl.load_workbook(STORAGE_FOLDER / "obr_efo" / "efo_expenditure.xlsx") + + +def _uc_targets(wb) -> dict: + return {t.name: t for t in obr._parse_welfare(wb) if "universal_credit" in t.name} + + +def test_universal_credit_is_one_gb_target(): + targets = _uc_targets(_committed_table()) + assert list(targets) == ["obr/universal_credit"] + target = targets["obr/universal_credit"] + assert target.variable == "universal_credit" + assert target.countries == GREAT_BRITAIN + # March 2026 EFO table 4.9, 2025-26: £66.411bn inside the welfare cap + # (row 18) plus £12.876bn outside it (row 43). + assert target.values[2025] == pytest.approx(79.287e9, abs=1e6) + + +def test_target_is_the_sum_of_both_rows_in_every_year(): + wb = _committed_table() + ws = wb["4.9"] + rows = [ + row + for row in range(1, 56) + if str(ws[f"B{row}"].value or "").startswith("Universal credit") + ] + assert rows == [18, 43] + target = _uc_targets(wb)["obr/universal_credit"] + assert sorted(target.values) == list(YEAR_COLUMNS.values()) + for column, year in YEAR_COLUMNS.items(): + expected = sum(ws[f"{column}{row}"].value for row in rows) * 1e9 + assert target.values[year] == pytest.approx(expected) + + +def _table(rows: list[tuple]) -> openpyxl.Workbook: + """A minimal sheet 4.9: (label, value per year) in column B onwards.""" + wb = openpyxl.Workbook() + ws = wb.active + ws.title = "4.9" + for i, (label, value) in enumerate(rows, start=6): + ws[f"B{i}"] = label + if value is not None: + for column in YEAR_COLUMNS: + ws[f"{column}{i}"] = value + return wb + + +def test_rows_are_found_by_section_not_position(): + wb = _table( + [ + ("Welfare cap", None), + ("Pension credit", 6.0), + ("Universal credit", 60.0), + ("Child benefit", 13.0), + ("Welfare spending outside the welfare cap", None), + ("State pension", 140.0), + ("Universal credit", 10.0), + ] + ) + target = _uc_targets(wb)["obr/universal_credit"] + assert set(target.values.values()) == {70e9} + + +@pytest.mark.parametrize( + "rows", + [ + # No outside-the-cap row: a partial total would understate UC. + [ + ("Universal credit", 60.0), + ("Welfare spending outside the welfare cap", None), + ("State pension", 140.0), + ], + # No section heading: the rows cannot be told apart. + [("Universal credit", 60.0), ("Universal credit", 10.0)], + # Two rows in one section. + [ + ("Universal credit", 60.0), + ("Universal credit", 1.0), + ("Welfare spending outside the welfare cap", None), + ("Universal credit", 10.0), + ], + ], +) +def test_no_target_unless_both_rows_are_unambiguous(rows): + assert _uc_targets(_table(rows)) == {} + + +def _fake_ctx(uc, benunit_household, household_country): + n_households = len(household_country) + variables = { + "universal_credit": SimpleNamespace(entity=SimpleNamespace(key="benunit")) + } + return SimpleNamespace( + sim=SimpleNamespace( + tax_benefit_system=SimpleNamespace(variables=variables), + calculate=lambda variable, *args, **kwargs: uc, + ), + household_from_family=lambda values: np.bincount( + benunit_household, weights=np.asarray(values, float), minlength=n_households + ), + country=household_country, + ) + + +@settings(max_examples=200, deadline=None) +@given( + st.lists(st.sampled_from(COUNTRIES), min_size=1, max_size=20).flatmap( + lambda countries: st.tuples( + st.just(np.array(countries)), + st.lists( + st.tuples( + st.integers(0, len(countries) - 1), + st.floats(0, 1e5, allow_nan=False), + ), + max_size=40, + ), + ) + ) +) +def test_column_counts_gb_universal_credit_once(case): + """Property: the column sums UC over benefit units in GB households, each + once, and is zero for every Northern Ireland household.""" + household_country, benunits = case + benunit_household = np.array([h for h, _ in benunits], dtype=int) + uc = np.array([amount for _, amount in benunits], dtype=float) + ctx = _fake_ctx(uc, benunit_household, household_country) + target = _uc_targets(_committed_table())["obr/universal_credit"] + + column = restrict_to_countries( + _compute_column(target, ctx, 2025), ctx.country, target.countries + ) + + in_gb = household_country[benunit_household] != "NORTHERN_IRELAND" + assert column.sum() == pytest.approx(uc[in_gb].sum()) + assert (column[household_country == "NORTHERN_IRELAND"] == 0).all() + assert (column >= 0).all() diff --git a/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py b/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py new file mode 100644 index 000000000..bccb3290e --- /dev/null +++ b/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py @@ -0,0 +1,150 @@ +"""Tests for the DWP UC payment distribution targets. + +The Stat-Xplore extract (storage/uc_national_payment_dist.xlsx, May 2025) +counts households on UC by monthly award band and family type. Its top band, +'£2500.01 or over', is open-ended; it used to parse to NaN bounds, so its four +targets (1.4k to 83k households) had a column of zeros and could never be met. +""" + +from types import SimpleNamespace + +import numpy as np +import pandas as pd +import pytest +from hypothesis import given, settings +from hypothesis import strategies as st + +from policyengine_uk_data.storage import STORAGE_FOLDER +from policyengine_uk_data.targets.compute.benefits import compute_uc_payment_dist +from policyengine_uk_data.targets.sources.dwp import _uc_payment_distribution_targets +from policyengine_uk_data.utils.uc_data import ( + _check_bands_disjoint, + parse_monthly_award_band, + uc_national_payment_dist, +) + +FAMILY_TYPES = { + "Single, no children": "SINGLE", + "Single, with children": "LONE_PARENT", + "Couple, no children": "COUPLE_NO_CHILDREN", + "Couple, with children": "COUPLE_WITH_CHILDREN", +} + + +def _raw_extract() -> pd.DataFrame: + """Household counts indexed by award band, one column per family type.""" + raw = pd.read_excel(STORAGE_FOLDER / "uc_national_payment_dist.xlsx", header=None) + counts = raw.iloc[9:, 3:7] + counts.index = raw.iloc[9:, 1] + counts.columns = raw.iloc[7, 3:7] + return counts + + +def test_parse_monthly_award_band(): + assert parse_monthly_award_band("£0.01 to £100.00") == (0, 1_200) + assert parse_monthly_award_band("£100.01 to £200.00") == (1_200, 2_400) + assert parse_monthly_award_band("£1000.01 to £1100.00") == (12_000, 13_200) + assert parse_monthly_award_band("£2,500.01 or over") == (30_000, np.inf) + with pytest.raises(ValueError): + parse_monthly_award_band("No payment") + + +def test_top_band_targets_are_reachable(): + targets = {t.name: t for t in _uc_payment_distribution_targets()} + top = _raw_extract().loc["£2500.01 or over"] + for label, family_type in FAMILY_TYPES.items(): + target = targets[ + f"dwp/uc_payment_dist/{family_type}_annual_payment_30_000_to_inf" + ] + assert target.lower_bound == 30_000 + assert target.upper_bound == np.inf + assert target.values[2025] == top[label] + for target in targets.values(): + assert not np.isnan(target.lower_bound) + assert not np.isnan(target.upper_bound) + assert "nan" not in target.name + + +def test_band_counts_sum_to_households_with_a_payment(): + """Conservation: the bands hold every household with a payment, once. + + Stat-Xplore perturbs each cell for disclosure control, so a total can + differ from the sum of its cells by a few households. + """ + raw = _raw_extract() + for label, family_type in FAMILY_TYPES.items(): + with_payment = raw.loc["Total", label] - raw.loc["No payment", label] + parsed = uc_national_payment_dist[ + uc_national_payment_dist.family_type == family_type + ] + assert abs(parsed.household_count.sum() - with_payment) <= 10, label + + +def test_bands_tile_the_positive_awards(): + for family_type, bands in uc_national_payment_dist.groupby("family_type"): + bands = bands.sort_values("uc_annual_payment_min") + lower = bands.uc_annual_payment_min.to_numpy() + upper = bands.uc_annual_payment_max.to_numpy() + assert lower[0] == 0, family_type + assert np.array_equal(lower[1:], upper[:-1]), family_type + assert upper[-1] == np.inf, family_type + + +def test_overlapping_summary_band_is_rejected(): + bands = pd.DataFrame( + { + "family_type": ["SINGLE"] * 3, + "uc_annual_payment_min": [18_000.0, 19_200.0, 18_000.0], + "uc_annual_payment_max": [19_200.0, 20_400.0, np.inf], + } + ) + with pytest.raises(ValueError): + _check_bands_disjoint(bands) + + +def _fake_ctx(uc, family_type, household): + def calculate(variable, map_to=None): + values = {"universal_credit": uc, "family_type": family_type}[variable] + return SimpleNamespace(values=values) + + n_households = household.max() + 1 + return SimpleNamespace( + sim=SimpleNamespace(calculate=calculate), + household_from_family=lambda values: np.bincount( + household, weights=np.asarray(values, float), minlength=n_households + ), + ) + + +_TARGETS = _uc_payment_distribution_targets() +_EDGES = sorted({t.upper_bound for t in _TARGETS if np.isfinite(t.upper_bound)}) +_AWARDS = st.one_of( + st.just(0.0), # no payment + st.floats(0, 1e6, allow_nan=False), # anywhere, including the open top band + st.sampled_from(_EDGES), # exactly on a band edge + st.sampled_from(_EDGES).map(lambda edge: np.nextafter(edge, np.inf)), + st.sampled_from(_EDGES).map(lambda edge: np.nextafter(edge, -np.inf)), +) + + +@settings(max_examples=200, deadline=None) +@given( + st.lists( + st.tuples( + _AWARDS, st.sampled_from(list(FAMILY_TYPES.values())), st.integers(0, 9) + ), + min_size=1, + max_size=60, + ) +) +def test_every_award_lands_in_exactly_one_band(benefit_units): + """Property: summed over bands, each household's column counts its benefit + units with a positive award, and never one without.""" + uc, family_type, household = (np.array(column) for column in zip(*benefit_units)) + ctx = _fake_ctx(uc.astype(float), family_type, household) + + total = sum(compute_uc_payment_dist(t, ctx) for t in _TARGETS) + expected = np.bincount( + household, weights=(uc > 0).astype(float), minlength=len(total) + ) + np.testing.assert_array_equal(total, expected) diff --git a/policyengine_uk_data/utils/uc_data.py b/policyengine_uk_data/utils/uc_data.py index 7bacebc57..3833d138d 100644 --- a/policyengine_uk_data/utils/uc_data.py +++ b/policyengine_uk_data/utils/uc_data.py @@ -1,7 +1,46 @@ +import numpy as np import pandas as pd from pathlib import Path +def parse_monthly_award_band(band: str) -> tuple[float, float]: + """Annual (lower, upper] payment bounds of a Stat-Xplore monthly award band. + + Awards are whole pence, so the band '£100.01 to £200.00' holds monthly + awards over £100.00 and up to £200.00: annual bounds (1,200, 2,400]. The + lower bound is the previous band's top, so consecutive bands meet with no + gap. The open top band '£2500.01 or over' is (30,000, inf). + """ + text = band.replace("£", "").replace(",", "").strip() + if text.endswith(" or over"): + lower = float(text.removesuffix(" or over")) + return round((lower - 0.01) * 12, 2), np.inf + parts = text.split(" to ") + if len(parts) != 2: + raise ValueError(f"Unrecognised UC monthly award band: {band!r}") + lower, upper = (float(part) for part in parts) + return round((lower - 0.01) * 12, 2), upper * 12 + + +def _check_bands_disjoint(bands: pd.DataFrame) -> None: + """Fail if any family type's payment bands overlap. + + Stat-Xplore also lists summary bands ('£1500.01 or over') that span the + finer bands below them. They are suppressed ('..') in the committed + extract; if a new extract filled one in, counting it as well would double + count those households. + """ + for family_type, group in bands.groupby("family_type"): + group = group.sort_values("uc_annual_payment_min") + lower = group.uc_annual_payment_min.to_numpy() + upper = group.uc_annual_payment_max.to_numpy() + if (lower[1:] < upper[:-1]).any(): + raise ValueError( + f"UC payment bands for {family_type} overlap: check for " + "summary bands spanning finer ones" + ) + + def _parse_uc_national_payment_dist(): """Parse UC national payment distribution into long format.""" storage_path = Path(__file__).parent.parent / "storage" @@ -44,17 +83,10 @@ def _parse_uc_national_payment_dist(): result_df = pd.DataFrame(data_rows) - # Parse monthly band into min and max, then convert to annual - def parse_band(band): - """Parse band like '£100.01 to £200.00' into (min, max).""" - parts = band.replace("£", "").replace(",", "").split(" to ") - if len(parts) == 2: - return float(parts[0]) * 12, float(parts[1]) * 12 - return None, None - - result_df[["uc_annual_payment_min", "uc_annual_payment_max"]] = result_df[ - "monthly_award_band" - ].apply(lambda x: pd.Series(parse_band(x))) + result_df[["uc_annual_payment_min", "uc_annual_payment_max"]] = [ + parse_monthly_award_band(band) for band in result_df["monthly_award_band"] + ] + _check_bands_disjoint(result_df) # Map family types to constant names family_type_mapping = { diff --git a/pyproject.toml b/pyproject.toml index beff7f1a4..58de5b1d3 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -38,6 +38,7 @@ dependencies = [ dev = [ "ruff>=0.9.0", "pytest", + "hypothesis", "torch", "l0-python>=0.4.0", "tables", diff --git a/uv.lock b/uv.lock index a3e9f44d9..c301c6c9f 100644 --- a/uv.lock +++ b/uv.lock @@ -577,6 +577,75 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/35/f4/124858007ddf3c61e9b144107304c9152fa80b5b6c168da07d86fe583cc1/huggingface_hub-1.1.5-py3-none-any.whl", hash = "sha256:e88ecc129011f37b868586bbcfae6c56868cae80cd56a79d61575426a3aa0d7d", size = 516000, upload-time = "2025-11-20T15:49:30.926Z" }, ] +[[package]] +name = "hypothesis" +version = "6.168.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "sortedcontainers" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/09/b7/13118bbc45d6d8b9d04e2de779e2a4ff23145ea39efa692b994fb874ca72/hypothesis-6.168.3.tar.gz", hash = "sha256:a43388f9067678fef6e13bdff325b6cfa6961a590498bb37f7ff31589c83bc75", size = 511022, upload-time = "2026-09-28T05:20:58.499Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2b/cd/746a2fc1e5e5ba34f07ea6ee1ad07525a3dde5ff4836db3a7980af49a891/hypothesis-6.168.3-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:d20972ca134e652e9928ecd200966a8a2857adc2812c1d74b12a872e5cd503eb", size = 791536, upload-time = "2026-09-28T05:18:57.162Z" }, + { url = "https://files.pythonhosted.org/packages/c3/0f/7a158e377b69556c8e12c25c8fd811d0103e54c24a12dab1f3b9819f202e/hypothesis-6.168.3-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:eafbec09d3e87d13d8242411d1f5868b5e879f1bd6d95e233e9ff80c527bea1c", size = 787313, upload-time = "2026-09-28T05:20:18.122Z" }, + { url = "https://files.pythonhosted.org/packages/45/c6/7df9104ee359e8fcd77791781c47e70794964ca69673665e0de2a6590f4c/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:73c5627497968cc62d140e9fde1ea21e12cac7b64dc419fea8286cd34ff1ad4e", size = 1124036, upload-time = "2026-09-28T05:20:04.372Z" }, + { url = "https://files.pythonhosted.org/packages/92/0f/8a61715404a73b9a82cb1f976803428d603cdea976a9c30266c72ad6a7ce/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:29dc56e6dc6eeb0aeb08ac463f847279bde2336c4786faf7ee0af4b93cd031a7", size = 1147875, upload-time = "2026-09-28T05:19:15.495Z" }, + { url = "https://files.pythonhosted.org/packages/e4/3f/2b16e95cc3b9a069afa23830a012aefe377ce4a457a227e5a5026eb827ff/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:753bb501f8560d2e321ed3b62596b4496c56f15668b0a0669231ccc3e6c4e80d", size = 1149489, upload-time = "2026-09-28T05:19:18.886Z" }, + { url = "https://files.pythonhosted.org/packages/f6/e7/d7cf6dc068bb02732b2a6e38a35a0dcb7f2137608198fc79d740f69d197c/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:71ab606c472449cb872ef2a7acaec679ee4f651b7f0a477a04ebc85d8933ef8c", size = 1191992, upload-time = "2026-09-28T05:19:21.83Z" }, + { url = "https://files.pythonhosted.org/packages/5f/26/f1e5b25dec15998e8375d722825ec3dae1075cbe7cb1f74221b2312a2e33/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:26928956c54748e4dfa587333741ab750246123b93484955646ff316cca27eef", size = 1169911, upload-time = "2026-09-28T05:19:27.996Z" }, + { url = "https://files.pythonhosted.org/packages/03/36/901234a49147e6bf8a5b9e54b775706a223c2b52befea1106ea2aaf15d48/hypothesis-6.168.3-cp310-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:0bdfc53c041b61c854fa3761735736b991bc2bddafee45a361cfe4d43027c1ef", size = 1129406, upload-time = "2026-09-28T05:18:42.632Z" }, + { url = "https://files.pythonhosted.org/packages/55/c1/5fc9ff91ec9fba6bb527833386c66812524b5cc75fe7d4a4f9122ee1c420/hypothesis-6.168.3-cp310-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:650528e2b1e2a45e95c2df624d4b9364b8ac5a027ba84e1eea2bc5398dd01dcd", size = 1160399, upload-time = "2026-09-28T05:20:23.911Z" }, + { url = "https://files.pythonhosted.org/packages/9b/d2/49ad1ef5c55547c8abfd0fc571b3493806dda530fd5cb3d8263dbad86a72/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:209dc54cdb1b4d6d7020ad8d09e44c49756361b7183adacea4d6f5705085b595", size = 1299906, upload-time = "2026-09-28T05:18:58.417Z" }, + { url = "https://files.pythonhosted.org/packages/0d/5f/f98a26094c4f462c4215807bed0f9a6b5508b27a3c329bcba1bc8adc7ac7/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:af8ca98cd11dc7f9427bb90483abcd9688d4df2e8e63893a30a2423b027ebb11", size = 1425538, upload-time = "2026-09-28T05:20:49.074Z" }, + { url = "https://files.pythonhosted.org/packages/d0/b8/c5b4406e3a6591e51a42f85469351d1d824e0a54b84167aec9b2353759fe/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:f8122cfdd0bc0ba843063effa22ac310bdebdb3a1319bf1e63f14baa119c72ab", size = 1377084, upload-time = "2026-09-28T05:20:06.618Z" }, + { url = "https://files.pythonhosted.org/packages/e0/01/0d84a8ea469d024c15f602799c0b8410fbe03d99fdb0370449b33c3c916f/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:4bedbb379eab34f792af7ee9a05aae04e9c08bbb52e0f90f5a2110d4fe4b2fbd", size = 1281162, upload-time = "2026-09-28T05:19:42.768Z" }, + { url = "https://files.pythonhosted.org/packages/22/b5/ea6435038de1a795f005cb603531e8ccc053febc50180f3ab55583c9c60e/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:5a953115b9f5c95133ab2d04efffeec96e5658c3207923dca7285f7db3e6bbef", size = 1300433, upload-time = "2026-09-28T05:18:55.634Z" }, + { url = "https://files.pythonhosted.org/packages/6c/78/fa6c77d9f64b6d592e69e582d83cbce7d7e56897faa4762885a2123b0166/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:6173558e676ad25ed1e20507fa4024a0d816dd90f715f77006e5a28f19109026", size = 1336273, upload-time = "2026-09-28T05:19:08.273Z" }, + { url = "https://files.pythonhosted.org/packages/f1/56/4acd0d778bab818c9ea336fcba768bf43bc70db7f84cc373579298df4641/hypothesis-6.168.3-cp310-abi3-win32.whl", hash = "sha256:dc66390fb12d80585aa9222bf538ce8b7aa22cf5d118250647355a1c9e8f62f4", size = 678211, upload-time = "2026-09-28T05:19:23.355Z" }, + { url = "https://files.pythonhosted.org/packages/b0/db/0a02b146ad1c30f9716362f2e1a379dfd6b0f7be03ac1e22c9da24c9e2a9/hypothesis-6.168.3-cp310-abi3-win_amd64.whl", hash = "sha256:92325b276360fe86c5bf71a568c0d53a6d140b0de36dcf029f9164a17803bb24", size = 684906, upload-time = "2026-09-28T05:18:35.099Z" }, + { url = "https://files.pythonhosted.org/packages/a3/c0/958deaf726848f96f52250740bf39f13f476b068e22f73e2b358eaa07532/hypothesis-6.168.3-cp310-abi3-win_arm64.whl", hash = "sha256:3cf6f1eeaf41cd8d60cf1f88fde905ca1dd77c906929a507d6ac7f66f2ccba2a", size = 683292, upload-time = "2026-09-28T05:19:57.404Z" }, + { url = "https://files.pythonhosted.org/packages/0c/a2/6787da846d929e52fc3344d803299c45782dbeac287528af08380e984bf8/hypothesis-6.168.3-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:1b230a850de63334c16654a34a2d547e0179d36b9071d4439b3e7237f6d077e7", size = 793254, upload-time = "2026-09-28T05:19:36.294Z" }, + { url = "https://files.pythonhosted.org/packages/89/ba/2893f5ca42501f4d562cba3229c8994c3a3e5fac66ef60c1e0e0b5220a5d/hypothesis-6.168.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d63b0226cd3e0d8bdd97c3384b22a21934ed4d53246c1c93575dada616672499", size = 784699, upload-time = "2026-09-28T05:18:54.194Z" }, + { url = "https://files.pythonhosted.org/packages/b7/aa/7d7349daf75b71f6f35876f8de115e974c5c04d600e86df1b2779b0883ab/hypothesis-6.168.3-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0369f5df055f96e117ab12e5f249668ff144731ab5280bc7a205fdf81f990b89", size = 1123016, upload-time = "2026-09-28T05:20:36.372Z" }, + { url = "https://files.pythonhosted.org/packages/00/f0/7774e1ea072708ea46cb25c4aeee5f9978b6c271764809069e8a6789858e/hypothesis-6.168.3-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f076bcd0f77fdcdb7797826c879099573d03a02e228ebe649ea917081b962ac0", size = 1168989, upload-time = "2026-09-28T05:20:08.44Z" }, + { url = "https://files.pythonhosted.org/packages/a7/7d/113992abed9efbd7944e3a496a382da58152dce126fddf6419ff31a0a57d/hypothesis-6.168.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:b823ba1fcec8da730f29316d010b06d3f7e0c3828dcf91020e24c55e7d24652a", size = 1298626, upload-time = "2026-09-28T05:19:40.986Z" }, + { url = "https://files.pythonhosted.org/packages/d5/65/3659fa5e733027e5b37a27486e57f4053535e40853d23501c422bb275043/hypothesis-6.168.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:dd2849c269d674e4618590f3b48d443bd4c06b5aef3d8d869086c2e6d213d248", size = 1335184, upload-time = "2026-09-28T05:20:20.087Z" }, + { url = "https://files.pythonhosted.org/packages/b4/6c/35aab2221b5ea65125340f236e1a765a9e9f3b28d89a389ea0033fed29f7/hypothesis-6.168.3-cp313-cp313-win_amd64.whl", hash = "sha256:3ef7d26f5789e691401d5f87eafed9bd2763f0dbe47af6d6b66012a509404766", size = 682173, upload-time = "2026-09-28T05:19:12.762Z" }, + { url = "https://files.pythonhosted.org/packages/6e/79/27b0cb56ff5d2bc92458bf6fcebdb1f0dc01f57f043a178e9f9d74498169/hypothesis-6.168.3-cp314-cp314-macosx_10_12_x86_64.whl", hash = "sha256:e2d4c68729a13df9af4998d2652cfb5d541c5609b88c306880a5dfb284938ec2", size = 793287, upload-time = "2026-09-28T05:20:38.522Z" }, + { url = "https://files.pythonhosted.org/packages/e8/bd/5342f95c3bc36586777ef8cb8eba87ac7acf0184210327fb0e6d8654e295/hypothesis-6.168.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:b345f818083ec99966a43ca4f7b38feb62c6920ce28572bc2bde4948a8b7eaba", size = 784857, upload-time = "2026-09-28T05:20:42.609Z" }, + { url = "https://files.pythonhosted.org/packages/4a/5f/fc774241518a0588d362680a3084275058aab86d6bc02a2c61e1073142d0/hypothesis-6.168.3-cp314-cp314-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0608c610fc002978fc5de0f471770d8817e9455cb8649a47983c9f8bccfc1897", size = 1123330, upload-time = "2026-09-28T05:19:29.735Z" }, + { url = "https://files.pythonhosted.org/packages/b0/e6/131f16775a3dca5f4fe27f0d6ad9a9851600098a54a872d163f6ebf0b664/hypothesis-6.168.3-cp314-cp314-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9dcf6448b1ecc37f2b23f2d1b3ddfc9ff6b6910614c15a642dc82f419b96037b", size = 1169178, upload-time = "2026-09-28T05:19:53.436Z" }, + { url = "https://files.pythonhosted.org/packages/06/61/c9f5bff8b73321c9fd12f2d69666c6e4861b97b60f5f24fdd6ac293007fd/hypothesis-6.168.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:ae4f9f094041dcce5119ebd7bab71062743ab02b6e654d056b370beda78c19e2", size = 1299127, upload-time = "2026-09-28T05:20:14.486Z" }, + { url = "https://files.pythonhosted.org/packages/e4/6d/d4618f8ab12dd76c4d58ea052252759aa2bf0c4e71f4b502ba5d577e0b9f/hypothesis-6.168.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:d479985fe73af97badfb72dc6d20c6a353e486a36f6d065f600026e0cf954a86", size = 1335390, upload-time = "2026-09-28T05:20:10.326Z" }, + { url = "https://files.pythonhosted.org/packages/95/b6/727161e17cc297df1aeea945de05c532033b66751625bec192bc5b5ff707/hypothesis-6.168.3-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:785e2c45f8c08e274b4bf1ccae97f1a4e09407e9a68e80790cfab1d17a4a45fa", size = 624291, upload-time = "2026-09-28T05:18:36.376Z" }, + { url = "https://files.pythonhosted.org/packages/77/23/2f9b506392a0b8712440bcb781abc5713be9ffac1303b6d10f3669050882/hypothesis-6.168.3-cp314-cp314-win_amd64.whl", hash = "sha256:320920b1e3dae8611eee8a03d063cf2187446f2a17c38cfb8a7fc1466f71eee2", size = 682064, upload-time = "2026-09-28T05:20:16.336Z" }, + { url = "https://files.pythonhosted.org/packages/50/93/efabfd95eb2b69c0c9fa1e1c83c8290aa2715d46baeb5a7764faf09c133a/hypothesis-6.168.3-cp314-cp314t-macosx_10_12_x86_64.whl", hash = "sha256:35380baa981108a7f60c4eab71e46acd8d8f58440520346a1a6aba06dca7e074", size = 791875, upload-time = "2026-09-28T05:19:37.789Z" }, + { url = "https://files.pythonhosted.org/packages/1b/fc/2a0ada1623a9a33048bbf9f00a7c92513398c6b8865cba5efb854262ce30/hypothesis-6.168.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:f0aaaed00438fa6673d856aaee12a4a8afbde82a9f3c5ff487cacfb064239afe", size = 783412, upload-time = "2026-09-28T05:18:48.266Z" }, + { url = "https://files.pythonhosted.org/packages/41/31/73c615d37eeb12da0209555612c32ced628d860267c23cfd2691c6207aea/hypothesis-6.168.3-cp314-cp314t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:f27e6df1576bf838e7f484d4ae5cba92114497ca30afb917056bf1ea33e95675", size = 1121662, upload-time = "2026-09-28T05:19:26.225Z" }, + { url = "https://files.pythonhosted.org/packages/92/54/1893344a3b8bbf83fa7c9bb7e96f6836580213854f46671cb749f7b4706c/hypothesis-6.168.3-cp314-cp314t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3bc85014577982ec6d2e266edc5cd6e7a7d3674c648791ca983c67a52f89a0c4", size = 1167761, upload-time = "2026-09-28T05:18:49.85Z" }, + { url = "https://files.pythonhosted.org/packages/26/97/a61d82968febd25a0f08ffbbfabfaeead66f68fc5fa73f3c9ece1c36fc8f/hypothesis-6.168.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:ccfc29505aa1cdcc254cb9cd701fd0811fef5480addd97d4127a00df023410f7", size = 1297309, upload-time = "2026-09-28T05:19:51.711Z" }, + { url = "https://files.pythonhosted.org/packages/bc/01/fe8e4cf6d39d02f4efadaa351427fa13722086cfc54a2da3579741c5177a/hypothesis-6.168.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:32d0699566aaa93f9e97a44705d7164386f1e91de78d277f53bada35e78bcc96", size = 1334260, upload-time = "2026-09-28T05:19:00.121Z" }, + { url = "https://files.pythonhosted.org/packages/39/6a/09177ebc62f4778f94379ecade6a51299d3d8c4f4b75ba636bf3a0202364/hypothesis-6.168.3-cp314-cp314t-win_amd64.whl", hash = "sha256:28d88fa174ecbd4ecbd7bb290f06d0db084a3971c3f511ae2830b65b5e25500f", size = 681992, upload-time = "2026-09-28T05:19:48.043Z" }, + { url = "https://files.pythonhosted.org/packages/f9/f6/890bf33d608cd63348d3146a3ca83362fbf09e589c5e5230a00c403070f4/hypothesis-6.168.3-cp315-abi3.abi3t-macosx_10_12_x86_64.whl", hash = "sha256:61f5782d807b1e6aa5c9beef1054cb2037e7d1add48a64972cd6ee447778781f", size = 791231, upload-time = "2026-09-28T05:18:43.998Z" }, + { url = "https://files.pythonhosted.org/packages/03/dd/fc36b204f8aa7437c676576d9437f0422de46aa8a729e6cd5bbb0224e759/hypothesis-6.168.3-cp315-abi3.abi3t-macosx_11_0_arm64.whl", hash = "sha256:7aacf3cf40c7ce8f9e4924d5b57bc0b068beafdfed2347bcf160a28b4cfbba7b", size = 783171, upload-time = "2026-09-28T05:19:20.353Z" }, + { url = "https://files.pythonhosted.org/packages/5c/d2/3cd087d577db67c404991d7723e64d7cd75570cdc82a0f7329aed6f2086c/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:01768a03a30dc54df7fe457c0b34016c84d598fa00d5965ededab93ba4eb3408", size = 1121152, upload-time = "2026-09-28T05:18:39.095Z" }, + { url = "https://files.pythonhosted.org/packages/42/6f/a05a68cc66e7a22701c50bf94fb9ddeb36f4f5d0de3cc130066709eda621/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:1a8a4ffc6c6e6e577f2bfbcfebf7cffbb310283ba2532c729570a0a752cebaaa", size = 1144069, upload-time = "2026-09-28T05:20:44.705Z" }, + { url = "https://files.pythonhosted.org/packages/4a/f4/fdd7a093fb2fb3a0ce24256c92ad5bf67936f4669bd06d5ceae4298150ec/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:b987d73eba95183a7e59cca6d1925c588aa4307922d9852cc2b4d282a2ae4128", size = 1146692, upload-time = "2026-09-28T05:20:25.83Z" }, + { url = "https://files.pythonhosted.org/packages/f1/c8/6dbd4377e935505ae4fc8ee4c7b18c69994c4bbaca015447c470218bdbee/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:d4569c39bd97d9573e946429ed676f3b55a7c8ed80920addd67d384a47d59b38", size = 1189480, upload-time = "2026-09-28T05:20:46.734Z" }, + { url = "https://files.pythonhosted.org/packages/ba/bf/ff288b496b690000d2686dcfa7f67855e4c5a6dddb464ed8e4880be83bb2/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:6042b8707a4b25bbbfe20b110258fb7549a68951e5525608a8ce06091039e9fc", size = 1167109, upload-time = "2026-09-28T05:19:24.758Z" }, + { url = "https://files.pythonhosted.org/packages/d9/d1/3fee2bc29fc747fabbb27e174515592e390809ca465d3bbbabbcfa3235a8/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_31_riscv64.whl", hash = "sha256:f413629de94d38a7a2ad259697a6752526143cb49c2e2d7bdba19381c80693f6", size = 1126823, upload-time = "2026-09-28T05:18:52.636Z" }, + { url = "https://files.pythonhosted.org/packages/58/6a/585294392fa6a6d9719446a042a777ba98ecca265d9e48cbedc10c88df0f/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:f071737e4e775bebba07e319eb645a880d1e0186b4d24bad23255e849d86d483", size = 1155773, upload-time = "2026-09-28T05:20:51.268Z" }, + { url = "https://files.pythonhosted.org/packages/cd/4d/fa283ff79996debf1ff08593e01f8b773eb8211b19b3d6d5d491317a65bf/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_aarch64.whl", hash = "sha256:f59a3912858d0609c26054aa1c474847c797937f5e0e1c77f0d3f08032e50f09", size = 1296671, upload-time = "2026-09-28T05:18:46.937Z" }, + { url = "https://files.pythonhosted.org/packages/42/05/98f9c2f628da5afabb6c3e8df5d2d299d2d50bf306b14a6e26d3954c4951/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_armv7l.whl", hash = "sha256:f09c05a23a8025dd5cad07c2ed299a46778e72d49ff0dd5bf353bcc0f7dc1580", size = 1422001, upload-time = "2026-09-28T05:19:04.188Z" }, + { url = "https://files.pythonhosted.org/packages/b8/c9/f2109ade29a7ec284d55bca328d784743e8b13e321576fd34ee7e65f99af/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_i686.whl", hash = "sha256:44ace770bda3a0301739fc5a413c764de790c739df1f1a7218dd049f8594d9f5", size = 1374182, upload-time = "2026-09-28T05:20:27.764Z" }, + { url = "https://files.pythonhosted.org/packages/ce/6d/e05d5f72441564a3bebc71fa155deafd0fd3b015d6014ec8a00edf6e42bb/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_ppc64le.whl", hash = "sha256:03b131043608f94a2578a2a896a1a72092acb7079815eef5b8513b08447e0e62", size = 1278402, upload-time = "2026-09-28T05:20:56.332Z" }, + { url = "https://files.pythonhosted.org/packages/96/d7/6988a7f1f69c5c530687dce03a32c157e2d878e3a2e32af9269ce016b78a/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_riscv64.whl", hash = "sha256:e9784aca26eddfe99b03a0292320db742cc8a74200ef864fdce949f526f973cd", size = 1297770, upload-time = "2026-09-28T05:19:11.387Z" }, + { url = "https://files.pythonhosted.org/packages/a2/28/5bb82b60b836bd2329e1fe01ad94efc2b14dfa806f8e50ed14b795d78770/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_x86_64.whl", hash = "sha256:2b52ac363096232bebc2add117e9178d91f4248f4cdb919fd1026b5f86a4bb16", size = 1333977, upload-time = "2026-09-28T05:20:02.483Z" }, + { url = "https://files.pythonhosted.org/packages/8e/7d/d419841b8f65481ea1a50c4ba36f2670d48b8c7a51e4e8586089564cf7f6/hypothesis-6.168.3-cp315-abi3.abi3t-win32.whl", hash = "sha256:d28e3a6b511a74ce37df5274b51f36c2b274fea7365e00b16f0c81e22acd5957", size = 675394, upload-time = "2026-09-28T05:19:55.237Z" }, + { url = "https://files.pythonhosted.org/packages/d8/d2/1de6a2ad100e44817f2e9a8e8ba3eeaf4769621f4f87f2d4966c4971fcc4/hypothesis-6.168.3-cp315-abi3.abi3t-win_amd64.whl", hash = "sha256:7b9638789548361a57d984f56619ac694a914c328181d911271618409261ff4a", size = 681699, upload-time = "2026-09-28T05:19:09.957Z" }, + { url = "https://files.pythonhosted.org/packages/c0/79/be3370fd02734d6b1d950183ae9224580d19fa8d356b95348d60f1e3eeb7/hypothesis-6.168.3-cp315-abi3.abi3t-win_arm64.whl", hash = "sha256:65d78e4357ec48ed2c67825f06740ee3599be4cfe770a6092bed07108d679ac5", size = 679849, upload-time = "2026-09-28T05:19:06.884Z" }, +] + [[package]] name = "idna" version = "3.11" @@ -1364,7 +1433,7 @@ wheels = [ [[package]] name = "policyengine-uk-data" -version = "1.56.16" +version = "1.57.4" source = { editable = "." } dependencies = [ { name = "google-auth" }, @@ -1392,6 +1461,7 @@ dependencies = [ dev = [ { name = "build" }, { name = "furo" }, + { name = "hypothesis" }, { name = "itables" }, { name = "l0-python" }, { name = "pytest" }, @@ -1410,6 +1480,7 @@ requires-dist = [ { name = "google-auth" }, { name = "google-cloud-storage" }, { name = "huggingface-hub" }, + { name = "hypothesis", marker = "extra == 'dev'" }, { name = "itables", marker = "extra == 'dev'" }, { name = "l0-python", marker = "extra == 'dev'", specifier = ">=0.4.0" }, { name = "microcalibrate", specifier = ">=0.18.0" }, From 4a1b1a6cb14ad15fc817896130b7cc83b1dec8f3 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 2 Oct 2026 22:44:26 -0400 Subject: [PATCH 2/3] Keep the UC payment bands' bounds; only open the top band MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Making the bands meet without gaps ((lower, upper]) moved penny-a-month awards into the bottom band: policyengine-uk leaves 1p a month payable after deductions (SI 2013/380 Sch 6), and 44 records (136k benefit units in a seeded main build) sit there. The old lower bound of £0.12 a year leaves them out, and the seeded build of the gapless version then reweighted against them. Whether they belong in Stat-Xplore's lowest award band turns on whether its award amount is before or after deductions, which is outside this change. The bounded bands now parse and compare exactly as before; only "£2500.01 or over" changes, to [30,000.12, inf). Co-Authored-By: Claude Opus 5.5 --- .../obr-uc-welfare-cap-targets.fixed.md | 2 +- .../targets/compute/benefits.py | 8 +-- .../test_uc_payment_distribution_targets.py | 68 ++++++++++++++----- policyengine_uk_data/utils/uc_data.py | 16 ++--- 4 files changed, 63 insertions(+), 31 deletions(-) diff --git a/changelog.d/obr-uc-welfare-cap-targets.fixed.md b/changelog.d/obr-uc-welfare-cap-targets.fixed.md index d2e0643ee..945db66c7 100644 --- a/changelog.d/obr-uc-welfare-cap-targets.fixed.md +++ b/changelog.d/obr-uc-welfare-cap-targets.fixed.md @@ -1 +1 @@ -Calibrate universal credit to one OBR total for Great Britain, the sum of EFO table 4.9's rows inside and outside the welfare cap. The outside-the-cap row is DWP's UC equivalent of JSA (the Intensive Work Search group), not UC for households the benefit cap leaves alone: computed that way, it asked for £12.9bn of nearly all UC (estimate £72bn, +460%), and the inside-the-cap row was compared with all UK UC. Also count the open top band of the DWP UC payment distribution ("£2,500.01 or over" a month): it parsed to missing bounds, so its four targets (1.4k to 83k households) could never be met, and the bands now meet without gaps. +Calibrate universal credit to one OBR total for Great Britain, the sum of EFO table 4.9's rows inside and outside the welfare cap. The outside-the-cap row is DWP's UC equivalent of JSA (the Intensive Work Search group), not UC for households the benefit cap leaves alone: computed that way, it asked for £12.9bn of nearly all UC (estimate £72bn, +460%), and the inside-the-cap row was compared with all UK UC. Also count the open top band of the DWP UC payment distribution ("£2500.01 or over" a month): it parsed to missing bounds, so its four targets (1.4k to 83k households) could never be met. diff --git a/policyengine_uk_data/targets/compute/benefits.py b/policyengine_uk_data/targets/compute/benefits.py index 47126b14b..e550a78d2 100644 --- a/policyengine_uk_data/targets/compute/benefits.py +++ b/policyengine_uk_data/targets/compute/benefits.py @@ -84,9 +84,9 @@ def ft_hh(value): def compute_uc_payment_dist(target, ctx) -> np.ndarray: """Compute UC payment distribution band x family type. - Bands are (lower, upper] annual amounts (see - utils.uc_data.parse_monthly_award_band), so consecutive bands meet and - the open top band has an infinite upper bound. + Bands are [lower, upper) annual amounts (see + utils.uc_data.parse_monthly_award_band); the open top band has an + infinite upper bound. """ name = target.name.removeprefix("dwp/uc_payment_dist/") idx = name.index("_annual_payment_") @@ -98,7 +98,7 @@ def compute_uc_payment_dist(target, ctx) -> np.ndarray: uc_family_type = ctx.sim.calculate("family_type", map_to="benunit").values in_band = ( - (uc_payments > lower) & (uc_payments <= upper) & (uc_family_type == family_type) + (uc_payments >= lower) & (uc_payments < upper) & (uc_family_type == family_type) ) return ctx.household_from_family(in_band) diff --git a/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py b/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py index bccb3290e..7bfc758d1 100644 --- a/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py +++ b/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py @@ -4,6 +4,7 @@ counts households on UC by monthly award band and family type. Its top band, '£2500.01 or over', is open-ended; it used to parse to NaN bounds, so its four targets (1.4k to 83k households) had a column of zeros and could never be met. +The other bands keep their bounds: [monthly minimum, monthly maximum) x 12. """ from types import SimpleNamespace @@ -41,10 +42,14 @@ def _raw_extract() -> pd.DataFrame: def test_parse_monthly_award_band(): - assert parse_monthly_award_band("£0.01 to £100.00") == (0, 1_200) - assert parse_monthly_award_band("£100.01 to £200.00") == (1_200, 2_400) - assert parse_monthly_award_band("£1000.01 to £1100.00") == (12_000, 13_200) - assert parse_monthly_award_band("£2,500.01 or over") == (30_000, np.inf) + # The bounded bands are parsed exactly as before this change. + for band, (low, high) in { + "£0.01 to £100.00": (0.01, 100.00), + "£100.01 to £200.00": (100.01, 200.00), + "£1000.01 to £1100.00": (1000.01, 1100.00), + }.items(): + assert parse_monthly_award_band(band) == (low * 12, high * 12) + assert parse_monthly_award_band("£2,500.01 or over") == (2500.01 * 12, np.inf) with pytest.raises(ValueError): parse_monthly_award_band("No payment") @@ -56,7 +61,7 @@ def test_top_band_targets_are_reachable(): target = targets[ f"dwp/uc_payment_dist/{family_type}_annual_payment_30_000_to_inf" ] - assert target.lower_bound == 30_000 + assert target.lower_bound == 2500.01 * 12 assert target.upper_bound == np.inf assert target.values[2025] == top[label] for target in targets.values(): @@ -80,13 +85,14 @@ def test_band_counts_sum_to_households_with_a_payment(): assert abs(parsed.household_count.sum() - with_payment) <= 10, label -def test_bands_tile_the_positive_awards(): +def test_bands_run_from_a_penny_to_infinity_without_overlap(): + """Each band starts a penny a month (12p a year) above the last one ends.""" for family_type, bands in uc_national_payment_dist.groupby("family_type"): bands = bands.sort_values("uc_annual_payment_min") lower = bands.uc_annual_payment_min.to_numpy() upper = bands.uc_annual_payment_max.to_numpy() - assert lower[0] == 0, family_type - assert np.array_equal(lower[1:], upper[:-1]), family_type + assert lower[0] == pytest.approx(0.12), family_type + np.testing.assert_allclose(lower[1:] - upper[:-1], 0.12, atol=1e-6) assert upper[-1] == np.inf, family_type @@ -117,7 +123,10 @@ def calculate(variable, map_to=None): _TARGETS = _uc_payment_distribution_targets() -_EDGES = sorted({t.upper_bound for t in _TARGETS if np.isfinite(t.upper_bound)}) +_EDGES = sorted( + {t.lower_bound for t in _TARGETS} + | {t.upper_bound for t in _TARGETS if np.isfinite(t.upper_bound)} +) _AWARDS = st.one_of( st.just(0.0), # no payment st.floats(0, 1e6, allow_nan=False), # anywhere, including the open top band @@ -127,6 +136,24 @@ def calculate(variable, map_to=None): ) +def _band_count(uc, family_type): + """Oracle: how many of the family type's bands hold each award, by an + interval search rather than the compute function's comparisons.""" + count = np.zeros(len(uc)) + for ft in np.unique(family_type): + bands = sorted( + (t.lower_bound, t.upper_bound) + for t in _TARGETS + if t.name.removeprefix("dwp/uc_payment_dist/").startswith(ft + "_annual") + ) + lower = np.array([b[0] for b in bands]) + upper = np.array([b[1] for b in bands]) + m = family_type == ft + i = np.searchsorted(lower, uc[m], side="right") - 1 + count[m] = (i >= 0) & (uc[m] < upper[np.maximum(i, 0)]) + return count + + @settings(max_examples=200, deadline=None) @given( st.lists( @@ -137,14 +164,21 @@ def calculate(variable, map_to=None): max_size=60, ) ) -def test_every_award_lands_in_exactly_one_band(benefit_units): - """Property: summed over bands, each household's column counts its benefit - units with a positive award, and never one without.""" +def test_each_award_lands_in_at_most_one_band(benefit_units): + """Property: summed over the bands, each household's column counts its + benefit units whose award lies in one of their family type's bands, once + (differential against an interval-search oracle). Awards from the top + band's lower bound up are always counted; zero awards never are.""" uc, family_type, household = (np.array(column) for column in zip(*benefit_units)) - ctx = _fake_ctx(uc.astype(float), family_type, household) + uc = uc.astype(float) + ctx = _fake_ctx(uc, family_type, household) - total = sum(compute_uc_payment_dist(t, ctx) for t in _TARGETS) - expected = np.bincount( - household, weights=(uc > 0).astype(float), minlength=len(total) - ) + per_target = [compute_uc_payment_dist(t, ctx) for t in _TARGETS] + total = sum(per_target) + in_band = _band_count(uc, family_type) + expected = np.bincount(household, weights=in_band, minlength=len(total)) np.testing.assert_array_equal(total, expected) + + top = 2500.01 * 12 + assert (in_band[uc >= top] == 1).all() + assert (in_band[uc == 0] == 0).all() diff --git a/policyengine_uk_data/utils/uc_data.py b/policyengine_uk_data/utils/uc_data.py index 3833d138d..17d56427c 100644 --- a/policyengine_uk_data/utils/uc_data.py +++ b/policyengine_uk_data/utils/uc_data.py @@ -4,22 +4,20 @@ def parse_monthly_award_band(band: str) -> tuple[float, float]: - """Annual (lower, upper] payment bounds of a Stat-Xplore monthly award band. + """Annual [lower, upper) payment bounds of a Stat-Xplore monthly award band. - Awards are whole pence, so the band '£100.01 to £200.00' holds monthly - awards over £100.00 and up to £200.00: annual bounds (1,200, 2,400]. The - lower bound is the previous band's top, so consecutive bands meet with no - gap. The open top band '£2500.01 or over' is (30,000, inf). + '£100.01 to £200.00' gives (1,200.12, 2,400). The open top band + '£2500.01 or over' gives (30,000.12, inf); it used to parse to missing + bounds, so its targets could never be met. """ text = band.replace("£", "").replace(",", "").strip() if text.endswith(" or over"): - lower = float(text.removesuffix(" or over")) - return round((lower - 0.01) * 12, 2), np.inf + return float(text.removesuffix(" or over")) * 12, np.inf parts = text.split(" to ") if len(parts) != 2: raise ValueError(f"Unrecognised UC monthly award band: {band!r}") - lower, upper = (float(part) for part in parts) - return round((lower - 0.01) * 12, 2), upper * 12 + lower, upper = (float(part) * 12 for part in parts) + return lower, upper def _check_bands_disjoint(bands: pd.DataFrame) -> None: From 974fa8cbb90b636fdc06e21195a420a3e0fad7cf Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 3 Oct 2026 07:34:13 -0400 Subject: [PATCH 3/3] Restore gapless UC payment bands; harden the parser and tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Stat-Xplore's monthly award is the UC due after deductions ("Monthly Award Amount (payment bands)" metadata), as universal_credit is, so a penny a month left after deductions belongs in "£0.01 to £100.00". The previous commit's reason for keeping the old [min x 12, max x 12) bounds was wrong: their £0.12 lower bound left those awards out only through float32 rounding. Bands are (lower, upper] again, as in the first commit. From the independent review of the first commit: - reject NaN, infinite or negative lower bounds and inverted bands; - scan the whole of table 4.9 for the UC rows, not the first 55 rows; - test GB scoping through create_target_matrix itself, a year missing from one UC row, and each payment band's membership against an interval-search oracle (float64 and float32 awards, edges and their neighbours); the old sum-over-bands property let an award move to the neighbouring band unnoticed; - parse the committed workbook once in the property test. Co-Authored-By: Claude Opus 5.5 --- .../obr-uc-welfare-cap-targets.fixed.md | 2 +- .../targets/compute/benefits.py | 10 +- policyengine_uk_data/targets/sources/obr.py | 3 +- .../tests/test_obr_universal_credit_target.py | 84 +++++++++++++- .../test_uc_payment_distribution_targets.py | 109 +++++++++--------- policyengine_uk_data/utils/uc_data.py | 31 ++--- 6 files changed, 166 insertions(+), 73 deletions(-) diff --git a/changelog.d/obr-uc-welfare-cap-targets.fixed.md b/changelog.d/obr-uc-welfare-cap-targets.fixed.md index 945db66c7..41872c214 100644 --- a/changelog.d/obr-uc-welfare-cap-targets.fixed.md +++ b/changelog.d/obr-uc-welfare-cap-targets.fixed.md @@ -1 +1 @@ -Calibrate universal credit to one OBR total for Great Britain, the sum of EFO table 4.9's rows inside and outside the welfare cap. The outside-the-cap row is DWP's UC equivalent of JSA (the Intensive Work Search group), not UC for households the benefit cap leaves alone: computed that way, it asked for £12.9bn of nearly all UC (estimate £72bn, +460%), and the inside-the-cap row was compared with all UK UC. Also count the open top band of the DWP UC payment distribution ("£2500.01 or over" a month): it parsed to missing bounds, so its four targets (1.4k to 83k households) could never be met. +Calibrate universal credit to one OBR total for Great Britain, the sum of EFO table 4.9's rows inside and outside the welfare cap. The outside-the-cap row is DWP's UC equivalent of JSA (the Intensive Work Search group), not UC for households the benefit cap leaves alone: computed that way, it asked for £12.9bn of nearly all UC (estimate £72bn, +460%), and the inside-the-cap row was compared with all UK UC. Also count the open top band of the DWP UC payment distribution ("£2500.01 or over" a month), which parsed to missing bounds so its four targets (1.4k to 83k households) could never be met, and make the bands meet without gaps, so awards of exactly a band's top, or a penny a month after deductions, land in their Stat-Xplore band. diff --git a/policyengine_uk_data/targets/compute/benefits.py b/policyengine_uk_data/targets/compute/benefits.py index e550a78d2..b1105855c 100644 --- a/policyengine_uk_data/targets/compute/benefits.py +++ b/policyengine_uk_data/targets/compute/benefits.py @@ -84,9 +84,11 @@ def ft_hh(value): def compute_uc_payment_dist(target, ctx) -> np.ndarray: """Compute UC payment distribution band x family type. - Bands are [lower, upper) annual amounts (see - utils.uc_data.parse_monthly_award_band); the open top band has an - infinite upper bound. + Stat-Xplore's monthly award is the UC due after deductions (its "Monthly + Award Amount (payment bands)" metadata), as universal_credit is. Bands + are (lower, upper] annual amounts (see + utils.uc_data.parse_monthly_award_band), so consecutive bands meet and + the open top band has an infinite upper bound. """ name = target.name.removeprefix("dwp/uc_payment_dist/") idx = name.index("_annual_payment_") @@ -98,7 +100,7 @@ def compute_uc_payment_dist(target, ctx) -> np.ndarray: uc_family_type = ctx.sim.calculate("family_type", map_to="benunit").values in_band = ( - (uc_payments >= lower) & (uc_payments < upper) & (uc_family_type == family_type) + (uc_payments > lower) & (uc_payments <= upper) & (uc_family_type == family_type) ) return ctx.household_from_family(in_band) diff --git a/policyengine_uk_data/targets/sources/obr.py b/policyengine_uk_data/targets/sources/obr.py index 1207aa435..457d6ccb0 100644 --- a/policyengine_uk_data/targets/sources/obr.py +++ b/policyengine_uk_data/targets/sources/obr.py @@ -455,13 +455,14 @@ def _parse_nics(wb: openpyxl.Workbook) -> list[Target]: return targets -def _universal_credit_rows(ws, max_row: int = 55) -> tuple[int, int]: +def _universal_credit_rows(ws) -> tuple[int, int]: """Table 4.9's universal credit rows inside and outside the welfare cap. Each section has exactly one row starting "Universal credit"; the outside-the-cap section starts at the row headed "Welfare spending outside the welfare cap". """ + max_row = ws.max_row boundary = _find_row( ws, "Welfare spending outside the welfare cap", max_row=max_row ) diff --git a/policyengine_uk_data/tests/test_obr_universal_credit_target.py b/policyengine_uk_data/tests/test_obr_universal_credit_target.py index d09f36353..8fcdb51e0 100644 --- a/policyengine_uk_data/tests/test_obr_universal_credit_target.py +++ b/policyengine_uk_data/tests/test_obr_universal_credit_target.py @@ -7,20 +7,23 @@ GB universal credit, the sum of both rows. """ +from functools import lru_cache from types import SimpleNamespace import numpy as np import openpyxl +import pandas as pd import pytest from hypothesis import given, settings from hypothesis import strategies as st from policyengine_uk_data.storage import STORAGE_FOLDER +from policyengine_uk_data.targets import build_loss_matrix from policyengine_uk_data.targets.build_loss_matrix import ( _compute_column, restrict_to_countries, ) -from policyengine_uk_data.targets.schema import GREAT_BRITAIN +from policyengine_uk_data.targets.schema import GREAT_BRITAIN, GeographicLevel from policyengine_uk_data.targets.sources import obr COUNTRIES = ("ENGLAND", "SCOTLAND", "WALES", "NORTHERN_IRELAND") @@ -35,6 +38,11 @@ def _uc_targets(wb) -> dict: return {t.name: t for t in obr._parse_welfare(wb) if "universal_credit" in t.name} +@lru_cache(maxsize=1) +def _committed_target(): + return _uc_targets(_committed_table())["obr/universal_credit"] + + def test_universal_credit_is_one_gb_target(): targets = _uc_targets(_committed_table()) assert list(targets) == ["obr/universal_credit"] @@ -115,6 +123,78 @@ def test_no_target_unless_both_rows_are_unambiguous(rows): assert _uc_targets(_table(rows)) == {} +def test_no_target_when_the_rows_cover_different_years(): + wb = _table( + [ + ("Universal credit", 60.0), + ("Welfare spending outside the welfare cap", None), + ("Universal credit", 10.0), + ] + ) + wb["4.9"]["I8"] = None # 2030-31 missing from the outside-the-cap row only + assert _uc_targets(wb) == {} + + +def test_rows_below_row_55_are_found(): + padding = [(f"Other benefit {i}", 1.0) for i in range(60)] + wb = _table( + [("Universal credit", 60.0)] + + padding + + [("Welfare spending outside the welfare cap", None)] + + padding + + [("Universal credit", 10.0)] + ) + target = _uc_targets(wb)["obr/universal_credit"] + assert set(target.values.values()) == {70e9} + + +def test_target_matrix_counts_gb_households_only(monkeypatch): + """Through create_target_matrix itself: England has two benefit units on + UC, Northern Ireland and Wales one each, Scotland none.""" + import policyengine_uk + + target = _committed_target() + uc = pd.Series([100.0, 50.0, 300.0, 20.0]) + benunit_household = np.array([0, 0, 1, 2]) + country = pd.Series(["ENGLAND", "NORTHERN_IRELAND", "WALES", "SCOTLAND"]) + + class FakeMicrosimulation: + tax_benefit_system = SimpleNamespace( + variables={ + "universal_credit": SimpleNamespace( + entity=SimpleNamespace(key="benunit") + ) + } + ) + + def __init__(self, dataset=None, reform=None): + pass + + def calculate(self, variable, *args, **kwargs): + return {"universal_credit": uc, "country": country}[variable] + + def map_result(self, values, source, target_entity): + assert (source, target_entity) == ("benunit", "household") + return np.bincount( + benunit_household, weights=np.asarray(values), minlength=len(country) + ) + + monkeypatch.setattr(policyengine_uk, "Microsimulation", FakeMicrosimulation) + monkeypatch.setattr( + build_loss_matrix, + "get_all_targets", + lambda geographic_level=None: ( + [target] if geographic_level == GeographicLevel.NATIONAL else [] + ), + ) + + matrix, values = build_loss_matrix.create_target_matrix( + SimpleNamespace(time_period="2025"), time_period="2025" + ) + np.testing.assert_array_equal(matrix["obr/universal_credit"], [150, 0, 20, 0]) + assert values["obr/universal_credit"] == target.values[2025] + + def _fake_ctx(uc, benunit_household, household_country): n_households = len(household_country) variables = { @@ -154,7 +234,7 @@ def test_column_counts_gb_universal_credit_once(case): benunit_household = np.array([h for h, _ in benunits], dtype=int) uc = np.array([amount for _, amount in benunits], dtype=float) ctx = _fake_ctx(uc, benunit_household, household_country) - target = _uc_targets(_committed_table())["obr/universal_credit"] + target = _committed_target() column = restrict_to_countries( _compute_column(target, ctx, 2025), ctx.country, target.countries diff --git a/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py b/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py index 7bfc758d1..6b0ff8888 100644 --- a/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py +++ b/policyengine_uk_data/tests/test_uc_payment_distribution_targets.py @@ -4,7 +4,6 @@ counts households on UC by monthly award band and family type. Its top band, '£2500.01 or over', is open-ended; it used to parse to NaN bounds, so its four targets (1.4k to 83k households) had a column of zeros and could never be met. -The other bands keep their bounds: [monthly minimum, monthly maximum) x 12. """ from types import SimpleNamespace @@ -42,18 +41,23 @@ def _raw_extract() -> pd.DataFrame: def test_parse_monthly_award_band(): - # The bounded bands are parsed exactly as before this change. - for band, (low, high) in { - "£0.01 to £100.00": (0.01, 100.00), - "£100.01 to £200.00": (100.01, 200.00), - "£1000.01 to £1100.00": (1000.01, 1100.00), - }.items(): - assert parse_monthly_award_band(band) == (low * 12, high * 12) - assert parse_monthly_award_band("£2,500.01 or over") == (2500.01 * 12, np.inf) + assert parse_monthly_award_band("£0.01 to £100.00") == (0, 1_200) + assert parse_monthly_award_band("£100.01 to £200.00") == (1_200, 2_400) + assert parse_monthly_award_band("£1000.01 to £1100.00") == (12_000, 13_200) + assert parse_monthly_award_band("£2,500.01 or over") == (30_000, np.inf) with pytest.raises(ValueError): parse_monthly_award_band("No payment") +@pytest.mark.parametrize( + "band", + ["£nan or over", "£inf or over", "£100.01 to £99.00", "£1.00 to £nan"], +) +def test_invalid_bands_are_rejected(band): + with pytest.raises(ValueError): + parse_monthly_award_band(band) + + def test_top_band_targets_are_reachable(): targets = {t.name: t for t in _uc_payment_distribution_targets()} top = _raw_extract().loc["£2500.01 or over"] @@ -61,7 +65,7 @@ def test_top_band_targets_are_reachable(): target = targets[ f"dwp/uc_payment_dist/{family_type}_annual_payment_30_000_to_inf" ] - assert target.lower_bound == 2500.01 * 12 + assert target.lower_bound == 30_000 assert target.upper_bound == np.inf assert target.values[2025] == top[label] for target in targets.values(): @@ -85,14 +89,13 @@ def test_band_counts_sum_to_households_with_a_payment(): assert abs(parsed.household_count.sum() - with_payment) <= 10, label -def test_bands_run_from_a_penny_to_infinity_without_overlap(): - """Each band starts a penny a month (12p a year) above the last one ends.""" +def test_bands_tile_the_positive_awards(): for family_type, bands in uc_national_payment_dist.groupby("family_type"): bands = bands.sort_values("uc_annual_payment_min") lower = bands.uc_annual_payment_min.to_numpy() upper = bands.uc_annual_payment_max.to_numpy() - assert lower[0] == pytest.approx(0.12), family_type - np.testing.assert_allclose(lower[1:] - upper[:-1], 0.12, atol=1e-6) + assert lower[0] == 0, family_type + assert np.array_equal(lower[1:], upper[:-1]), family_type assert upper[-1] == np.inf, family_type @@ -123,62 +126,64 @@ def calculate(variable, map_to=None): _TARGETS = _uc_payment_distribution_targets() -_EDGES = sorted( - {t.lower_bound for t in _TARGETS} - | {t.upper_bound for t in _TARGETS if np.isfinite(t.upper_bound)} -) +_EDGES = sorted({t.upper_bound for t in _TARGETS if np.isfinite(t.upper_bound)}) _AWARDS = st.one_of( st.just(0.0), # no payment st.floats(0, 1e6, allow_nan=False), # anywhere, including the open top band st.sampled_from(_EDGES), # exactly on a band edge st.sampled_from(_EDGES).map(lambda edge: np.nextafter(edge, np.inf)), st.sampled_from(_EDGES).map(lambda edge: np.nextafter(edge, -np.inf)), + st.sampled_from(_EDGES).map(lambda e: np.nextafter(np.float32(e), np.inf)), + st.sampled_from(_EDGES).map(lambda e: np.nextafter(np.float32(e), -np.inf)), +) +_BENEFIT_UNITS = st.lists( + st.tuples(_AWARDS, st.sampled_from(list(FAMILY_TYPES.values())), st.integers(0, 9)), + min_size=1, + max_size=60, ) -def _band_count(uc, family_type): - """Oracle: how many of the family type's bands hold each award, by an - interval search rather than the compute function's comparisons.""" - count = np.zeros(len(uc)) +def _oracle_band(uc, family_type): + """Target name of the band holding each award (None for no payment), found + by searching the family type's sorted upper bounds rather than by the + compute function's comparisons.""" + names = np.full(len(uc), None, dtype=object) for ft in np.unique(family_type): bands = sorted( - (t.lower_bound, t.upper_bound) + (t.upper_bound, t.lower_bound, t.name) for t in _TARGETS if t.name.removeprefix("dwp/uc_payment_dist/").startswith(ft + "_annual") ) - lower = np.array([b[0] for b in bands]) - upper = np.array([b[1] for b in bands]) + upper = np.array([b[0] for b in bands]) m = family_type == ft - i = np.searchsorted(lower, uc[m], side="right") - 1 - count[m] = (i >= 0) & (uc[m] < upper[np.maximum(i, 0)]) - return count + i = np.searchsorted(upper, uc[m], side="left") + names[m] = [ + bands[k][2] if k < len(bands) and award > bands[k][1] else None + for k, award in zip(i, uc[m]) + ] + return names @settings(max_examples=200, deadline=None) -@given( - st.lists( - st.tuples( - _AWARDS, st.sampled_from(list(FAMILY_TYPES.values())), st.integers(0, 9) - ), - min_size=1, - max_size=60, - ) -) -def test_each_award_lands_in_at_most_one_band(benefit_units): - """Property: summed over the bands, each household's column counts its - benefit units whose award lies in one of their family type's bands, once - (differential against an interval-search oracle). Awards from the top - band's lower bound up are always counted; zero awards never are.""" +@given(_BENEFIT_UNITS, st.booleans()) +def test_each_award_lands_in_its_own_band(benefit_units, as_float32): + """Property: every target's column counts exactly the household's benefit + units whose award an interval search places in that target's band (so no + award is counted twice, moved to a neighbouring band, or lost), for float64 + and float32 awards, on band edges and one ulp either side.""" uc, family_type, household = (np.array(column) for column in zip(*benefit_units)) - uc = uc.astype(float) + uc = uc.astype(np.float32 if as_float32 else float) ctx = _fake_ctx(uc, family_type, household) + band = _oracle_band(uc, family_type) - per_target = [compute_uc_payment_dist(t, ctx) for t in _TARGETS] - total = sum(per_target) - in_band = _band_count(uc, family_type) - expected = np.bincount(household, weights=in_band, minlength=len(total)) - np.testing.assert_array_equal(total, expected) - - top = 2500.01 * 12 - assert (in_band[uc >= top] == 1).all() - assert (in_band[uc == 0] == 0).all() + for target in _TARGETS: + expected = np.bincount( + household, + weights=(band == target.name).astype(float), + minlength=household.max() + 1, + ) + np.testing.assert_array_equal( + compute_uc_payment_dist(target, ctx), expected, err_msg=target.name + ) + # Every positive award is in some band; no payment is in none. + assert all((b is not None) == (award > 0) for b, award in zip(band, uc)) diff --git a/policyengine_uk_data/utils/uc_data.py b/policyengine_uk_data/utils/uc_data.py index 17d56427c..ffe20896e 100644 --- a/policyengine_uk_data/utils/uc_data.py +++ b/policyengine_uk_data/utils/uc_data.py @@ -4,29 +4,34 @@ def parse_monthly_award_band(band: str) -> tuple[float, float]: - """Annual [lower, upper) payment bounds of a Stat-Xplore monthly award band. + """Annual (lower, upper] payment bounds of a Stat-Xplore monthly award band. - '£100.01 to £200.00' gives (1,200.12, 2,400). The open top band - '£2500.01 or over' gives (30,000.12, inf); it used to parse to missing - bounds, so its targets could never be met. + Awards are whole pence, so the band '£100.01 to £200.00' holds monthly + awards over £100.00 and up to £200.00: annual bounds (1,200, 2,400]. The + lower bound is the previous band's top, so consecutive bands meet with no + gap. The open top band '£2500.01 or over' is (30,000, inf). """ text = band.replace("£", "").replace(",", "").strip() if text.endswith(" or over"): - return float(text.removesuffix(" or over")) * 12, np.inf - parts = text.split(" to ") - if len(parts) != 2: - raise ValueError(f"Unrecognised UC monthly award band: {band!r}") - lower, upper = (float(part) * 12 for part in parts) + lower, upper = float(text.removesuffix(" or over")), np.inf + else: + parts = text.split(" to ") + if len(parts) != 2: + raise ValueError(f"Unrecognised UC monthly award band: {band!r}") + lower, upper = float(parts[0]), float(parts[1]) * 12 + lower = round((lower - 0.01) * 12, 2) + if not (np.isfinite(lower) and lower >= 0 and upper > lower): + raise ValueError(f"Invalid UC monthly award band: {band!r}") return lower, upper def _check_bands_disjoint(bands: pd.DataFrame) -> None: """Fail if any family type's payment bands overlap. - Stat-Xplore also lists summary bands ('£1500.01 or over') that span the - finer bands below them. They are suppressed ('..') in the committed - extract; if a new extract filled one in, counting it as well would double - count those households. + Stat-Xplore's '£1500.01 or over' band is its top band for months up to + August 2022 and spans the finer bands added from September 2022. It is + suppressed ('..') in the committed extract; if a new extract filled it + in, counting it as well would double count those households. """ for family_type, group in bands.groupby("family_type"): group = group.sort_values("uc_annual_payment_min")