From 40aaab0069d50900dff3413bfc0fcd3dd584c3f0 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Wed, 30 Sep 2026 06:40:57 -0400 Subject: [PATCH 1/4] Impute UC deductions and private school attendance in every data-built simulation uc_deduction_random_draw, uc_deduction_type_random_draw and attends_private_school decided between microdata imputation and household defaults by testing whether total weight was below 1e6. policyengine.py builds constituency and local-authority simulations by filtering rows from the national data (RowFilterStrategy), so those fell below the threshold and got household defaults: no UC deductions and no private school attendance. Each now reads Simulation.built_from_dataset (added in #1899). filter_dataset carries each person's attends_private_school into an extract, since one household alone would rank at the 100th income percentile. Co-Authored-By: Claude Opus 5.5 --- ...ilt-from-dataset-imputation-gates.fixed.md | 1 + docs/book/validation/uc-deductions.md | 2 +- policyengine_uk/data/filter_dataset.py | 22 +- .../tests/test_data_built_imputations.py | 201 ++++++++++++++++++ .../contrib/labour/attends_private_school.py | 11 +- .../deductions/uc_deduction_random_draw.py | 17 +- .../uc_deduction_type_random_draw.py | 16 +- 7 files changed, 238 insertions(+), 32 deletions(-) create mode 100644 changelog.d/built-from-dataset-imputation-gates.fixed.md create mode 100644 policyengine_uk/tests/test_data_built_imputations.py diff --git a/changelog.d/built-from-dataset-imputation-gates.fixed.md b/changelog.d/built-from-dataset-imputation-gates.fixed.md new file mode 100644 index 0000000000..eaa94b75c5 --- /dev/null +++ b/changelog.d/built-from-dataset-imputation-gates.fixed.md @@ -0,0 +1 @@ +- UC deduction draws and private school attendance now apply to every simulation built from data, including a constituency or local authority filtered from the national data, instead of only those carrying over a million people of weight. `filter_dataset` carries each person's private school attendance into the household it extracts. diff --git a/docs/book/validation/uc-deductions.md b/docs/book/validation/uc-deductions.md index 5d7d4fc4a4..7264e11ee0 100644 --- a/docs/book/validation/uc-deductions.md +++ b/docs/book/validation/uc-deductions.md @@ -81,4 +81,4 @@ A cap of 1 − *x* is equivalent to a protected minimum floor at *x* of the stan Statutory parameters — the cap, the protected floor, the minimum payable penny, the abolition switches — live under `gov.dwp.universal_credit.deductions`. The calibrated distributions live under `gov.simulation.uc_deductions`: they describe the world, not the law. -The assignment formulas are an explicit fallback. The end state imputes `uc_latent_deduction_rate` and `uc_deduction_combination` at dataset build, at which point the model consumes them as plain inputs and the fallback retires; raw-FRS users keep working through the fallback until then. Assignment uses deterministic splitmix64 hashes of `benunit_id`, reproducible across runs and machines, and overridable by datasets or situations through `uc_deduction_random_draw` and `uc_deduction_type_random_draw`. Single-household simulations get no deductions unless set explicitly. +The assignment formulas are an explicit fallback. The end state imputes `uc_latent_deduction_rate` and `uc_deduction_combination` at dataset build, at which point the model consumes them as plain inputs and the fallback retires; raw-FRS users keep working through the fallback until then. Assignment uses deterministic splitmix64 hashes of `benunit_id`, reproducible across runs and machines, and overridable by datasets or situations through `uc_deduction_random_draw` and `uc_deduction_type_random_draw`. Every simulation built from data gets the hashed draws, including a constituency or local authority filtered from the national data, and a benefit unit gets the same draw there as in the national run. Household situations get no deductions unless set explicitly. diff --git a/policyengine_uk/data/filter_dataset.py b/policyengine_uk/data/filter_dataset.py index 3b70dd4c0d..09e8ef3a2f 100644 --- a/policyengine_uk/data/filter_dataset.py +++ b/policyengine_uk/data/filter_dataset.py @@ -37,17 +37,17 @@ def filter_dataset( dataset: UKSingleYearDataset = sim.dataset[year] new_dataset = dataset.copy() person = new_dataset.person - if "months_since_last_birthday" not in person.columns: - # Birthdays are spread over the year across the whole population, so - # carry each person's place across rather than recompute it for one - # household. - months = pd.Series( - np.asarray(sim.calculate("months_since_last_birthday", year)), - index=np.asarray(sim.calculate("person_id", year)), - ) - person = person.assign( - months_since_last_birthday=months.loc[person.person_id].values - ) + # These are imputed across the whole population: birthdays are spread over + # the year, and private school attendance follows each household's income + # percentile (one household alone would rank at the 100th). Carry each + # person's value across rather than recompute it for one household. + for variable in ("months_since_last_birthday", "attends_private_school"): + if variable not in person.columns: + values = pd.Series( + np.asarray(sim.calculate(variable, year)), + index=np.asarray(sim.calculate("person_id", year)), + ) + person = person.assign(**{variable: values.loc[person.person_id].values}) new_dataset.person = person[person.person_household_id == household_id] new_dataset.household = new_dataset.household[ new_dataset.household.household_id == household_id diff --git a/policyengine_uk/tests/test_data_built_imputations.py b/policyengine_uk/tests/test_data_built_imputations.py new file mode 100644 index 0000000000..53441f64f6 --- /dev/null +++ b/policyengine_uk/tests/test_data_built_imputations.py @@ -0,0 +1,201 @@ +"""Imputations that apply to simulations built from data, not to situations. + +UC deduction draws and private school attendance are imputed across a +population. How the simulation was built decides whether they apply, not how +much weight it carries: a constituency or local authority filtered from the +national data (as policyengine.py's RowFilterStrategy does) carries well under +a million people of weight and is still data. +""" + +import numpy as np +import pandas as pd +import pytest +from hypothesis import given, settings +from hypothesis import strategies as st + +from policyengine_uk import Microsimulation, Simulation +from policyengine_uk.data import ( + UKMultiYearDataset, + UKSingleYearDataset, + filter_dataset, +) +from policyengine_uk.utils.stochastic import splitmix64_uniform + +YEAR = 2025 +REGIONS = np.array(["LONDON", "NORTH_EAST", "SCOTLAND", "WALES"]) +DRAWS = ("uc_deduction_random_draw", "uc_deduction_type_random_draw") +UC_OUTPUTS = ( + *DRAWS, + "uc_has_deduction", + "uc_deduction_combination", + "uc_deductions", +) + + +def tables(n: int = 400, weight_scale: float = 1.0) -> dict: + """n households of one adult and one child, all on Universal Credit, with + incomes spread from 0 to 150k and draws for private school attendance.""" + ids = np.arange(n) + rng = np.random.default_rng(0) + person = pd.DataFrame( + { + "person_id": np.concatenate([ids * 10 + 1, ids * 10 + 2]), + "person_benunit_id": np.concatenate([ids, ids]), + "person_household_id": np.concatenate([ids, ids]), + "age": np.concatenate([np.full(n, 35), np.full(n, 10)]), + "employment_income": np.concatenate( + [np.linspace(0, 150_000, n), np.zeros(n)] + ), + "attends_private_school_random_draw": np.concatenate( + [np.ones(n), rng.uniform(0, 0.5, n)] + ), + } + ) + benunit = pd.DataFrame( + { + "benunit_id": ids, + "would_claim_uc": np.ones(n, dtype=bool), + "universal_credit_pre_benefit_cap": np.full(n, 6_000.0), + "benefit_cap_reduction": np.zeros(n), + } + ) + household = pd.DataFrame( + { + "household_id": ids, + "household_weight": weight_scale * rng.lognormal(0, 0.5, n), + "region": REGIONS[ids % len(REGIONS)], + } + ) + return {"person": person, "benunit": benunit, "household": household} + + +def data_simulation(t: dict) -> Microsimulation: + # Copies: building encodes enum columns in place. + year = UKSingleYearDataset( + person=t["person"].copy(), + benunit=t["benunit"].copy(), + household=t["household"].copy(), + fiscal_year=YEAR, + ) + return Microsimulation(dataset=UKMultiYearDataset(datasets=[year])) + + +def row_filter(t: dict, keep) -> dict: + """Keep households where keep(household table) holds, with their benefit + units and people: what policyengine.py does for a constituency.""" + household = t["household"][keep(t["household"])] + person = t["person"][t["person"].person_household_id.isin(household.household_id)] + benunit = t["benunit"][t["benunit"].benunit_id.isin(person.person_benunit_id)] + return { + "person": person.reset_index(drop=True), + "benunit": benunit.reset_index(drop=True), + "household": household.reset_index(drop=True), + } + + +def values(sim, variable: str) -> np.ndarray: + return np.asarray(sim.calculate(variable, YEAR)) + + +@pytest.fixture(scope="module") +def national(): + t = tables() + return t, data_simulation(t) + + +def test_small_data_built_simulation_gets_hashed_draws(): + small = tables(weight_scale=1e-3) + assert small["household"].household_weight.sum() < 1e6 + sim = data_simulation(small) + ids = small["benunit"].benunit_id.values + for salt, draw in enumerate(DRAWS): + expected = splitmix64_uniform(ids, salt=salt).astype(np.float32) + assert np.array_equal(values(sim, draw), expected), draw + # Every benefit unit is on UC, so some have deductions. + assert values(sim, "uc_deductions").max() > 0 + assert values(sim, "attends_private_school").any() + + +@settings(max_examples=6, deadline=None) +@given(log_scale=st.floats(min_value=-4, max_value=9)) +def test_imputations_do_not_depend_on_total_weight(national, log_scale): + """Scaling every weight by the same factor changes nothing: hashed draws + depend only on ids, and income percentiles only on relative weights.""" + t, reference = national + sim = data_simulation(tables(weight_scale=10**log_scale)) + for variable in (*UC_OUTPUTS, "attends_private_school"): + assert np.array_equal(values(sim, variable), values(reference, variable)), ( + variable + ) + + +def test_region_filtered_from_data_matches_the_national_run(national): + """Every benefit unit in a region filtered from the data has the UC + deductions it has in the national simulation.""" + t, sim = national + region = row_filter(t, lambda h: h.region == "LONDON") + assert region["household"].household_weight.sum() < 1e6 + regional = data_simulation(region) + kept = np.isin(t["benunit"].benunit_id, region["benunit"].benunit_id) + for variable in UC_OUTPUTS: + assert np.array_equal( + values(regional, variable), values(sim, variable)[kept] + ), variable + assert values(regional, "uc_deductions").max() > 0 + # Private school attendance ranks incomes within the simulated population, + # so a region ranks against itself; it is still imputed. + assert values(regional, "attends_private_school").any() + + +def test_extracted_household_keeps_its_imputations(national): + """filter_dataset extracts one household from the data. Its UC draws hash + the same ids, and it carries its private school attendance across rather + than ranking at the 100th percentile of a population of one.""" + t, sim = national + attends = values(sim, "attends_private_school") + person_household = t["person"].person_household_id.values + deductions = values(sim, "uc_deductions") + with_deductions = np.flatnonzero(deductions > 0) + attending = np.unique(person_household[attends]) + not_attending = np.setdiff1d(t["household"].household_id, attending) + assert len(with_deductions) and len(attending) + for household in [*with_deductions[:3], *attending[:3], *not_attending[-3:]]: + extract = filter_dataset(sim, household_id=int(household), year=YEAR) + extracted = Microsimulation(dataset=UKMultiYearDataset(datasets=[extract])) + assert values(extracted, "uc_deductions")[0] == deductions[household] + assert np.array_equal( + values(extracted, "attends_private_school"), + attends[person_household == household], + ) + + +def test_situations_get_defaults_whatever_their_weight(): + """A household situation is not data, even with the weight of a nation.""" + situation = { + "people": { + "adult": {"age": {YEAR: 35}}, + "child": { + "age": {YEAR: 10}, + "attends_private_school_random_draw": {YEAR: 0}, + }, + }, + "benunits": { + "benunit": { + "members": ["adult", "child"], + "would_claim_uc": {YEAR: True}, + "universal_credit_pre_benefit_cap": {YEAR: 6_000}, + "benefit_cap_reduction": {YEAR: 0}, + } + }, + "households": { + "household": { + "members": ["adult", "child"], + "household_weight": {YEAR: 1e9}, + } + }, + } + sim = Simulation(situation=situation) + for draw in DRAWS: + assert values(sim, draw)[0] == 1.0 + assert values(sim, "uc_deductions")[0] == 0 + assert not values(sim, "attends_private_school").any() diff --git a/policyengine_uk/variables/contrib/labour/attends_private_school.py b/policyengine_uk/variables/contrib/labour/attends_private_school.py index 27cc135f61..54bb90ddfb 100644 --- a/policyengine_uk/variables/contrib/labour/attends_private_school.py +++ b/policyengine_uk/variables/contrib/labour/attends_private_school.py @@ -35,7 +35,11 @@ class attends_private_school(Variable): value_type = bool def formula(person, period, parameters): - if not hasattr(person.simulation, "dataset"): + # Imputed only in simulations built from data, including a region or + # constituency filtered from them. A household situation has no + # income distribution to rank within, so it attends no private school + # unless set. + if not getattr(person.simulation, "built_from_dataset", False): return 0 household = person.household # To ensure that our model matches @@ -63,9 +67,8 @@ def formula(person, period, parameters): household_weight = household("household_weight", period) weighted_income = MicroSeries(net_income, weights=household_weight) - if household_weight.sum() < 1e6: - return 0 - + # Percentiles rank households within the simulated population, so a + # region filtered from the data ranks against itself, not the UK. percentile = np.zeros_like(weighted_income).astype(numpy.int64) mask = household_weight > 0 diff --git a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py index 22c73c1c30..7cb5a53aa8 100644 --- a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py +++ b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py @@ -6,9 +6,10 @@ class uc_deduction_random_draw(Variable): label = "UC deduction random draw" documentation = ( "Uniform draw on [0, 1) determining deduction incidence and size. " - "Deterministic hash of the benefit unit id in dataset simulations; " - "1.0 in single-household simulations, so no deduction unless set. " - "Datasets and situations can override it directly." + "Deterministic hash of the benefit unit id in simulations built from " + "data, including a region or constituency filtered from them; 1.0 in " + "household situations, so no deduction unless set. Datasets and " + "situations can override it directly." ) entity = BenUnit definition_period = YEAR @@ -16,11 +17,11 @@ class uc_deduction_random_draw(Variable): default_value = 1.0 def formula(benunit, period, parameters): - # Representative microdata carries tens of millions of households of - # weight; single-household situations carry ~1. Only assign hashed - # draws in representative simulations: the 1.0 default never falls - # below any incidence, so calculators get no deductions unless set. - if benunit("benunit_weight", period).sum() < 1e6: + # Hashed draws in every simulation built from data, however little + # weight it carries: a constituency filtered from the national data is + # still data. Household situations get 1.0, which never falls below + # any incidence, so calculators get no deductions unless set. + if not getattr(benunit.simulation, "built_from_dataset", False): return np.ones(benunit.count) ids = benunit("benunit_id", period) return splitmix64_uniform(ids, salt=0) diff --git a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py index 284739d134..89f2020991 100644 --- a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py +++ b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py @@ -6,9 +6,10 @@ class uc_deduction_type_random_draw(Variable): label = "UC deduction type random draw" documentation = ( "Uniform draw on [0, 1) determining the deduction type combination. " - "Deterministic hash of the benefit unit id in dataset simulations; " - "1.0 in single-household simulations. Datasets and situations can " - "override it directly." + "Deterministic hash of the benefit unit id in simulations built from " + "data, including a region or constituency filtered from them; 1.0 in " + "household situations. Datasets and situations can override it " + "directly." ) entity = BenUnit definition_period = YEAR @@ -16,11 +17,10 @@ class uc_deduction_type_random_draw(Variable): default_value = 1.0 def formula(benunit, period, parameters): - # Representative microdata carries tens of millions of households of - # weight; single-household situations carry ~1. Only assign hashed - # draws in representative simulations; 1.0 maps to the last type - # combination but only matters when a deduction is assigned. - if benunit("benunit_weight", period).sum() < 1e6: + # Hashed draws in every simulation built from data, however little + # weight it carries. Household situations get 1.0, which maps to the + # last type combination but only matters when a deduction is assigned. + if not getattr(benunit.simulation, "built_from_dataset", False): return np.ones(benunit.count) ids = benunit("benunit_id", period) return splitmix64_uniform(ids, salt=1) From 84cc1a6c4b7f78c0bdead775092aeeb2aec6dc01 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Wed, 30 Sep 2026 06:56:51 -0400 Subject: [PATCH 2/4] Address review: zero-weight data, exact scale property, clearer tests - attends_private_school no longer raises when no household has weight (the old 1e6 gate returned early; MicroSeries cannot rank zero weight). - The weight-scale property scales by powers of two, so invariance is exact; the national fixture carries national-scale weight, so the region test compares a national run with a constituency-sized one. - Replace the vacuous attends_private_school YAML cases with situation cases that fail under the old gate. - Document what filter_dataset carries, and that the changelog's filtered regions rank private school attendance locally. Co-Authored-By: Claude Opus 5.5 --- ...ilt-from-dataset-imputation-gates.fixed.md | 2 +- policyengine_uk/data/filter_dataset.py | 6 +++ .../labour/attends_private_school.yaml | 29 +++++------- .../tests/test_data_built_imputations.py | 45 ++++++++++++++----- .../contrib/labour/attends_private_school.py | 15 ++++--- 5 files changed, 61 insertions(+), 36 deletions(-) diff --git a/changelog.d/built-from-dataset-imputation-gates.fixed.md b/changelog.d/built-from-dataset-imputation-gates.fixed.md index eaa94b75c5..2db9e7f4eb 100644 --- a/changelog.d/built-from-dataset-imputation-gates.fixed.md +++ b/changelog.d/built-from-dataset-imputation-gates.fixed.md @@ -1 +1 @@ -- UC deduction draws and private school attendance now apply to every simulation built from data, including a constituency or local authority filtered from the national data, instead of only those carrying over a million people of weight. `filter_dataset` carries each person's private school attendance into the household it extracts. +- UC deduction draws and private school attendance now apply to every simulation built from data, including a constituency or local authority filtered from the national data, instead of only those carrying over a million people of weight. A filtered region ranks private school attendance by its own income distribution. `filter_dataset` carries each person's private school attendance into the household it extracts. diff --git a/policyengine_uk/data/filter_dataset.py b/policyengine_uk/data/filter_dataset.py index 09e8ef3a2f..51e1a7590f 100644 --- a/policyengine_uk/data/filter_dataset.py +++ b/policyengine_uk/data/filter_dataset.py @@ -20,6 +20,12 @@ def filter_dataset( This function creates a new dataset containing only the specified household and the associated benefit units and people within that household. + Values imputed across the whole population (months_since_last_birthday and + attends_private_school) are taken from sim for the given year and carried + in as inputs, so the extract keeps them in later years too. Private school + attendance ranks incomes, so the first extract from a simulation computes + its taxes and benefits. + Parameters ---------- sim : Microsimulation diff --git a/policyengine_uk/tests/policy/baseline/contrib/labour/attends_private_school.yaml b/policyengine_uk/tests/policy/baseline/contrib/labour/attends_private_school.yaml index da00699567..0db7ac9ade 100644 --- a/policyengine_uk/tests/policy/baseline/contrib/labour/attends_private_school.yaml +++ b/policyengine_uk/tests/policy/baseline/contrib/labour/attends_private_school.yaml @@ -1,41 +1,32 @@ -- name: Attends private school returns False when attendance rate is 0% +- name: A household situation attends no private school unless set, even at the top of the income distribution period: 2024 input: - gov.simulation.private_school_vat.private_school_attendance_rate.100: 0 - gov.simulation.private_school_vat.private_school_attendance_rate.95: 0 people: - adult: + adult: age: 25 child: age: 10 + attends_private_school_random_draw: 0 households: household: - household_weight: 0.001 + household_weight: 1_000_000_000 household_market_income: 1_000_000_000 household_benefits: 0 - members: [ - adult, - child - ] + members: [adult, child] output: attends_private_school: [False, False] -- name: Attends private school successfully returns False when household weights are 0 +- name: A household situation keeps attendance it sets period: 2024 input: people: - adult: + adult: age: 25 child: age: 10 + attends_private_school: true households: household: - household_weight: 0 - household_market_income: 1_000_000_000 - household_benefits: 0 - members: [ - adult, - child - ] + members: [adult, child] output: - attends_private_school: [False, False] + attends_private_school: [False, True] diff --git a/policyengine_uk/tests/test_data_built_imputations.py b/policyengine_uk/tests/test_data_built_imputations.py index 53441f64f6..e57f41e4bf 100644 --- a/policyengine_uk/tests/test_data_built_imputations.py +++ b/policyengine_uk/tests/test_data_built_imputations.py @@ -30,6 +30,9 @@ "uc_deduction_combination", "uc_deductions", ) +# Weight per household for a national-scale population: 400 households carry +# about 2.3m of weight, a quarter of them (one region) about 0.6m. +NATIONAL_SCALE = 5_000 def tables(n: int = 400, weight_scale: float = 1.0) -> dict: @@ -99,7 +102,8 @@ def values(sim, variable: str) -> np.ndarray: @pytest.fixture(scope="module") def national(): - t = tables() + t = tables(weight_scale=NATIONAL_SCALE) + assert t["household"].household_weight.sum() > 1e6 return t, data_simulation(t) @@ -117,12 +121,14 @@ def test_small_data_built_simulation_gets_hashed_draws(): @settings(max_examples=6, deadline=None) -@given(log_scale=st.floats(min_value=-4, max_value=9)) -def test_imputations_do_not_depend_on_total_weight(national, log_scale): +@given(power=st.integers(min_value=-14, max_value=30)) +def test_imputations_do_not_depend_on_total_weight(national, power): """Scaling every weight by the same factor changes nothing: hashed draws - depend only on ids, and income percentiles only on relative weights.""" + depend only on ids, and income percentiles only on relative weights. + Powers of two scale exactly in floating point, so this holds exactly for + totals from about a hundred to about 10^15.""" t, reference = national - sim = data_simulation(tables(weight_scale=10**log_scale)) + sim = data_simulation(tables(weight_scale=NATIONAL_SCALE * 2.0**power)) for variable in (*UC_OUTPUTS, "attends_private_school"): assert np.array_equal(values(sim, variable), values(reference, variable)), ( variable @@ -134,6 +140,7 @@ def test_region_filtered_from_data_matches_the_national_run(national): deductions it has in the national simulation.""" t, sim = national region = row_filter(t, lambda h: h.region == "LONDON") + # A constituency-sized share of a national-sized population. assert region["household"].household_weight.sum() < 1e6 regional = data_simulation(region) kept = np.isin(t["benunit"].benunit_id, region["benunit"].benunit_id) @@ -153,22 +160,40 @@ def test_extracted_household_keeps_its_imputations(national): than ranking at the 100th percentile of a population of one.""" t, sim = national attends = values(sim, "attends_private_school") - person_household = t["person"].person_household_id.values - deductions = values(sim, "uc_deductions") - with_deductions = np.flatnonzero(deductions > 0) + person_household = values(sim, "person_household_id") + deductions = pd.Series( + values(sim, "uc_deductions"), index=values(sim, "benunit_id") + ) + # One benefit unit per household, sharing its id. + with_deductions = deductions.index[deductions > 0] attending = np.unique(person_household[attends]) - not_attending = np.setdiff1d(t["household"].household_id, attending) + not_attending = np.setdiff1d(values(sim, "household_id"), attending) assert len(with_deductions) and len(attending) for household in [*with_deductions[:3], *attending[:3], *not_attending[-3:]]: extract = filter_dataset(sim, household_id=int(household), year=YEAR) extracted = Microsimulation(dataset=UKMultiYearDataset(datasets=[extract])) - assert values(extracted, "uc_deductions")[0] == deductions[household] + assert np.array_equal( + values(extracted, "uc_deductions"), + deductions.loc[values(extracted, "benunit_id")].values, + ) assert np.array_equal( values(extracted, "attends_private_school"), attends[person_household == household], ) +def test_data_without_weight_attends_no_private_school(): + """Every household at zero weight: no income ranking is possible, so no + household reaches a percentile with a positive rate, and nothing fails.""" + sim = data_simulation(tables(n=40, weight_scale=0)) + assert not values(sim, "attends_private_school").any() + ids = values(sim, "benunit_id") + assert np.array_equal( + values(sim, "uc_deduction_random_draw"), + splitmix64_uniform(ids, salt=0).astype(np.float32), + ) + + def test_situations_get_defaults_whatever_their_weight(): """A household situation is not data, even with the weight of a nation.""" situation = { diff --git a/policyengine_uk/variables/contrib/labour/attends_private_school.py b/policyengine_uk/variables/contrib/labour/attends_private_school.py index 54bb90ddfb..78be836c55 100644 --- a/policyengine_uk/variables/contrib/labour/attends_private_school.py +++ b/policyengine_uk/variables/contrib/labour/attends_private_school.py @@ -69,15 +69,18 @@ def formula(person, period, parameters): # Percentiles rank households within the simulated population, so a # region filtered from the data ranks against itself, not the UK. + # Households without weight stay at percentile 0 (a rate of 0 unless + # reformed), including when no household has weight. percentile = np.zeros_like(weighted_income).astype(numpy.int64) mask = household_weight > 0 - percentile[mask] = ( - weighted_income[mask] - .percentile_rank() - .clip(0, 100) - .values.astype(numpy.int64) - ) + if mask.any(): + percentile[mask] = ( + weighted_income[mask] + .percentile_rank() + .clip(0, 100) + .values.astype(numpy.int64) + ) # STUDENT_POPULATION_ADJUSTMENT_FACTOR = 0.78 STUDENT_POPULATION_ADJUSTMENT_FACTOR = population_adjustment_factor From 77141d92b53838f95564c360a4a000312082e999 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 00:46:35 -0400 Subject: [PATCH 3/4] Address review: set the data-built flag in the builders, honour core datasets - build_from_situation clears Simulation.built_from_dataset and build_from_ids (every data path) sets it, so a simulation rebuilt in place follows its new source. __init__ no longer sets it separately. - utils.data_source.built_from_data falls back to policyengine-core's is_over_dataset, so a core Simulation built over the UK system from data gets the imputations. The four gates (both UC draws, private school attendance, months_since_last_birthday) use it. - The region test's population now sits below the old threshold for people as well as benefit units, so its private school assertion fails under the old gate. New tests cover both rebuild directions and a core simulation. Co-Authored-By: Claude Opus 5.5 --- ...ilt-from-dataset-imputation-gates.fixed.md | 2 +- policyengine_uk/simulation.py | 10 +- .../tests/test_data_built_imputations.py | 117 +++++++++++++----- policyengine_uk/utils/data_source.py | 14 +++ .../contrib/labour/attends_private_school.py | 3 +- .../deductions/uc_deduction_random_draw.py | 3 +- .../uc_deduction_type_random_draw.py | 3 +- .../demographic/months_since_last_birthday.py | 3 +- 8 files changed, 117 insertions(+), 38 deletions(-) create mode 100644 policyengine_uk/utils/data_source.py diff --git a/changelog.d/built-from-dataset-imputation-gates.fixed.md b/changelog.d/built-from-dataset-imputation-gates.fixed.md index 2db9e7f4eb..69acb71256 100644 --- a/changelog.d/built-from-dataset-imputation-gates.fixed.md +++ b/changelog.d/built-from-dataset-imputation-gates.fixed.md @@ -1 +1 @@ -- UC deduction draws and private school attendance now apply to every simulation built from data, including a constituency or local authority filtered from the national data, instead of only those carrying over a million people of weight. A filtered region ranks private school attendance by its own income distribution. `filter_dataset` carries each person's private school attendance into the household it extracts. +- UC deduction draws and private school attendance now apply to every simulation built from data, including a constituency or local authority filtered from the national data, instead of only those carrying over a million people of weight. A filtered region ranks private school attendance by its own income distribution. `filter_dataset` carries each person's private school attendance into the household it extracts. Simulations rebuilt in place, and policyengine-core simulations built over the UK system from data, are recognised as data-built too. diff --git a/policyengine_uk/simulation.py b/policyengine_uk/simulation.py index cd39f80a3a..25dd7b9b5c 100644 --- a/policyengine_uk/simulation.py +++ b/policyengine_uk/simulation.py @@ -107,8 +107,10 @@ class Simulation(CoreSimulation): dataset = None # True when built from survey or other microdata rather than a situation # dictionary. Variables that impute unobserved detail across a population - # (such as months_since_last_birthday) read it; unlike the sum of weights, - # it stays true for a region or constituency filtered from the data. + # (such as months_since_last_birthday) read it through + # utils.data_source.built_from_data; unlike the sum of weights, it stays + # true for a region or constituency filtered from the data. The builders + # set it, so rebuilding a simulation in place keeps it right. built_from_dataset: bool = False def __init__( @@ -181,7 +183,6 @@ def __init__( self.build_from_dataset_source(get_default_dataset_url()) else: raise ValueError(f"Unsupported dataset type: {dataset.__class__}") - self.built_from_dataset = situation is None # Universal Credit reform (July 2025). Needs closer integration in the baseline, # but adding here for ease of toggling on/off via the 'active' parameter. @@ -270,6 +271,7 @@ def build_from_situation(self, situation: Dict) -> None: Args: situation: Dictionary describing household composition and characteristics """ + self.built_from_dataset = False self.build_from_populations(self.tax_benefit_system.instantiate_entities()) from policyengine_core.simulations.simulation_builder import ( SimulationBuilder, @@ -536,6 +538,8 @@ def build_from_ids( benunit_id: Array of benefit unit IDs household_id: Array of household IDs """ + # Every data source (DataFrame, dataset, file, URL) builds through here. + self.built_from_dataset = True from policyengine_core.simulations.simulation_builder import ( SimulationBuilder, ) # Import here to avoid circular dependency diff --git a/policyengine_uk/tests/test_data_built_imputations.py b/policyengine_uk/tests/test_data_built_imputations.py index e57f41e4bf..6d02202a37 100644 --- a/policyengine_uk/tests/test_data_built_imputations.py +++ b/policyengine_uk/tests/test_data_built_imputations.py @@ -19,6 +19,7 @@ UKSingleYearDataset, filter_dataset, ) +from policyengine_uk.utils.data_source import built_from_data from policyengine_uk.utils.stochastic import splitmix64_uniform YEAR = 2025 @@ -31,8 +32,32 @@ "uc_deductions", ) # Weight per household for a national-scale population: 400 households carry -# about 2.3m of weight, a quarter of them (one region) about 0.6m. -NATIONAL_SCALE = 5_000 +# about 1.6m of weight. One region (a quarter of them) carries about 0.4m, or +# 0.8m of people, under the old threshold of a million for both. +NATIONAL_SCALE = 3_500 +SITUATION = { + "people": { + "adult": {"age": {YEAR: 35}}, + "child": { + "age": {YEAR: 10}, + "attends_private_school_random_draw": {YEAR: 0}, + }, + }, + "benunits": { + "benunit": { + "members": ["adult", "child"], + "would_claim_uc": {YEAR: True}, + "universal_credit_pre_benefit_cap": {YEAR: 6_000}, + "benefit_cap_reduction": {YEAR: 0}, + } + }, + "households": { + "household": { + "members": ["adult", "child"], + "household_weight": {YEAR: 1e9}, + } + }, +} def tables(n: int = 400, weight_scale: float = 1.0) -> dict: @@ -72,7 +97,7 @@ def tables(n: int = 400, weight_scale: float = 1.0) -> dict: return {"person": person, "benunit": benunit, "household": household} -def data_simulation(t: dict) -> Microsimulation: +def multi_year(t: dict) -> UKMultiYearDataset: # Copies: building encodes enum columns in place. year = UKSingleYearDataset( person=t["person"].copy(), @@ -80,7 +105,11 @@ def data_simulation(t: dict) -> Microsimulation: household=t["household"].copy(), fiscal_year=YEAR, ) - return Microsimulation(dataset=UKMultiYearDataset(datasets=[year])) + return UKMultiYearDataset(datasets=[year]) + + +def data_simulation(t: dict) -> Microsimulation: + return Microsimulation(dataset=multi_year(t)) def row_filter(t: dict, keep) -> dict: @@ -140,8 +169,11 @@ def test_region_filtered_from_data_matches_the_national_run(national): deductions it has in the national simulation.""" t, sim = national region = row_filter(t, lambda h: h.region == "LONDON") - # A constituency-sized share of a national-sized population. - assert region["household"].household_weight.sum() < 1e6 + # A small share of a national-sized population: under the old threshold + # for benefit units, and for people (the old private school test summed + # household weight over people). + weight = region["household"].set_index("household_id").household_weight + assert weight.loc[region["person"].person_household_id].sum() < 1e6 regional = data_simulation(region) kept = np.isin(t["benunit"].benunit_id, region["benunit"].benunit_id) for variable in UC_OUTPUTS: @@ -196,31 +228,56 @@ def test_data_without_weight_attends_no_private_school(): def test_situations_get_defaults_whatever_their_weight(): """A household situation is not data, even with the weight of a nation.""" - situation = { - "people": { - "adult": {"age": {YEAR: 35}}, - "child": { - "age": {YEAR: 10}, - "attends_private_school_random_draw": {YEAR: 0}, - }, - }, - "benunits": { - "benunit": { - "members": ["adult", "child"], - "would_claim_uc": {YEAR: True}, - "universal_credit_pre_benefit_cap": {YEAR: 6_000}, - "benefit_cap_reduction": {YEAR: 0}, - } - }, - "households": { - "household": { - "members": ["adult", "child"], - "household_weight": {YEAR: 1e9}, - } - }, - } - sim = Simulation(situation=situation) + sim = Simulation(situation=SITUATION) + assert not built_from_data(sim) for draw in DRAWS: assert values(sim, draw)[0] == 1.0 assert values(sim, "uc_deductions")[0] == 0 assert not values(sim, "attends_private_school").any() + + +def test_rebuilding_in_place_follows_the_new_source(): + """The builders set the flag, so a data simulation rebuilt from a + situation gets household defaults, and a situation rebuilt from data gets + the imputations.""" + data = tables(n=40, weight_scale=NATIONAL_SCALE) + sim = data_simulation(data) + sim.build_from_situation(SITUATION) + assert not built_from_data(sim) + for draw in DRAWS: + assert values(sim, draw)[0] == 1.0 + assert not values(sim, "attends_private_school").any() + + sim = Simulation(situation=SITUATION) + sim.build_from_multi_year_dataset(multi_year(data)) + assert built_from_data(sim) + ids = values(sim, "benunit_id") + for salt, draw in enumerate(DRAWS): + expected = splitmix64_uniform(ids, salt=salt).astype(np.float32) + assert np.array_equal(values(sim, draw), expected), draw + assert values(sim, "attends_private_school").any() + + +def test_core_simulation_over_data_gets_imputations(): + """A policyengine-core Simulation built over the UK system from data + records is_over_dataset rather than built_from_dataset.""" + from policyengine_core.simulations import Simulation as CoreSimulation + + from policyengine_uk.system import system + + t = tables(n=40, weight_scale=NATIONAL_SCALE) + frame = ( + t["person"] + .merge(t["benunit"], left_on="person_benunit_id", right_on="benunit_id") + .merge(t["household"], left_on="person_household_id", right_on="household_id") + ) + sim = CoreSimulation( + tax_benefit_system=system, + dataset=frame.rename(columns=lambda column: f"{column}__{YEAR}"), + ) + assert built_from_data(sim) + ids = values(sim, "benunit_id") + for salt, draw in enumerate(DRAWS): + expected = splitmix64_uniform(ids, salt=salt).astype(np.float32) + assert np.array_equal(values(sim, draw), expected), draw + assert values(sim, "attends_private_school").any() diff --git a/policyengine_uk/utils/data_source.py b/policyengine_uk/utils/data_source.py new file mode 100644 index 0000000000..cbc9656906 --- /dev/null +++ b/policyengine_uk/utils/data_source.py @@ -0,0 +1,14 @@ +def built_from_data(simulation) -> bool: + """Whether a simulation was built from survey or other microdata rather + than from a situation dictionary. + + Variables that impute unobserved detail across a population read this, + not the sum of weights: a region or constituency filtered from the data + carries little weight and is still data. policyengine_uk.Simulation + records it as ``built_from_dataset``; a policyengine-core Simulation built + over the UK system records ``is_over_dataset``. + """ + flag = getattr(simulation, "built_from_dataset", None) + if flag is None: + flag = getattr(simulation, "is_over_dataset", False) + return bool(flag) diff --git a/policyengine_uk/variables/contrib/labour/attends_private_school.py b/policyengine_uk/variables/contrib/labour/attends_private_school.py index 63f93d011c..aedb920d9d 100644 --- a/policyengine_uk/variables/contrib/labour/attends_private_school.py +++ b/policyengine_uk/variables/contrib/labour/attends_private_school.py @@ -1,4 +1,5 @@ from policyengine_uk.model_api import * +from policyengine_uk.utils.data_source import built_from_data def interpolate_percentile(param, percentile): @@ -39,7 +40,7 @@ def formula(person, period, parameters): # constituency filtered from them. A household situation has no # income distribution to rank within, so it attends no private school # unless set. - if not getattr(person.simulation, "built_from_dataset", False): + if not built_from_data(person.simulation): return 0 household = person.household # To ensure that our model matches diff --git a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py index 7cb5a53aa8..8e791bb96b 100644 --- a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py +++ b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py @@ -1,4 +1,5 @@ from policyengine_uk.model_api import * +from policyengine_uk.utils.data_source import built_from_data from policyengine_uk.utils.stochastic import splitmix64_uniform @@ -21,7 +22,7 @@ def formula(benunit, period, parameters): # weight it carries: a constituency filtered from the national data is # still data. Household situations get 1.0, which never falls below # any incidence, so calculators get no deductions unless set. - if not getattr(benunit.simulation, "built_from_dataset", False): + if not built_from_data(benunit.simulation): return np.ones(benunit.count) ids = benunit("benunit_id", period) return splitmix64_uniform(ids, salt=0) diff --git a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py index 89f2020991..1472d9de31 100644 --- a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py +++ b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py @@ -1,4 +1,5 @@ from policyengine_uk.model_api import * +from policyengine_uk.utils.data_source import built_from_data from policyengine_uk.utils.stochastic import splitmix64_uniform @@ -20,7 +21,7 @@ def formula(benunit, period, parameters): # Hashed draws in every simulation built from data, however little # weight it carries. Household situations get 1.0, which maps to the # last type combination but only matters when a deduction is assigned. - if not getattr(benunit.simulation, "built_from_dataset", False): + if not built_from_data(benunit.simulation): return np.ones(benunit.count) ids = benunit("benunit_id", period) return splitmix64_uniform(ids, salt=1) diff --git a/policyengine_uk/variables/household/demographic/months_since_last_birthday.py b/policyengine_uk/variables/household/demographic/months_since_last_birthday.py index d8a037891b..b5a35bc6a6 100644 --- a/policyengine_uk/variables/household/demographic/months_since_last_birthday.py +++ b/policyengine_uk/variables/household/demographic/months_since_last_birthday.py @@ -1,4 +1,5 @@ from policyengine_uk.model_api import * +from policyengine_uk.utils.data_source import built_from_data from policyengine_uk.utils.stochastic import splitmix64_uniform, stratified_uniform @@ -30,7 +31,7 @@ def formula(person, period, parameters): age = person("age", period) whole_years = np.floor(age) fraction = age - whole_years - if getattr(person.simulation, "built_from_dataset", False): + if built_from_data(person.simulation): position = stratified_uniform( strata=whole_years * 2 + person("is_male", period), draws=splitmix64_uniform(person("person_id", period), salt=2), From e951934977ab0e1f1ca86775eea00acb07ee60b3 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 06:16:22 -0400 Subject: [PATCH 4/4] Drop caches for the old population when rebuilding in place A calculated clone rebuilt from data kept core's simulation-level _fast_cache, so months_since_last_birthday (now enabled by the data-built flag) read a two-person person_weight against an 80-person population and raised. build_from_situation and build_from_ids now record the source and reset _fast_cache and _user_input_keys (a clone shares the latter with its original). The rebuild test calculates on a clone before rebuilding, in both directions. Co-Authored-By: Claude Opus 5.5 --- policyengine_uk/simulation.py | 15 +++++++++-- .../tests/test_data_built_imputations.py | 27 +++++++++++++++---- 2 files changed, 35 insertions(+), 7 deletions(-) diff --git a/policyengine_uk/simulation.py b/policyengine_uk/simulation.py index 25dd7b9b5c..032fb6fdea 100644 --- a/policyengine_uk/simulation.py +++ b/policyengine_uk/simulation.py @@ -271,7 +271,7 @@ def build_from_situation(self, situation: Dict) -> None: Args: situation: Dictionary describing household composition and characteristics """ - self.built_from_dataset = False + self._start_new_population(built_from_dataset=False) self.build_from_populations(self.tax_benefit_system.instantiate_entities()) from policyengine_core.simulations.simulation_builder import ( SimulationBuilder, @@ -521,6 +521,17 @@ def build_from_multi_year_dataset(self, dataset: UKMultiYearDataset) -> None: self.dataset = dataset + def _start_new_population(self, built_from_dataset: bool) -> None: + """Record the new population's source, and drop what the simulation + cached for the previous one, so rebuilding in place (e.g. a clone) + never reads arrays sized for the old population.""" + self.built_from_dataset = built_from_dataset + if getattr(self, "_fast_cache", None) is not None: + self._fast_cache = {} + if getattr(self, "_user_input_keys", None) is not None: + # A clone shares this set with its original, so replace it. + self._user_input_keys = set() + def build_from_ids( self, person_id: np.ndarray, @@ -539,7 +550,7 @@ def build_from_ids( household_id: Array of household IDs """ # Every data source (DataFrame, dataset, file, URL) builds through here. - self.built_from_dataset = True + self._start_new_population(built_from_dataset=True) from policyengine_core.simulations.simulation_builder import ( SimulationBuilder, ) # Import here to avoid circular dependency diff --git a/policyengine_uk/tests/test_data_built_imputations.py b/policyengine_uk/tests/test_data_built_imputations.py index 6d02202a37..8524ec36c2 100644 --- a/policyengine_uk/tests/test_data_built_imputations.py +++ b/policyengine_uk/tests/test_data_built_imputations.py @@ -236,21 +236,38 @@ def test_situations_get_defaults_whatever_their_weight(): assert not values(sim, "attends_private_school").any() +def calculate_imputations(sim) -> None: + """Fill the simulation's caches for its current population.""" + for variable in ( + *DRAWS, + "attends_private_school", + "months_since_last_birthday", + "person_weight", + "is_male", + ): + values(sim, variable) + + def test_rebuilding_in_place_follows_the_new_source(): - """The builders set the flag, so a data simulation rebuilt from a - situation gets household defaults, and a situation rebuilt from data gets - the imputations.""" + """The builders set the flag and drop what was cached for the old + population, so a calculated clone rebuilt from a situation gets household + defaults, and one rebuilt from data gets the imputations.""" data = tables(n=40, weight_scale=NATIONAL_SCALE) - sim = data_simulation(data) + sim = data_simulation(data).clone() + calculate_imputations(sim) sim.build_from_situation(SITUATION) assert not built_from_data(sim) for draw in DRAWS: assert values(sim, draw)[0] == 1.0 assert not values(sim, "attends_private_school").any() - sim = Simulation(situation=SITUATION) + assert values(sim, "months_since_last_birthday").tolist() == [6, 6] + + sim = Simulation(situation=SITUATION).clone() + calculate_imputations(sim) sim.build_from_multi_year_dataset(multi_year(data)) assert built_from_data(sim) + assert values(sim, "months_since_last_birthday").shape == (80,) ids = values(sim, "benunit_id") for salt, draw in enumerate(DRAWS): expected = splitmix64_uniform(ids, salt=salt).astype(np.float32)