diff --git a/changelog.d/built-from-dataset-imputation-gates.fixed.md b/changelog.d/built-from-dataset-imputation-gates.fixed.md new file mode 100644 index 0000000000..69acb71256 --- /dev/null +++ b/changelog.d/built-from-dataset-imputation-gates.fixed.md @@ -0,0 +1 @@ +- UC deduction draws and private school attendance now apply to every simulation built from data, including a constituency or local authority filtered from the national data, instead of only those carrying over a million people of weight. A filtered region ranks private school attendance by its own income distribution. `filter_dataset` carries each person's private school attendance into the household it extracts. Simulations rebuilt in place, and policyengine-core simulations built over the UK system from data, are recognised as data-built too. diff --git a/docs/book/validation/uc-deductions.md b/docs/book/validation/uc-deductions.md index 5d7d4fc4a4..7264e11ee0 100644 --- a/docs/book/validation/uc-deductions.md +++ b/docs/book/validation/uc-deductions.md @@ -81,4 +81,4 @@ A cap of 1 − *x* is equivalent to a protected minimum floor at *x* of the stan Statutory parameters — the cap, the protected floor, the minimum payable penny, the abolition switches — live under `gov.dwp.universal_credit.deductions`. The calibrated distributions live under `gov.simulation.uc_deductions`: they describe the world, not the law. -The assignment formulas are an explicit fallback. The end state imputes `uc_latent_deduction_rate` and `uc_deduction_combination` at dataset build, at which point the model consumes them as plain inputs and the fallback retires; raw-FRS users keep working through the fallback until then. Assignment uses deterministic splitmix64 hashes of `benunit_id`, reproducible across runs and machines, and overridable by datasets or situations through `uc_deduction_random_draw` and `uc_deduction_type_random_draw`. Single-household simulations get no deductions unless set explicitly. +The assignment formulas are an explicit fallback. The end state imputes `uc_latent_deduction_rate` and `uc_deduction_combination` at dataset build, at which point the model consumes them as plain inputs and the fallback retires; raw-FRS users keep working through the fallback until then. Assignment uses deterministic splitmix64 hashes of `benunit_id`, reproducible across runs and machines, and overridable by datasets or situations through `uc_deduction_random_draw` and `uc_deduction_type_random_draw`. Every simulation built from data gets the hashed draws, including a constituency or local authority filtered from the national data, and a benefit unit gets the same draw there as in the national run. Household situations get no deductions unless set explicitly. diff --git a/policyengine_uk/data/filter_dataset.py b/policyengine_uk/data/filter_dataset.py index 3b70dd4c0d..51e1a7590f 100644 --- a/policyengine_uk/data/filter_dataset.py +++ b/policyengine_uk/data/filter_dataset.py @@ -20,6 +20,12 @@ def filter_dataset( This function creates a new dataset containing only the specified household and the associated benefit units and people within that household. + Values imputed across the whole population (months_since_last_birthday and + attends_private_school) are taken from sim for the given year and carried + in as inputs, so the extract keeps them in later years too. Private school + attendance ranks incomes, so the first extract from a simulation computes + its taxes and benefits. + Parameters ---------- sim : Microsimulation @@ -37,17 +43,17 @@ def filter_dataset( dataset: UKSingleYearDataset = sim.dataset[year] new_dataset = dataset.copy() person = new_dataset.person - if "months_since_last_birthday" not in person.columns: - # Birthdays are spread over the year across the whole population, so - # carry each person's place across rather than recompute it for one - # household. - months = pd.Series( - np.asarray(sim.calculate("months_since_last_birthday", year)), - index=np.asarray(sim.calculate("person_id", year)), - ) - person = person.assign( - months_since_last_birthday=months.loc[person.person_id].values - ) + # These are imputed across the whole population: birthdays are spread over + # the year, and private school attendance follows each household's income + # percentile (one household alone would rank at the 100th). Carry each + # person's value across rather than recompute it for one household. + for variable in ("months_since_last_birthday", "attends_private_school"): + if variable not in person.columns: + values = pd.Series( + np.asarray(sim.calculate(variable, year)), + index=np.asarray(sim.calculate("person_id", year)), + ) + person = person.assign(**{variable: values.loc[person.person_id].values}) new_dataset.person = person[person.person_household_id == household_id] new_dataset.household = new_dataset.household[ new_dataset.household.household_id == household_id diff --git a/policyengine_uk/simulation.py b/policyengine_uk/simulation.py index cd39f80a3a..032fb6fdea 100644 --- a/policyengine_uk/simulation.py +++ b/policyengine_uk/simulation.py @@ -107,8 +107,10 @@ class Simulation(CoreSimulation): dataset = None # True when built from survey or other microdata rather than a situation # dictionary. Variables that impute unobserved detail across a population - # (such as months_since_last_birthday) read it; unlike the sum of weights, - # it stays true for a region or constituency filtered from the data. + # (such as months_since_last_birthday) read it through + # utils.data_source.built_from_data; unlike the sum of weights, it stays + # true for a region or constituency filtered from the data. The builders + # set it, so rebuilding a simulation in place keeps it right. built_from_dataset: bool = False def __init__( @@ -181,7 +183,6 @@ def __init__( self.build_from_dataset_source(get_default_dataset_url()) else: raise ValueError(f"Unsupported dataset type: {dataset.__class__}") - self.built_from_dataset = situation is None # Universal Credit reform (July 2025). Needs closer integration in the baseline, # but adding here for ease of toggling on/off via the 'active' parameter. @@ -270,6 +271,7 @@ def build_from_situation(self, situation: Dict) -> None: Args: situation: Dictionary describing household composition and characteristics """ + self._start_new_population(built_from_dataset=False) self.build_from_populations(self.tax_benefit_system.instantiate_entities()) from policyengine_core.simulations.simulation_builder import ( SimulationBuilder, @@ -519,6 +521,17 @@ def build_from_multi_year_dataset(self, dataset: UKMultiYearDataset) -> None: self.dataset = dataset + def _start_new_population(self, built_from_dataset: bool) -> None: + """Record the new population's source, and drop what the simulation + cached for the previous one, so rebuilding in place (e.g. a clone) + never reads arrays sized for the old population.""" + self.built_from_dataset = built_from_dataset + if getattr(self, "_fast_cache", None) is not None: + self._fast_cache = {} + if getattr(self, "_user_input_keys", None) is not None: + # A clone shares this set with its original, so replace it. + self._user_input_keys = set() + def build_from_ids( self, person_id: np.ndarray, @@ -536,6 +549,8 @@ def build_from_ids( benunit_id: Array of benefit unit IDs household_id: Array of household IDs """ + # Every data source (DataFrame, dataset, file, URL) builds through here. + self._start_new_population(built_from_dataset=True) from policyengine_core.simulations.simulation_builder import ( SimulationBuilder, ) # Import here to avoid circular dependency diff --git a/policyengine_uk/tests/policy/baseline/contrib/labour/attends_private_school.yaml b/policyengine_uk/tests/policy/baseline/contrib/labour/attends_private_school.yaml index da00699567..0db7ac9ade 100644 --- a/policyengine_uk/tests/policy/baseline/contrib/labour/attends_private_school.yaml +++ b/policyengine_uk/tests/policy/baseline/contrib/labour/attends_private_school.yaml @@ -1,41 +1,32 @@ -- name: Attends private school returns False when attendance rate is 0% +- name: A household situation attends no private school unless set, even at the top of the income distribution period: 2024 input: - gov.simulation.private_school_vat.private_school_attendance_rate.100: 0 - gov.simulation.private_school_vat.private_school_attendance_rate.95: 0 people: - adult: + adult: age: 25 child: age: 10 + attends_private_school_random_draw: 0 households: household: - household_weight: 0.001 + household_weight: 1_000_000_000 household_market_income: 1_000_000_000 household_benefits: 0 - members: [ - adult, - child - ] + members: [adult, child] output: attends_private_school: [False, False] -- name: Attends private school successfully returns False when household weights are 0 +- name: A household situation keeps attendance it sets period: 2024 input: people: - adult: + adult: age: 25 child: age: 10 + attends_private_school: true households: household: - household_weight: 0 - household_market_income: 1_000_000_000 - household_benefits: 0 - members: [ - adult, - child - ] + members: [adult, child] output: - attends_private_school: [False, False] + attends_private_school: [False, True] diff --git a/policyengine_uk/tests/test_data_built_imputations.py b/policyengine_uk/tests/test_data_built_imputations.py new file mode 100644 index 0000000000..8524ec36c2 --- /dev/null +++ b/policyengine_uk/tests/test_data_built_imputations.py @@ -0,0 +1,300 @@ +"""Imputations that apply to simulations built from data, not to situations. + +UC deduction draws and private school attendance are imputed across a +population. How the simulation was built decides whether they apply, not how +much weight it carries: a constituency or local authority filtered from the +national data (as policyengine.py's RowFilterStrategy does) carries well under +a million people of weight and is still data. +""" + +import numpy as np +import pandas as pd +import pytest +from hypothesis import given, settings +from hypothesis import strategies as st + +from policyengine_uk import Microsimulation, Simulation +from policyengine_uk.data import ( + UKMultiYearDataset, + UKSingleYearDataset, + filter_dataset, +) +from policyengine_uk.utils.data_source import built_from_data +from policyengine_uk.utils.stochastic import splitmix64_uniform + +YEAR = 2025 +REGIONS = np.array(["LONDON", "NORTH_EAST", "SCOTLAND", "WALES"]) +DRAWS = ("uc_deduction_random_draw", "uc_deduction_type_random_draw") +UC_OUTPUTS = ( + *DRAWS, + "uc_has_deduction", + "uc_deduction_combination", + "uc_deductions", +) +# Weight per household for a national-scale population: 400 households carry +# about 1.6m of weight. One region (a quarter of them) carries about 0.4m, or +# 0.8m of people, under the old threshold of a million for both. +NATIONAL_SCALE = 3_500 +SITUATION = { + "people": { + "adult": {"age": {YEAR: 35}}, + "child": { + "age": {YEAR: 10}, + "attends_private_school_random_draw": {YEAR: 0}, + }, + }, + "benunits": { + "benunit": { + "members": ["adult", "child"], + "would_claim_uc": {YEAR: True}, + "universal_credit_pre_benefit_cap": {YEAR: 6_000}, + "benefit_cap_reduction": {YEAR: 0}, + } + }, + "households": { + "household": { + "members": ["adult", "child"], + "household_weight": {YEAR: 1e9}, + } + }, +} + + +def tables(n: int = 400, weight_scale: float = 1.0) -> dict: + """n households of one adult and one child, all on Universal Credit, with + incomes spread from 0 to 150k and draws for private school attendance.""" + ids = np.arange(n) + rng = np.random.default_rng(0) + person = pd.DataFrame( + { + "person_id": np.concatenate([ids * 10 + 1, ids * 10 + 2]), + "person_benunit_id": np.concatenate([ids, ids]), + "person_household_id": np.concatenate([ids, ids]), + "age": np.concatenate([np.full(n, 35), np.full(n, 10)]), + "employment_income": np.concatenate( + [np.linspace(0, 150_000, n), np.zeros(n)] + ), + "attends_private_school_random_draw": np.concatenate( + [np.ones(n), rng.uniform(0, 0.5, n)] + ), + } + ) + benunit = pd.DataFrame( + { + "benunit_id": ids, + "would_claim_uc": np.ones(n, dtype=bool), + "universal_credit_pre_benefit_cap": np.full(n, 6_000.0), + "benefit_cap_reduction": np.zeros(n), + } + ) + household = pd.DataFrame( + { + "household_id": ids, + "household_weight": weight_scale * rng.lognormal(0, 0.5, n), + "region": REGIONS[ids % len(REGIONS)], + } + ) + return {"person": person, "benunit": benunit, "household": household} + + +def multi_year(t: dict) -> UKMultiYearDataset: + # Copies: building encodes enum columns in place. + year = UKSingleYearDataset( + person=t["person"].copy(), + benunit=t["benunit"].copy(), + household=t["household"].copy(), + fiscal_year=YEAR, + ) + return UKMultiYearDataset(datasets=[year]) + + +def data_simulation(t: dict) -> Microsimulation: + return Microsimulation(dataset=multi_year(t)) + + +def row_filter(t: dict, keep) -> dict: + """Keep households where keep(household table) holds, with their benefit + units and people: what policyengine.py does for a constituency.""" + household = t["household"][keep(t["household"])] + person = t["person"][t["person"].person_household_id.isin(household.household_id)] + benunit = t["benunit"][t["benunit"].benunit_id.isin(person.person_benunit_id)] + return { + "person": person.reset_index(drop=True), + "benunit": benunit.reset_index(drop=True), + "household": household.reset_index(drop=True), + } + + +def values(sim, variable: str) -> np.ndarray: + return np.asarray(sim.calculate(variable, YEAR)) + + +@pytest.fixture(scope="module") +def national(): + t = tables(weight_scale=NATIONAL_SCALE) + assert t["household"].household_weight.sum() > 1e6 + return t, data_simulation(t) + + +def test_small_data_built_simulation_gets_hashed_draws(): + small = tables(weight_scale=1e-3) + assert small["household"].household_weight.sum() < 1e6 + sim = data_simulation(small) + ids = small["benunit"].benunit_id.values + for salt, draw in enumerate(DRAWS): + expected = splitmix64_uniform(ids, salt=salt).astype(np.float32) + assert np.array_equal(values(sim, draw), expected), draw + # Every benefit unit is on UC, so some have deductions. + assert values(sim, "uc_deductions").max() > 0 + assert values(sim, "attends_private_school").any() + + +@settings(max_examples=6, deadline=None) +@given(power=st.integers(min_value=-14, max_value=30)) +def test_imputations_do_not_depend_on_total_weight(national, power): + """Scaling every weight by the same factor changes nothing: hashed draws + depend only on ids, and income percentiles only on relative weights. + Powers of two scale exactly in floating point, so this holds exactly for + totals from about a hundred to about 10^15.""" + t, reference = national + sim = data_simulation(tables(weight_scale=NATIONAL_SCALE * 2.0**power)) + for variable in (*UC_OUTPUTS, "attends_private_school"): + assert np.array_equal(values(sim, variable), values(reference, variable)), ( + variable + ) + + +def test_region_filtered_from_data_matches_the_national_run(national): + """Every benefit unit in a region filtered from the data has the UC + deductions it has in the national simulation.""" + t, sim = national + region = row_filter(t, lambda h: h.region == "LONDON") + # A small share of a national-sized population: under the old threshold + # for benefit units, and for people (the old private school test summed + # household weight over people). + weight = region["household"].set_index("household_id").household_weight + assert weight.loc[region["person"].person_household_id].sum() < 1e6 + regional = data_simulation(region) + kept = np.isin(t["benunit"].benunit_id, region["benunit"].benunit_id) + for variable in UC_OUTPUTS: + assert np.array_equal( + values(regional, variable), values(sim, variable)[kept] + ), variable + assert values(regional, "uc_deductions").max() > 0 + # Private school attendance ranks incomes within the simulated population, + # so a region ranks against itself; it is still imputed. + assert values(regional, "attends_private_school").any() + + +def test_extracted_household_keeps_its_imputations(national): + """filter_dataset extracts one household from the data. Its UC draws hash + the same ids, and it carries its private school attendance across rather + than ranking at the 100th percentile of a population of one.""" + t, sim = national + attends = values(sim, "attends_private_school") + person_household = values(sim, "person_household_id") + deductions = pd.Series( + values(sim, "uc_deductions"), index=values(sim, "benunit_id") + ) + # One benefit unit per household, sharing its id. + with_deductions = deductions.index[deductions > 0] + attending = np.unique(person_household[attends]) + not_attending = np.setdiff1d(values(sim, "household_id"), attending) + assert len(with_deductions) and len(attending) + for household in [*with_deductions[:3], *attending[:3], *not_attending[-3:]]: + extract = filter_dataset(sim, household_id=int(household), year=YEAR) + extracted = Microsimulation(dataset=UKMultiYearDataset(datasets=[extract])) + assert np.array_equal( + values(extracted, "uc_deductions"), + deductions.loc[values(extracted, "benunit_id")].values, + ) + assert np.array_equal( + values(extracted, "attends_private_school"), + attends[person_household == household], + ) + + +def test_data_without_weight_attends_no_private_school(): + """Every household at zero weight: no income ranking is possible, so no + household reaches a percentile with a positive rate, and nothing fails.""" + sim = data_simulation(tables(n=40, weight_scale=0)) + assert not values(sim, "attends_private_school").any() + ids = values(sim, "benunit_id") + assert np.array_equal( + values(sim, "uc_deduction_random_draw"), + splitmix64_uniform(ids, salt=0).astype(np.float32), + ) + + +def test_situations_get_defaults_whatever_their_weight(): + """A household situation is not data, even with the weight of a nation.""" + sim = Simulation(situation=SITUATION) + assert not built_from_data(sim) + for draw in DRAWS: + assert values(sim, draw)[0] == 1.0 + assert values(sim, "uc_deductions")[0] == 0 + assert not values(sim, "attends_private_school").any() + + +def calculate_imputations(sim) -> None: + """Fill the simulation's caches for its current population.""" + for variable in ( + *DRAWS, + "attends_private_school", + "months_since_last_birthday", + "person_weight", + "is_male", + ): + values(sim, variable) + + +def test_rebuilding_in_place_follows_the_new_source(): + """The builders set the flag and drop what was cached for the old + population, so a calculated clone rebuilt from a situation gets household + defaults, and one rebuilt from data gets the imputations.""" + data = tables(n=40, weight_scale=NATIONAL_SCALE) + sim = data_simulation(data).clone() + calculate_imputations(sim) + sim.build_from_situation(SITUATION) + assert not built_from_data(sim) + for draw in DRAWS: + assert values(sim, draw)[0] == 1.0 + assert not values(sim, "attends_private_school").any() + + assert values(sim, "months_since_last_birthday").tolist() == [6, 6] + + sim = Simulation(situation=SITUATION).clone() + calculate_imputations(sim) + sim.build_from_multi_year_dataset(multi_year(data)) + assert built_from_data(sim) + assert values(sim, "months_since_last_birthday").shape == (80,) + ids = values(sim, "benunit_id") + for salt, draw in enumerate(DRAWS): + expected = splitmix64_uniform(ids, salt=salt).astype(np.float32) + assert np.array_equal(values(sim, draw), expected), draw + assert values(sim, "attends_private_school").any() + + +def test_core_simulation_over_data_gets_imputations(): + """A policyengine-core Simulation built over the UK system from data + records is_over_dataset rather than built_from_dataset.""" + from policyengine_core.simulations import Simulation as CoreSimulation + + from policyengine_uk.system import system + + t = tables(n=40, weight_scale=NATIONAL_SCALE) + frame = ( + t["person"] + .merge(t["benunit"], left_on="person_benunit_id", right_on="benunit_id") + .merge(t["household"], left_on="person_household_id", right_on="household_id") + ) + sim = CoreSimulation( + tax_benefit_system=system, + dataset=frame.rename(columns=lambda column: f"{column}__{YEAR}"), + ) + assert built_from_data(sim) + ids = values(sim, "benunit_id") + for salt, draw in enumerate(DRAWS): + expected = splitmix64_uniform(ids, salt=salt).astype(np.float32) + assert np.array_equal(values(sim, draw), expected), draw + assert values(sim, "attends_private_school").any() diff --git a/policyengine_uk/utils/data_source.py b/policyengine_uk/utils/data_source.py new file mode 100644 index 0000000000..cbc9656906 --- /dev/null +++ b/policyengine_uk/utils/data_source.py @@ -0,0 +1,14 @@ +def built_from_data(simulation) -> bool: + """Whether a simulation was built from survey or other microdata rather + than from a situation dictionary. + + Variables that impute unobserved detail across a population read this, + not the sum of weights: a region or constituency filtered from the data + carries little weight and is still data. policyengine_uk.Simulation + records it as ``built_from_dataset``; a policyengine-core Simulation built + over the UK system records ``is_over_dataset``. + """ + flag = getattr(simulation, "built_from_dataset", None) + if flag is None: + flag = getattr(simulation, "is_over_dataset", False) + return bool(flag) diff --git a/policyengine_uk/variables/contrib/labour/attends_private_school.py b/policyengine_uk/variables/contrib/labour/attends_private_school.py index 6bd463ec02..aedb920d9d 100644 --- a/policyengine_uk/variables/contrib/labour/attends_private_school.py +++ b/policyengine_uk/variables/contrib/labour/attends_private_school.py @@ -1,4 +1,5 @@ from policyengine_uk.model_api import * +from policyengine_uk.utils.data_source import built_from_data def interpolate_percentile(param, percentile): @@ -35,7 +36,11 @@ class attends_private_school(Variable): value_type = bool def formula(person, period, parameters): - if not hasattr(person.simulation, "dataset"): + # Imputed only in simulations built from data, including a region or + # constituency filtered from them. A household situation has no + # income distribution to rank within, so it attends no private school + # unless set. + if not built_from_data(person.simulation): return 0 household = person.household # To ensure that our model matches @@ -65,18 +70,20 @@ def formula(person, period, parameters): household_weight = household("household_weight", period) weighted_income = MicroSeries(net_income, weights=household_weight) - if household_weight.sum() < 1e6: - return 0 - + # Percentiles rank households within the simulated population, so a + # region filtered from the data ranks against itself, not the UK. + # Households without weight stay at percentile 0 (a rate of 0 unless + # reformed), including when no household has weight. percentile = np.zeros_like(weighted_income).astype(numpy.int64) mask = household_weight > 0 - percentile[mask] = ( - weighted_income[mask] - .percentile_rank() - .clip(0, 100) - .values.astype(numpy.int64) - ) + if mask.any(): + percentile[mask] = ( + weighted_income[mask] + .percentile_rank() + .clip(0, 100) + .values.astype(numpy.int64) + ) # STUDENT_POPULATION_ADJUSTMENT_FACTOR = 0.78 STUDENT_POPULATION_ADJUSTMENT_FACTOR = population_adjustment_factor diff --git a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py index 22c73c1c30..8e791bb96b 100644 --- a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py +++ b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_random_draw.py @@ -1,4 +1,5 @@ from policyengine_uk.model_api import * +from policyengine_uk.utils.data_source import built_from_data from policyengine_uk.utils.stochastic import splitmix64_uniform @@ -6,9 +7,10 @@ class uc_deduction_random_draw(Variable): label = "UC deduction random draw" documentation = ( "Uniform draw on [0, 1) determining deduction incidence and size. " - "Deterministic hash of the benefit unit id in dataset simulations; " - "1.0 in single-household simulations, so no deduction unless set. " - "Datasets and situations can override it directly." + "Deterministic hash of the benefit unit id in simulations built from " + "data, including a region or constituency filtered from them; 1.0 in " + "household situations, so no deduction unless set. Datasets and " + "situations can override it directly." ) entity = BenUnit definition_period = YEAR @@ -16,11 +18,11 @@ class uc_deduction_random_draw(Variable): default_value = 1.0 def formula(benunit, period, parameters): - # Representative microdata carries tens of millions of households of - # weight; single-household situations carry ~1. Only assign hashed - # draws in representative simulations: the 1.0 default never falls - # below any incidence, so calculators get no deductions unless set. - if benunit("benunit_weight", period).sum() < 1e6: + # Hashed draws in every simulation built from data, however little + # weight it carries: a constituency filtered from the national data is + # still data. Household situations get 1.0, which never falls below + # any incidence, so calculators get no deductions unless set. + if not built_from_data(benunit.simulation): return np.ones(benunit.count) ids = benunit("benunit_id", period) return splitmix64_uniform(ids, salt=0) diff --git a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py index 284739d134..1472d9de31 100644 --- a/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py +++ b/policyengine_uk/variables/gov/dwp/universal_credit/deductions/uc_deduction_type_random_draw.py @@ -1,4 +1,5 @@ from policyengine_uk.model_api import * +from policyengine_uk.utils.data_source import built_from_data from policyengine_uk.utils.stochastic import splitmix64_uniform @@ -6,9 +7,10 @@ class uc_deduction_type_random_draw(Variable): label = "UC deduction type random draw" documentation = ( "Uniform draw on [0, 1) determining the deduction type combination. " - "Deterministic hash of the benefit unit id in dataset simulations; " - "1.0 in single-household simulations. Datasets and situations can " - "override it directly." + "Deterministic hash of the benefit unit id in simulations built from " + "data, including a region or constituency filtered from them; 1.0 in " + "household situations. Datasets and situations can override it " + "directly." ) entity = BenUnit definition_period = YEAR @@ -16,11 +18,10 @@ class uc_deduction_type_random_draw(Variable): default_value = 1.0 def formula(benunit, period, parameters): - # Representative microdata carries tens of millions of households of - # weight; single-household situations carry ~1. Only assign hashed - # draws in representative simulations; 1.0 maps to the last type - # combination but only matters when a deduction is assigned. - if benunit("benunit_weight", period).sum() < 1e6: + # Hashed draws in every simulation built from data, however little + # weight it carries. Household situations get 1.0, which maps to the + # last type combination but only matters when a deduction is assigned. + if not built_from_data(benunit.simulation): return np.ones(benunit.count) ids = benunit("benunit_id", period) return splitmix64_uniform(ids, salt=1) diff --git a/policyengine_uk/variables/household/demographic/months_since_last_birthday.py b/policyengine_uk/variables/household/demographic/months_since_last_birthday.py index d8a037891b..b5a35bc6a6 100644 --- a/policyengine_uk/variables/household/demographic/months_since_last_birthday.py +++ b/policyengine_uk/variables/household/demographic/months_since_last_birthday.py @@ -1,4 +1,5 @@ from policyengine_uk.model_api import * +from policyengine_uk.utils.data_source import built_from_data from policyengine_uk.utils.stochastic import splitmix64_uniform, stratified_uniform @@ -30,7 +31,7 @@ def formula(person, period, parameters): age = person("age", period) whole_years = np.floor(age) fraction = age - whole_years - if getattr(person.simulation, "built_from_dataset", False): + if built_from_data(person.simulation): position = stratified_uniform( strata=whole_years * 2 + person("is_male", period), draws=splitmix64_uniform(person("person_id", period), salt=2),