From 96672af5f52586a279d40d37cf83816c3b2ab07e Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 2 Oct 2026 01:16:36 -0400 Subject: [PATCH 1/4] Stop SPI-synthetic rows carrying benefit claims nobody observed The second-stage QRF draws each SPI-donor person's benefit reports from age, gender, region and incomes, with no view of the benefit unit or of health. Those reports then act as existing claims in policyengine-uk. The take-up anchors, receives_benefits_in_own_right and ssmg_reported were also left as the FRS donor's. On SPI-donor rows, this zeroes the income-related awards (UC, Pension Credit, Housing Benefit, CTR, IS, tax credits, income-related ESA and JSA, SSMG), the out-of-work benefits (contributory ESA and JSA, incapacity benefit, SDA) and Child Benefit, whose only use is the take-up anchor. It then rebuilds the anchors and receives_benefits_in_own_right from the rows' own reports. The zeroed columns stay in the QRF chain, so the values of the reports kept do not change. Also adds hypothesis as a dev extra, for the property tests. Co-Authored-By: Claude Opus 5.5 --- .gitignore | 3 + .../spi-synthetic-reported-benefits.fixed.md | 1 + policyengine_uk_data/datasets/frs.py | 47 +++-- .../datasets/imputations/frs_only.py | 102 +++++++++- .../tests/test_frs_only_imputation.py | 25 +-- .../tests/test_spi_donor_benefit_rules.py | 187 ++++++++++++++++++ pyproject.toml | 5 +- uv.lock | 73 ++++++- 8 files changed, 409 insertions(+), 34 deletions(-) create mode 100644 changelog.d/spi-synthetic-reported-benefits.fixed.md create mode 100644 policyengine_uk_data/tests/test_spi_donor_benefit_rules.py diff --git a/.gitignore b/.gitignore index 9a741bc49..5c1ab2385 100644 --- a/.gitignore +++ b/.gitignore @@ -19,3 +19,6 @@ **/_build !policyengine_uk_data/storage/*.csv **/version.json + +# Hypothesis example database (property-based tests) +.hypothesis/ diff --git a/changelog.d/spi-synthetic-reported-benefits.fixed.md b/changelog.d/spi-synthetic-reported-benefits.fixed.md new file mode 100644 index 000000000..f315e0118 --- /dev/null +++ b/changelog.d/spi-synthetic-reported-benefits.fixed.md @@ -0,0 +1 @@ +SPI-synthetic rows no longer report income-related, out-of-work or Child Benefit receipt, and their take-up flags and `receives_benefits_in_own_right` come from their own reports instead of the FRS donor's. diff --git a/policyengine_uk_data/datasets/frs.py b/policyengine_uk_data/datasets/frs.py index e440d4869..82f31fe55 100644 --- a/policyengine_uk_data/datasets/frs.py +++ b/policyengine_uk_data/datasets/frs.py @@ -74,6 +74,14 @@ "esa_contrib_reported", "esa_income_reported", ) +# Take-up flags anchored on reported receipt: flag -> (take-up rate +# parameter, person-level report column). A benefit unit with any member +# reporting receipt claims with certainty; the rest are filled at random. +REPORTED_TAKEUP_ANCHORS = { + "would_claim_child_benefit": ("child_benefit", "child_benefit_reported"), + "would_claim_pc": ("pension_credit", "pension_credit_reported"), + "would_claim_uc": ("universal_credit", "universal_credit_reported"), +} NON_ADVANCED_EDUCATION_LEVELS = ( "PRE_PRIMARY", "PRIMARY", @@ -211,6 +219,16 @@ def derive_esa_support_group_proxy( ) +def reported_benunit_mask( + person: pd.DataFrame, benunit: pd.DataFrame, person_column: str +) -> np.ndarray: + """Benefit units with any member reporting a positive ``person_column``.""" + reporter_benunits = set( + person.loc[person[person_column] > 0, "person_benunit_id"].values + ) + return benunit["benunit_id"].isin(reporter_benunits).values + + def derive_receives_benefits_in_own_right(pe_person: pd.DataFrame) -> pd.Series: """Identify people reporting adult benefits that end QYP status.""" @@ -1511,11 +1529,14 @@ def determine_education_level(fted_val, typeed2_val, age_val): assign_takeup_with_reported_anchors, ) - def _reported_benunit_mask(person_column: str) -> np.ndarray: - reporter_benunits = set( - pe_person.loc[pe_person[person_column] > 0, "person_benunit_id"].values + def _anchored_takeup(flag: str, rate: float) -> np.ndarray: + return assign_takeup_with_reported_anchors( + generator.random(len(pe_benunit)), + rate, + reported_mask=reported_benunit_mask( + pe_person, pe_benunit, REPORTED_TAKEUP_ANCHORS[flag][1] + ), ) - return pe_benunit["benunit_id"].isin(reporter_benunits).values # Person-level pe_person["would_claim_marriage_allowance"] = ( @@ -1524,23 +1545,17 @@ def _reported_benunit_mask(person_column: str) -> np.ndarray: # Benefit unit-level — anchor on any adult in the benefit unit having # reported positive receipt in the FRS benefits table. - pe_benunit["would_claim_child_benefit"] = assign_takeup_with_reported_anchors( - generator.random(len(pe_benunit)), - child_benefit_rate, - reported_mask=_reported_benunit_mask("child_benefit_reported"), + pe_benunit["would_claim_child_benefit"] = _anchored_takeup( + "would_claim_child_benefit", child_benefit_rate ) pe_benunit["child_benefit_opts_out"] = ( generator.random(len(pe_benunit)) < child_benefit_opts_out_rate ) - pe_benunit["would_claim_pc"] = assign_takeup_with_reported_anchors( - generator.random(len(pe_benunit)), - pension_credit_rate, - reported_mask=_reported_benunit_mask("pension_credit_reported"), + pe_benunit["would_claim_pc"] = _anchored_takeup( + "would_claim_pc", pension_credit_rate ) - pe_benunit["would_claim_uc"] = assign_takeup_with_reported_anchors( - generator.random(len(pe_benunit)), - universal_credit_rate, - reported_mask=_reported_benunit_mask("universal_credit_reported"), + pe_benunit["would_claim_uc"] = _anchored_takeup( + "would_claim_uc", universal_credit_rate ) pe_benunit["would_claim_tfc"] = generator.random(len(pe_benunit)) < tfc_rate diff --git a/policyengine_uk_data/datasets/imputations/frs_only.py b/policyengine_uk_data/datasets/imputations/frs_only.py index 242fbf247..036215650 100644 --- a/policyengine_uk_data/datasets/imputations/frs_only.py +++ b/policyengine_uk_data/datasets/imputations/frs_only.py @@ -40,6 +40,13 @@ add_disability_benefit_categories_from_reported_amounts, add_disability_benefit_flags_from_reported_amounts, ) +from policyengine_uk_data.datasets.frs import ( + BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS, + REPORTED_TAKEUP_ANCHORS, + reported_benunit_mask, +) +from policyengine_uk_data.parameters import load_take_up_rate +from policyengine_uk_data.utils.takeup import assign_takeup_with_reported_anchors logger = logging.getLogger(__name__) @@ -102,6 +109,91 @@ "esa_income_reported", ] +# Benefit reports set to zero on SPI-donor rows once the QRF has drawn them. +# The QRF draws each person's reports from their age, gender, region and +# incomes; it sees nothing of their benefit unit (partner, children, rent, +# capital) or of their health. A report says the person is an existing +# claimant, and policyengine-uk acts on that, so these are zeroed: +# +# - Income-related awards. Entitlement turns on the benefit unit's joint +# means and make-up, which on these rows come from the imputed incomes, so +# the model's own means test is the only coherent source. A report would +# instead be a continuing-award gate (housing benefit, income support, tax +# credits, income-related ESA and JSA), the paid amount (income-related +# ESA and JSA; council tax reduction where the model has no scheme) or a +# certain-claim anchor (UC, Pension Credit). Sure Start Maternity Grant +# needs one of these awards and is copied from the donor, not drawn. +# - Benefits paid only to people out of work or incapable of it: ESA and +# JSA (contributory), incapacity benefit and severe disablement allowance. +# On the 2024-25 build, 76% of SPI-row ESA (contributory) reporters earned +# more than ESA's permitted-work limit, against 2% of FRS reporters. +# - Child Benefit, whose only use is the take-up anchor. The draw ignores +# the children: on the 2024-25 build, 44% of SPI-row reports were in +# benefit units without a child under 16 (FRS rows: 8%). +# +# State pension (driven by age), winter fuel payment (not read by the +# model), the disability benefits, carer's allowance (the draws respect its +# earnings limit) and IIDB, AFCS and bereavement support (payable at any +# income) keep their drawn values. The zeroed columns stay in the QRF chain +# above so that the values kept do not change. +SPI_DONOR_ZEROED_PERSON_VARIABLES = [ + "universal_credit_reported", + "pension_credit_reported", + "housing_benefit_reported", + "council_tax_benefit_reported", + "income_support_reported", + "working_tax_credit_reported", + "child_tax_credit_reported", + "jsa_income_reported", + "esa_income_reported", + "ssmg_reported", + "jsa_contrib_reported", + "esa_contrib_reported", + "incapacity_benefit_reported", + "sda_reported", + "child_benefit_reported", +] + +# Seed for the take-up draws on SPI-donor rows; create_frs uses 100. +SPI_DONOR_TAKEUP_SEED = 101 + + +def apply_spi_donor_benefit_rules( + dataset: UKSingleYearDataset, +) -> UKSingleYearDataset: + """Zero SPI-donor benefit reports and re-derive the flags built from them. + + ``create_frs`` sets ``receives_benefits_in_own_right`` and the + report-anchored take-up flags from the donor's own reports, before the + SPI rows exist. Here they are rebuilt from the rows' own reports: a unit + reporting receipt claims, and the rest draw at the take-up rate. With the + anchoring reports zeroed, every SPI-donor unit draws at the rate, since + whether a synthetic family claims is unobserved. + """ + dataset = dataset.copy() + person, benunit = dataset.person, dataset.benunit + for column in SPI_DONOR_ZEROED_PERSON_VARIABLES: + if column in person.columns: + person[column] = 0.0 + + own_right = [c for c in BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS if c in person] + if "receives_benefits_in_own_right" in person.columns: + person["receives_benefits_in_own_right"] = ( + person[own_right].fillna(0).sum(axis=1) > 0 + ) + + year = int(str(dataset.time_period)[:4]) + generator = np.random.default_rng(seed=SPI_DONOR_TAKEUP_SEED) + for flag, (rate_name, column) in REPORTED_TAKEUP_ANCHORS.items(): + if flag not in benunit.columns or column not in person.columns: + continue + benunit[flag] = assign_takeup_with_reported_anchors( + generator.random(len(benunit)), + load_take_up_rate(rate_name, year), + reported_mask=reported_benunit_mask(person, benunit, column), + ) + return dataset + def _one_hot_encode(df: pd.DataFrame, columns: list[str]) -> pd.DataFrame: """Return ``df`` with object-typed ``columns`` one-hot encoded. @@ -177,7 +269,10 @@ def impute_frs_only_variables( to predict values for every row of ``target_dataset``; predictions replace the existing (donor-leaked) values in ``FRS_ONLY_PERSON_VARIABLES`` only. Variables absent from either - frame are skipped silently. + frame are skipped silently. ``apply_spi_donor_benefit_rules`` then + zeroes ``SPI_DONOR_ZEROED_PERSON_VARIABLES`` and rebuilds the flags + derived from reports, before the disability categories and flags are + derived from the remaining reports. """ from policyengine_uk_data.utils.qrf import QRF @@ -203,9 +298,9 @@ def impute_frs_only_variables( if not outputs: logger.warning( "Stage-2 FRS-only imputation: no output variables available; " - "returning target_dataset unchanged." + "applying only the SPI-donor benefit rules." ) - return target_dataset + return apply_spi_donor_benefit_rules(target_dataset) train_inputs_raw = _build_predictor_frame(train_dataset) target_inputs_raw = _build_predictor_frame(target_dataset) @@ -241,6 +336,7 @@ def impute_frs_only_variables( values = np.maximum(predictions[column].values, 0.0) target_dataset.person[column] = values + target_dataset = apply_spi_donor_benefit_rules(target_dataset) target_dataset.person = add_disability_benefit_categories_from_reported_amounts( target_dataset.person, int(str(target_dataset.time_period)[:4]), diff --git a/policyengine_uk_data/tests/test_frs_only_imputation.py b/policyengine_uk_data/tests/test_frs_only_imputation.py index 1961274e4..17ba59fd1 100644 --- a/policyengine_uk_data/tests/test_frs_only_imputation.py +++ b/policyengine_uk_data/tests/test_frs_only_imputation.py @@ -178,25 +178,26 @@ def test_frs_only_skips_missing_output_columns(): def test_frs_only_reported_values_correlate_with_training_pattern(): - """UC ``_reported`` predictions should respect the training-data pattern. + """Drawn ``_reported`` values should respect the training-data pattern. The stage-1 QRF-imputed income on the SPI-donor side gets fed back - as a stage-2 predictor. If the training data only has non-zero UC - for low-income respondents, the QRF should preferentially draw - near-zero values when predicting for high-income target rows, - compared with low-income target rows. + as a stage-2 predictor. If the training data only has non-zero + carer's allowance for low earners (it has an earnings limit), the QRF + should preferentially draw near-zero values when predicting for + high-income target rows, compared with low-income target rows. (UC + would be the obvious case, but SPI-donor rows have it zeroed.) """ from policyengine_uk_data.datasets.imputations.frs_only import ( impute_frs_only_variables, ) - # Train set with a clean employment-income → UC relationship: - # low-income respondents sometimes claim UC, high-income never do. + # Train set with a clean employment-income → carer's allowance relationship: + # low earners sometimes receive it, high earners never do. rng = np.random.default_rng(42) train = _fake_dataset(person_rows=2_000, seed=0) low_income_mask = train.person["employment_income"] < 20_000 - train.person["universal_credit_reported"] = 0.0 - train.person.loc[low_income_mask, "universal_credit_reported"] = rng.gamma( + train.person["carers_allowance_reported"] = 0.0 + train.person.loc[low_income_mask, "carers_allowance_reported"] = rng.gamma( 2, 4_000, size=int(low_income_mask.sum()) ) @@ -215,10 +216,10 @@ def test_frs_only_reported_values_correlate_with_training_pattern(): target_dataset=low_target, ) - high_mean = high_result.person["universal_credit_reported"].mean() - low_mean = low_result.person["universal_credit_reported"].mean() + high_mean = high_result.person["carers_allowance_reported"].mean() + low_mean = low_result.person["carers_allowance_reported"].mean() assert high_mean < low_mean, ( - "Stage-2 QRF should produce lower UC-receipt predictions for high-" + "Stage-2 QRF should produce lower carer's allowance predictions for high-" f"income target rows (got high={high_mean:.2f} vs low={low_mean:.2f})." ) diff --git a/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py b/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py new file mode 100644 index 000000000..c221e54ed --- /dev/null +++ b/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py @@ -0,0 +1,187 @@ +"""Benefit reports and the flags built from them on SPI-donor rows. + +Properties of ``apply_spi_donor_benefit_rules`` for any input: + +1. Every column in ``SPI_DONOR_ZEROED_PERSON_VARIABLES`` is zero afterwards. +2. Nothing else changes except ``receives_benefits_in_own_right`` and the + report-anchored take-up flags; the input is not mutated. +3. A benefit unit with a member reporting an anchoring benefit claims it. +4. ``receives_benefits_in_own_right`` is true exactly when the person + reports one of ``BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS``. +5. The rules are deterministic and idempotent. + +The fixtures are synthetic, not survey records. +""" + +import numpy as np +import pandas as pd +import pytest +from hypothesis import HealthCheck, given, settings +from hypothesis import strategies as st +from policyengine_uk.data import UKSingleYearDataset + +from policyengine_uk_data.datasets.disability_benefits import ( + add_disability_benefit_flags_from_reported_amounts, +) +from policyengine_uk_data.datasets.frs import ( + BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS, + REPORTED_TAKEUP_ANCHORS, +) +from policyengine_uk_data.datasets.imputations import frs_only +from policyengine_uk_data.datasets.imputations.frs_only import ( + FRS_ONLY_PERSON_VARIABLES, + SPI_DONOR_ZEROED_PERSON_VARIABLES, + apply_spi_donor_benefit_rules, +) +from policyengine_uk_data.parameters import load_take_up_rate + +YEAR = 2024 +REPORT_COLUMNS = sorted( + set(FRS_ONLY_PERSON_VARIABLES) | set(SPI_DONOR_ZEROED_PERSON_VARIABLES) +) +DERIVED = {"receives_benefits_in_own_right", *REPORTED_TAKEUP_ANCHORS} + + +def _dataset(benunit_sizes, reports, flags) -> UKSingleYearDataset: + n_benunits, n_people = len(benunit_sizes), sum(benunit_sizes) + benunit_of_person = np.repeat(np.arange(n_benunits), benunit_sizes) + person = pd.DataFrame( + { + "person_id": np.arange(n_people), + "person_benunit_id": benunit_of_person, + "person_household_id": benunit_of_person, + "age": np.full(n_people, 40), + "receives_benefits_in_own_right": np.asarray(flags[:n_people]), + } + ) + for i, column in enumerate(REPORT_COLUMNS): + person[column] = np.asarray(reports[i][:n_people], dtype=float) + benunit = pd.DataFrame({"benunit_id": np.arange(n_benunits)}) + for j, flag in enumerate(REPORTED_TAKEUP_ANCHORS): + benunit[flag] = np.roll(np.asarray(flags[:n_benunits]), j) + household = pd.DataFrame( + {"household_id": np.arange(n_benunits), "household_weight": 0.0} + ) + return UKSingleYearDataset( + person=person, benunit=benunit, household=household, fiscal_year=YEAR + ) + + +@st.composite +def datasets(draw): + sizes = draw(st.lists(st.integers(1, 4), min_size=1, max_size=12)) + n = sum(sizes) + amount = st.one_of(st.just(0.0), st.floats(0.01, 50_000)) + reports = [draw(st.lists(amount, min_size=n, max_size=n)) for _ in REPORT_COLUMNS] + flags = draw(st.lists(st.booleans(), min_size=n, max_size=n)) + return _dataset(sizes, reports, flags) + + +def _reporting_benunits(person, benunit, column): + reporters = person.loc[person[column] > 0, "person_benunit_id"] + return benunit.benunit_id.isin(set(reporters)).values + + +@settings(max_examples=60, deadline=None) +@given(datasets()) +def test_rules_zero_listed_reports_and_touch_nothing_else(dataset): + before = dataset.copy() + after = apply_spi_donor_benefit_rules(dataset) + + for column in SPI_DONOR_ZEROED_PERSON_VARIABLES: + assert (after.person[column] == 0).all(), column + kept = [c for c in after.person.columns if c not in DERIVED] + unchanged = [c for c in kept if c not in SPI_DONOR_ZEROED_PERSON_VARIABLES] + pd.testing.assert_frame_equal(after.person[unchanged], before.person[unchanged]) + pd.testing.assert_frame_equal(after.household, before.household) + pd.testing.assert_frame_equal(dataset.person, before.person) + pd.testing.assert_frame_equal(dataset.benunit, before.benunit) + + +@settings(max_examples=60, deadline=None) +@given(datasets()) +def test_rules_rebuild_own_right_flag_from_own_reports(dataset): + after = apply_spi_donor_benefit_rules(dataset).person + own = after[list(BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS)].sum(axis=1) > 0 + assert (after.receives_benefits_in_own_right == own).all() + + +@settings( + max_examples=60, + deadline=None, + suppress_health_check=[HealthCheck.function_scoped_fixture], +) +@given(dataset=datasets()) +def test_reporters_claim_whatever_is_zeroed(dataset, monkeypatch): + # With no anchoring report zeroed, reporters must still claim: the + # anchors are rebuilt from the rows' own reports, not left as copies. + anchoring = {column for _, column in REPORTED_TAKEUP_ANCHORS.values()} + monkeypatch.setattr( + frs_only, + "SPI_DONOR_ZEROED_PERSON_VARIABLES", + [c for c in SPI_DONOR_ZEROED_PERSON_VARIABLES if c not in anchoring], + ) + after = apply_spi_donor_benefit_rules(dataset) + for flag, (_, column) in REPORTED_TAKEUP_ANCHORS.items(): + reports = _reporting_benunits(after.person, after.benunit, column) + assert after.benunit[flag].values[reports].all(), flag + + +@settings(max_examples=40, deadline=None) +@given(datasets()) +def test_rules_are_deterministic_and_idempotent(dataset): + once = apply_spi_donor_benefit_rules(dataset) + again = apply_spi_donor_benefit_rules(dataset) + twice = apply_spi_donor_benefit_rules(once) + for frame in ("person", "benunit", "household"): + pd.testing.assert_frame_equal(getattr(once, frame), getattr(again, frame)) + pd.testing.assert_frame_equal(getattr(once, frame), getattr(twice, frame)) + + +def test_unreported_units_claim_at_the_take_up_rate(): + n = 40_000 + rng = np.random.default_rng(0) + reports = [rng.gamma(2, 1_000, n) for _ in REPORT_COLUMNS] + dataset = _dataset([1] * n, reports, np.ones(n, dtype=bool)) + after = apply_spi_donor_benefit_rules(dataset).benunit + for flag, (rate_name, _) in REPORTED_TAKEUP_ANCHORS.items(): + rate = load_take_up_rate(rate_name, YEAR) + assert after[flag].mean() == pytest.approx(rate, abs=0.01), flag + + +def test_stage_two_keeps_drawn_values_of_kept_reports(monkeypatch): + """Zeroing does not change the QRF draws of the reports that are kept.""" + from policyengine_uk_data.tests.test_frs_only_imputation import _fake_dataset + + train = _fake_dataset(person_rows=400, seed=0) + for column in ("esa_contrib_reported", "state_pension_reported", "ssmg_reported"): + train.person[column] = np.where( + np.random.default_rng(1).random(400) < 0.3, 5_000.0, 0.0 + ) + target = _fake_dataset(person_rows=80, seed=1) + target.person["ssmg_reported"] = 600.0 + target.person["receives_benefits_in_own_right"] = True + target.benunit["would_claim_uc"] = True + + ruled = frs_only.impute_frs_only_variables(train, target) + monkeypatch.setattr(frs_only, "SPI_DONOR_ZEROED_PERSON_VARIABLES", []) + unruled = frs_only.impute_frs_only_variables(train, target) + + for column in SPI_DONOR_ZEROED_PERSON_VARIABLES: + assert (ruled.person[column] == 0).all(), column + kept = [ + c + for c in FRS_ONLY_PERSON_VARIABLES + if c not in SPI_DONOR_ZEROED_PERSON_VARIABLES + ] + pd.testing.assert_frame_equal(ruled.person[kept], unruled.person[kept]) + assert not ruled.person.receives_benefits_in_own_right.any() + + # Disability flags come from the zeroed reports: ESA (contributory) was + # drawn for some people but no longer marks them disabled. + assert (unruled.person.esa_contrib_reported > 0).any() + flags = ["is_disabled_for_benefits", "is_severely_disabled_for_benefits"] + recomputed = add_disability_benefit_flags_from_reported_amounts( + ruled.person.drop(columns=flags), int(str(ruled.time_period)[:4]) + ) + pd.testing.assert_frame_equal(ruled.person[flags], recomputed[flags]) diff --git a/pyproject.toml b/pyproject.toml index beff7f1a4..7973343cb 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -45,8 +45,9 @@ dev = [ "yaml-changelog>=0.1.7", "itables", "quantile-forest", - "build", "towncrier>=24.8.0", - + "build", + "towncrier>=24.8.0", + "hypothesis>=6.168.3", ] [tool.setuptools] diff --git a/uv.lock b/uv.lock index a3e9f44d9..672737ea1 100644 --- a/uv.lock +++ b/uv.lock @@ -577,6 +577,75 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/35/f4/124858007ddf3c61e9b144107304c9152fa80b5b6c168da07d86fe583cc1/huggingface_hub-1.1.5-py3-none-any.whl", hash = "sha256:e88ecc129011f37b868586bbcfae6c56868cae80cd56a79d61575426a3aa0d7d", size = 516000, upload-time = "2025-11-20T15:49:30.926Z" }, ] +[[package]] +name = "hypothesis" +version = "6.168.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "sortedcontainers" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/09/b7/13118bbc45d6d8b9d04e2de779e2a4ff23145ea39efa692b994fb874ca72/hypothesis-6.168.3.tar.gz", hash = "sha256:a43388f9067678fef6e13bdff325b6cfa6961a590498bb37f7ff31589c83bc75", size = 511022, upload-time = "2026-09-28T05:20:58.499Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2b/cd/746a2fc1e5e5ba34f07ea6ee1ad07525a3dde5ff4836db3a7980af49a891/hypothesis-6.168.3-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:d20972ca134e652e9928ecd200966a8a2857adc2812c1d74b12a872e5cd503eb", size = 791536, upload-time = "2026-09-28T05:18:57.162Z" }, + { url = "https://files.pythonhosted.org/packages/c3/0f/7a158e377b69556c8e12c25c8fd811d0103e54c24a12dab1f3b9819f202e/hypothesis-6.168.3-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:eafbec09d3e87d13d8242411d1f5868b5e879f1bd6d95e233e9ff80c527bea1c", size = 787313, upload-time = "2026-09-28T05:20:18.122Z" }, + { url = "https://files.pythonhosted.org/packages/45/c6/7df9104ee359e8fcd77791781c47e70794964ca69673665e0de2a6590f4c/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:73c5627497968cc62d140e9fde1ea21e12cac7b64dc419fea8286cd34ff1ad4e", size = 1124036, upload-time = "2026-09-28T05:20:04.372Z" }, + { url = "https://files.pythonhosted.org/packages/92/0f/8a61715404a73b9a82cb1f976803428d603cdea976a9c30266c72ad6a7ce/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:29dc56e6dc6eeb0aeb08ac463f847279bde2336c4786faf7ee0af4b93cd031a7", size = 1147875, upload-time = "2026-09-28T05:19:15.495Z" }, + { url = "https://files.pythonhosted.org/packages/e4/3f/2b16e95cc3b9a069afa23830a012aefe377ce4a457a227e5a5026eb827ff/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:753bb501f8560d2e321ed3b62596b4496c56f15668b0a0669231ccc3e6c4e80d", size = 1149489, upload-time = "2026-09-28T05:19:18.886Z" }, + { url = "https://files.pythonhosted.org/packages/f6/e7/d7cf6dc068bb02732b2a6e38a35a0dcb7f2137608198fc79d740f69d197c/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:71ab606c472449cb872ef2a7acaec679ee4f651b7f0a477a04ebc85d8933ef8c", size = 1191992, upload-time = "2026-09-28T05:19:21.83Z" }, + { url = "https://files.pythonhosted.org/packages/5f/26/f1e5b25dec15998e8375d722825ec3dae1075cbe7cb1f74221b2312a2e33/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:26928956c54748e4dfa587333741ab750246123b93484955646ff316cca27eef", size = 1169911, upload-time = "2026-09-28T05:19:27.996Z" }, + { url = "https://files.pythonhosted.org/packages/03/36/901234a49147e6bf8a5b9e54b775706a223c2b52befea1106ea2aaf15d48/hypothesis-6.168.3-cp310-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:0bdfc53c041b61c854fa3761735736b991bc2bddafee45a361cfe4d43027c1ef", size = 1129406, upload-time = "2026-09-28T05:18:42.632Z" }, + { url = "https://files.pythonhosted.org/packages/55/c1/5fc9ff91ec9fba6bb527833386c66812524b5cc75fe7d4a4f9122ee1c420/hypothesis-6.168.3-cp310-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:650528e2b1e2a45e95c2df624d4b9364b8ac5a027ba84e1eea2bc5398dd01dcd", size = 1160399, upload-time = "2026-09-28T05:20:23.911Z" }, + { url = "https://files.pythonhosted.org/packages/9b/d2/49ad1ef5c55547c8abfd0fc571b3493806dda530fd5cb3d8263dbad86a72/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:209dc54cdb1b4d6d7020ad8d09e44c49756361b7183adacea4d6f5705085b595", size = 1299906, upload-time = "2026-09-28T05:18:58.417Z" }, + { url = "https://files.pythonhosted.org/packages/0d/5f/f98a26094c4f462c4215807bed0f9a6b5508b27a3c329bcba1bc8adc7ac7/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:af8ca98cd11dc7f9427bb90483abcd9688d4df2e8e63893a30a2423b027ebb11", size = 1425538, upload-time = "2026-09-28T05:20:49.074Z" }, + { url = "https://files.pythonhosted.org/packages/d0/b8/c5b4406e3a6591e51a42f85469351d1d824e0a54b84167aec9b2353759fe/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:f8122cfdd0bc0ba843063effa22ac310bdebdb3a1319bf1e63f14baa119c72ab", size = 1377084, upload-time = "2026-09-28T05:20:06.618Z" }, + { url = "https://files.pythonhosted.org/packages/e0/01/0d84a8ea469d024c15f602799c0b8410fbe03d99fdb0370449b33c3c916f/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:4bedbb379eab34f792af7ee9a05aae04e9c08bbb52e0f90f5a2110d4fe4b2fbd", size = 1281162, upload-time = "2026-09-28T05:19:42.768Z" }, + { url = "https://files.pythonhosted.org/packages/22/b5/ea6435038de1a795f005cb603531e8ccc053febc50180f3ab55583c9c60e/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:5a953115b9f5c95133ab2d04efffeec96e5658c3207923dca7285f7db3e6bbef", size = 1300433, upload-time = "2026-09-28T05:18:55.634Z" }, + { url = "https://files.pythonhosted.org/packages/6c/78/fa6c77d9f64b6d592e69e582d83cbce7d7e56897faa4762885a2123b0166/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:6173558e676ad25ed1e20507fa4024a0d816dd90f715f77006e5a28f19109026", size = 1336273, upload-time = "2026-09-28T05:19:08.273Z" }, + { url = "https://files.pythonhosted.org/packages/f1/56/4acd0d778bab818c9ea336fcba768bf43bc70db7f84cc373579298df4641/hypothesis-6.168.3-cp310-abi3-win32.whl", hash = "sha256:dc66390fb12d80585aa9222bf538ce8b7aa22cf5d118250647355a1c9e8f62f4", size = 678211, upload-time = "2026-09-28T05:19:23.355Z" }, + { url = "https://files.pythonhosted.org/packages/b0/db/0a02b146ad1c30f9716362f2e1a379dfd6b0f7be03ac1e22c9da24c9e2a9/hypothesis-6.168.3-cp310-abi3-win_amd64.whl", hash = "sha256:92325b276360fe86c5bf71a568c0d53a6d140b0de36dcf029f9164a17803bb24", size = 684906, upload-time = "2026-09-28T05:18:35.099Z" }, + { url = "https://files.pythonhosted.org/packages/a3/c0/958deaf726848f96f52250740bf39f13f476b068e22f73e2b358eaa07532/hypothesis-6.168.3-cp310-abi3-win_arm64.whl", hash = "sha256:3cf6f1eeaf41cd8d60cf1f88fde905ca1dd77c906929a507d6ac7f66f2ccba2a", size = 683292, upload-time = "2026-09-28T05:19:57.404Z" }, + { url = "https://files.pythonhosted.org/packages/0c/a2/6787da846d929e52fc3344d803299c45782dbeac287528af08380e984bf8/hypothesis-6.168.3-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:1b230a850de63334c16654a34a2d547e0179d36b9071d4439b3e7237f6d077e7", size = 793254, upload-time = "2026-09-28T05:19:36.294Z" }, + { url = "https://files.pythonhosted.org/packages/89/ba/2893f5ca42501f4d562cba3229c8994c3a3e5fac66ef60c1e0e0b5220a5d/hypothesis-6.168.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d63b0226cd3e0d8bdd97c3384b22a21934ed4d53246c1c93575dada616672499", size = 784699, upload-time = "2026-09-28T05:18:54.194Z" }, + { url = "https://files.pythonhosted.org/packages/b7/aa/7d7349daf75b71f6f35876f8de115e974c5c04d600e86df1b2779b0883ab/hypothesis-6.168.3-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0369f5df055f96e117ab12e5f249668ff144731ab5280bc7a205fdf81f990b89", size = 1123016, upload-time = "2026-09-28T05:20:36.372Z" }, + { url = "https://files.pythonhosted.org/packages/00/f0/7774e1ea072708ea46cb25c4aeee5f9978b6c271764809069e8a6789858e/hypothesis-6.168.3-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f076bcd0f77fdcdb7797826c879099573d03a02e228ebe649ea917081b962ac0", size = 1168989, upload-time = "2026-09-28T05:20:08.44Z" }, + { url = "https://files.pythonhosted.org/packages/a7/7d/113992abed9efbd7944e3a496a382da58152dce126fddf6419ff31a0a57d/hypothesis-6.168.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:b823ba1fcec8da730f29316d010b06d3f7e0c3828dcf91020e24c55e7d24652a", size = 1298626, upload-time = "2026-09-28T05:19:40.986Z" }, + { url = "https://files.pythonhosted.org/packages/d5/65/3659fa5e733027e5b37a27486e57f4053535e40853d23501c422bb275043/hypothesis-6.168.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:dd2849c269d674e4618590f3b48d443bd4c06b5aef3d8d869086c2e6d213d248", size = 1335184, upload-time = "2026-09-28T05:20:20.087Z" }, + { url = "https://files.pythonhosted.org/packages/b4/6c/35aab2221b5ea65125340f236e1a765a9e9f3b28d89a389ea0033fed29f7/hypothesis-6.168.3-cp313-cp313-win_amd64.whl", hash = "sha256:3ef7d26f5789e691401d5f87eafed9bd2763f0dbe47af6d6b66012a509404766", size = 682173, upload-time = "2026-09-28T05:19:12.762Z" }, + { url = "https://files.pythonhosted.org/packages/6e/79/27b0cb56ff5d2bc92458bf6fcebdb1f0dc01f57f043a178e9f9d74498169/hypothesis-6.168.3-cp314-cp314-macosx_10_12_x86_64.whl", hash = "sha256:e2d4c68729a13df9af4998d2652cfb5d541c5609b88c306880a5dfb284938ec2", size = 793287, upload-time = "2026-09-28T05:20:38.522Z" }, + { url = "https://files.pythonhosted.org/packages/e8/bd/5342f95c3bc36586777ef8cb8eba87ac7acf0184210327fb0e6d8654e295/hypothesis-6.168.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:b345f818083ec99966a43ca4f7b38feb62c6920ce28572bc2bde4948a8b7eaba", size = 784857, upload-time = "2026-09-28T05:20:42.609Z" }, + { url = "https://files.pythonhosted.org/packages/4a/5f/fc774241518a0588d362680a3084275058aab86d6bc02a2c61e1073142d0/hypothesis-6.168.3-cp314-cp314-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0608c610fc002978fc5de0f471770d8817e9455cb8649a47983c9f8bccfc1897", size = 1123330, upload-time = "2026-09-28T05:19:29.735Z" }, + { url = "https://files.pythonhosted.org/packages/b0/e6/131f16775a3dca5f4fe27f0d6ad9a9851600098a54a872d163f6ebf0b664/hypothesis-6.168.3-cp314-cp314-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9dcf6448b1ecc37f2b23f2d1b3ddfc9ff6b6910614c15a642dc82f419b96037b", size = 1169178, upload-time = "2026-09-28T05:19:53.436Z" }, + { url = "https://files.pythonhosted.org/packages/06/61/c9f5bff8b73321c9fd12f2d69666c6e4861b97b60f5f24fdd6ac293007fd/hypothesis-6.168.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:ae4f9f094041dcce5119ebd7bab71062743ab02b6e654d056b370beda78c19e2", size = 1299127, upload-time = "2026-09-28T05:20:14.486Z" }, + { url = "https://files.pythonhosted.org/packages/e4/6d/d4618f8ab12dd76c4d58ea052252759aa2bf0c4e71f4b502ba5d577e0b9f/hypothesis-6.168.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:d479985fe73af97badfb72dc6d20c6a353e486a36f6d065f600026e0cf954a86", size = 1335390, upload-time = "2026-09-28T05:20:10.326Z" }, + { url = "https://files.pythonhosted.org/packages/95/b6/727161e17cc297df1aeea945de05c532033b66751625bec192bc5b5ff707/hypothesis-6.168.3-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:785e2c45f8c08e274b4bf1ccae97f1a4e09407e9a68e80790cfab1d17a4a45fa", size = 624291, upload-time = "2026-09-28T05:18:36.376Z" }, + { url = "https://files.pythonhosted.org/packages/77/23/2f9b506392a0b8712440bcb781abc5713be9ffac1303b6d10f3669050882/hypothesis-6.168.3-cp314-cp314-win_amd64.whl", hash = "sha256:320920b1e3dae8611eee8a03d063cf2187446f2a17c38cfb8a7fc1466f71eee2", size = 682064, upload-time = "2026-09-28T05:20:16.336Z" }, + { url = "https://files.pythonhosted.org/packages/50/93/efabfd95eb2b69c0c9fa1e1c83c8290aa2715d46baeb5a7764faf09c133a/hypothesis-6.168.3-cp314-cp314t-macosx_10_12_x86_64.whl", hash = "sha256:35380baa981108a7f60c4eab71e46acd8d8f58440520346a1a6aba06dca7e074", size = 791875, upload-time = "2026-09-28T05:19:37.789Z" }, + { url = "https://files.pythonhosted.org/packages/1b/fc/2a0ada1623a9a33048bbf9f00a7c92513398c6b8865cba5efb854262ce30/hypothesis-6.168.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:f0aaaed00438fa6673d856aaee12a4a8afbde82a9f3c5ff487cacfb064239afe", size = 783412, upload-time = "2026-09-28T05:18:48.266Z" }, + { url = "https://files.pythonhosted.org/packages/41/31/73c615d37eeb12da0209555612c32ced628d860267c23cfd2691c6207aea/hypothesis-6.168.3-cp314-cp314t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:f27e6df1576bf838e7f484d4ae5cba92114497ca30afb917056bf1ea33e95675", size = 1121662, upload-time = "2026-09-28T05:19:26.225Z" }, + { url = "https://files.pythonhosted.org/packages/92/54/1893344a3b8bbf83fa7c9bb7e96f6836580213854f46671cb749f7b4706c/hypothesis-6.168.3-cp314-cp314t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3bc85014577982ec6d2e266edc5cd6e7a7d3674c648791ca983c67a52f89a0c4", size = 1167761, upload-time = "2026-09-28T05:18:49.85Z" }, + { url = "https://files.pythonhosted.org/packages/26/97/a61d82968febd25a0f08ffbbfabfaeead66f68fc5fa73f3c9ece1c36fc8f/hypothesis-6.168.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:ccfc29505aa1cdcc254cb9cd701fd0811fef5480addd97d4127a00df023410f7", size = 1297309, upload-time = "2026-09-28T05:19:51.711Z" }, + { url = "https://files.pythonhosted.org/packages/bc/01/fe8e4cf6d39d02f4efadaa351427fa13722086cfc54a2da3579741c5177a/hypothesis-6.168.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:32d0699566aaa93f9e97a44705d7164386f1e91de78d277f53bada35e78bcc96", size = 1334260, upload-time = "2026-09-28T05:19:00.121Z" }, + { url = "https://files.pythonhosted.org/packages/39/6a/09177ebc62f4778f94379ecade6a51299d3d8c4f4b75ba636bf3a0202364/hypothesis-6.168.3-cp314-cp314t-win_amd64.whl", hash = "sha256:28d88fa174ecbd4ecbd7bb290f06d0db084a3971c3f511ae2830b65b5e25500f", size = 681992, upload-time = "2026-09-28T05:19:48.043Z" }, + { url = "https://files.pythonhosted.org/packages/f9/f6/890bf33d608cd63348d3146a3ca83362fbf09e589c5e5230a00c403070f4/hypothesis-6.168.3-cp315-abi3.abi3t-macosx_10_12_x86_64.whl", hash = "sha256:61f5782d807b1e6aa5c9beef1054cb2037e7d1add48a64972cd6ee447778781f", size = 791231, upload-time = "2026-09-28T05:18:43.998Z" }, + { url = "https://files.pythonhosted.org/packages/03/dd/fc36b204f8aa7437c676576d9437f0422de46aa8a729e6cd5bbb0224e759/hypothesis-6.168.3-cp315-abi3.abi3t-macosx_11_0_arm64.whl", hash = "sha256:7aacf3cf40c7ce8f9e4924d5b57bc0b068beafdfed2347bcf160a28b4cfbba7b", size = 783171, upload-time = "2026-09-28T05:19:20.353Z" }, + { url = "https://files.pythonhosted.org/packages/5c/d2/3cd087d577db67c404991d7723e64d7cd75570cdc82a0f7329aed6f2086c/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:01768a03a30dc54df7fe457c0b34016c84d598fa00d5965ededab93ba4eb3408", size = 1121152, upload-time = "2026-09-28T05:18:39.095Z" }, + { url = "https://files.pythonhosted.org/packages/42/6f/a05a68cc66e7a22701c50bf94fb9ddeb36f4f5d0de3cc130066709eda621/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:1a8a4ffc6c6e6e577f2bfbcfebf7cffbb310283ba2532c729570a0a752cebaaa", size = 1144069, upload-time = "2026-09-28T05:20:44.705Z" }, + { url = "https://files.pythonhosted.org/packages/4a/f4/fdd7a093fb2fb3a0ce24256c92ad5bf67936f4669bd06d5ceae4298150ec/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:b987d73eba95183a7e59cca6d1925c588aa4307922d9852cc2b4d282a2ae4128", size = 1146692, upload-time = "2026-09-28T05:20:25.83Z" }, + { url = "https://files.pythonhosted.org/packages/f1/c8/6dbd4377e935505ae4fc8ee4c7b18c69994c4bbaca015447c470218bdbee/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:d4569c39bd97d9573e946429ed676f3b55a7c8ed80920addd67d384a47d59b38", size = 1189480, upload-time = "2026-09-28T05:20:46.734Z" }, + { url = "https://files.pythonhosted.org/packages/ba/bf/ff288b496b690000d2686dcfa7f67855e4c5a6dddb464ed8e4880be83bb2/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:6042b8707a4b25bbbfe20b110258fb7549a68951e5525608a8ce06091039e9fc", size = 1167109, upload-time = "2026-09-28T05:19:24.758Z" }, + { url = "https://files.pythonhosted.org/packages/d9/d1/3fee2bc29fc747fabbb27e174515592e390809ca465d3bbbabbcfa3235a8/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_31_riscv64.whl", hash = "sha256:f413629de94d38a7a2ad259697a6752526143cb49c2e2d7bdba19381c80693f6", size = 1126823, upload-time = "2026-09-28T05:18:52.636Z" }, + { url = "https://files.pythonhosted.org/packages/58/6a/585294392fa6a6d9719446a042a777ba98ecca265d9e48cbedc10c88df0f/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:f071737e4e775bebba07e319eb645a880d1e0186b4d24bad23255e849d86d483", size = 1155773, upload-time = "2026-09-28T05:20:51.268Z" }, + { url = "https://files.pythonhosted.org/packages/cd/4d/fa283ff79996debf1ff08593e01f8b773eb8211b19b3d6d5d491317a65bf/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_aarch64.whl", hash = "sha256:f59a3912858d0609c26054aa1c474847c797937f5e0e1c77f0d3f08032e50f09", size = 1296671, upload-time = "2026-09-28T05:18:46.937Z" }, + { url = "https://files.pythonhosted.org/packages/42/05/98f9c2f628da5afabb6c3e8df5d2d299d2d50bf306b14a6e26d3954c4951/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_armv7l.whl", hash = "sha256:f09c05a23a8025dd5cad07c2ed299a46778e72d49ff0dd5bf353bcc0f7dc1580", size = 1422001, upload-time = "2026-09-28T05:19:04.188Z" }, + { url = "https://files.pythonhosted.org/packages/b8/c9/f2109ade29a7ec284d55bca328d784743e8b13e321576fd34ee7e65f99af/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_i686.whl", hash = "sha256:44ace770bda3a0301739fc5a413c764de790c739df1f1a7218dd049f8594d9f5", size = 1374182, upload-time = "2026-09-28T05:20:27.764Z" }, + { url = "https://files.pythonhosted.org/packages/ce/6d/e05d5f72441564a3bebc71fa155deafd0fd3b015d6014ec8a00edf6e42bb/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_ppc64le.whl", hash = "sha256:03b131043608f94a2578a2a896a1a72092acb7079815eef5b8513b08447e0e62", size = 1278402, upload-time = "2026-09-28T05:20:56.332Z" }, + { url = "https://files.pythonhosted.org/packages/96/d7/6988a7f1f69c5c530687dce03a32c157e2d878e3a2e32af9269ce016b78a/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_riscv64.whl", hash = "sha256:e9784aca26eddfe99b03a0292320db742cc8a74200ef864fdce949f526f973cd", size = 1297770, upload-time = "2026-09-28T05:19:11.387Z" }, + { url = "https://files.pythonhosted.org/packages/a2/28/5bb82b60b836bd2329e1fe01ad94efc2b14dfa806f8e50ed14b795d78770/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_x86_64.whl", hash = "sha256:2b52ac363096232bebc2add117e9178d91f4248f4cdb919fd1026b5f86a4bb16", size = 1333977, upload-time = "2026-09-28T05:20:02.483Z" }, + { url = "https://files.pythonhosted.org/packages/8e/7d/d419841b8f65481ea1a50c4ba36f2670d48b8c7a51e4e8586089564cf7f6/hypothesis-6.168.3-cp315-abi3.abi3t-win32.whl", hash = "sha256:d28e3a6b511a74ce37df5274b51f36c2b274fea7365e00b16f0c81e22acd5957", size = 675394, upload-time = "2026-09-28T05:19:55.237Z" }, + { url = "https://files.pythonhosted.org/packages/d8/d2/1de6a2ad100e44817f2e9a8e8ba3eeaf4769621f4f87f2d4966c4971fcc4/hypothesis-6.168.3-cp315-abi3.abi3t-win_amd64.whl", hash = "sha256:7b9638789548361a57d984f56619ac694a914c328181d911271618409261ff4a", size = 681699, upload-time = "2026-09-28T05:19:09.957Z" }, + { url = "https://files.pythonhosted.org/packages/c0/79/be3370fd02734d6b1d950183ae9224580d19fa8d356b95348d60f1e3eeb7/hypothesis-6.168.3-cp315-abi3.abi3t-win_arm64.whl", hash = "sha256:65d78e4357ec48ed2c67825f06740ee3599be4cfe770a6092bed07108d679ac5", size = 679849, upload-time = "2026-09-28T05:19:06.884Z" }, +] + [[package]] name = "idna" version = "3.11" @@ -1364,7 +1433,7 @@ wheels = [ [[package]] name = "policyengine-uk-data" -version = "1.56.16" +version = "1.57.4" source = { editable = "." } dependencies = [ { name = "google-auth" }, @@ -1392,6 +1461,7 @@ dependencies = [ dev = [ { name = "build" }, { name = "furo" }, + { name = "hypothesis" }, { name = "itables" }, { name = "l0-python" }, { name = "pytest" }, @@ -1410,6 +1480,7 @@ requires-dist = [ { name = "google-auth" }, { name = "google-cloud-storage" }, { name = "huggingface-hub" }, + { name = "hypothesis", marker = "extra == 'dev'", specifier = ">=6.168.3" }, { name = "itables", marker = "extra == 'dev'" }, { name = "l0-python", marker = "extra == 'dev'", specifier = ">=0.4.0" }, { name = "microcalibrate", specifier = ">=0.18.0" }, From eb14fd07907cbf19dbc4cb968d0248a0e35cfde6 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 2 Oct 2026 04:13:03 -0400 Subject: [PATCH 2/4] Restore donor status benefits on SPI rows, keep the donor's Child Benefit flag Review of 96672af (subfleet 20261002-011737-spi-514-review): - IIDB, AFCS and bereavement support follow an injury, service or a death, not income, and the QRF drew them at 6.2, 2.4 and 4.6 times the FRS rate by weight. SPI rows now take the donor's own values. - The Child Benefit take-up flag keeps the donor's value: the award does not depend on the replaced incomes and the claim is for the same children. Only the UC and Pension Credit flags are redrawn. - The comment no longer says the model's means test replaces the zeroed income-related reports: in policyengine-uk 2.93.0 housing benefit, CTR, IS, tax credits and income-related ESA and JSA need a report, so SPI rows no longer get them. Evidence is now cited by weight, and Child Benefit counts 16-19 qualifying young people. - One assign_reported_takeup helper serves create_frs and the SPI rows, so the rate and the anchoring rule have one source. The SPI draws no longer depend on which columns are present. - Tests pin the three rule sets, use shuffled, gapped ids and check the rebuilt flags against reports that are not zeroed. Co-Authored-By: Claude Opus 5.5 --- .../spi-synthetic-reported-benefits.fixed.md | 2 +- policyengine_uk_data/datasets/frs.py | 51 +++--- .../datasets/imputations/frs_only.py | 156 +++++++++------- .../tests/test_spi_donor_benefit_rules.py | 172 +++++++++++++----- 4 files changed, 246 insertions(+), 135 deletions(-) diff --git a/changelog.d/spi-synthetic-reported-benefits.fixed.md b/changelog.d/spi-synthetic-reported-benefits.fixed.md index f315e0118..1310079b8 100644 --- a/changelog.d/spi-synthetic-reported-benefits.fixed.md +++ b/changelog.d/spi-synthetic-reported-benefits.fixed.md @@ -1 +1 @@ -SPI-synthetic rows no longer report income-related, out-of-work or Child Benefit receipt, and their take-up flags and `receives_benefits_in_own_right` come from their own reports instead of the FRS donor's. +SPI-synthetic rows no longer report income-related, out-of-work or Child Benefit receipt, take their industrial injuries, armed forces compensation and bereavement support from the FRS donor, and get UC, Pension Credit and `receives_benefits_in_own_right` flags from their own reports instead of the donor's. diff --git a/policyengine_uk_data/datasets/frs.py b/policyengine_uk_data/datasets/frs.py index 82f31fe55..1e2e9b8bf 100644 --- a/policyengine_uk_data/datasets/frs.py +++ b/policyengine_uk_data/datasets/frs.py @@ -33,6 +33,7 @@ STORAGE_FOLDER, ) from policyengine_uk_data.parameters import load_take_up_rate, load_parameter +from policyengine_uk_data.utils.takeup import assign_takeup_with_reported_anchors from policyengine_uk_data.datasets.childcare.assumptions import ( EXTENDED_HOURS_MEAN, EXTENDED_HOURS_SD, @@ -229,6 +230,24 @@ def reported_benunit_mask( return benunit["benunit_id"].isin(reporter_benunits).values +def assign_reported_takeup( + person: pd.DataFrame, + benunit: pd.DataFrame, + flag: str, + year: int, + draws: np.ndarray, +) -> np.ndarray: + """Take-up for one ``REPORTED_TAKEUP_ANCHORS`` flag: benefit units + reporting receipt claim, and the rest are filled at random so the share + claiming matches the take-up rate (see ``utils/takeup.py``).""" + rate_name, report_column = REPORTED_TAKEUP_ANCHORS[flag] + return assign_takeup_with_reported_anchors( + draws, + load_take_up_rate(rate_name, year), + reported_mask=reported_benunit_mask(person, benunit, report_column), + ) + + def derive_receives_benefits_in_own_right(pe_person: pd.DataFrame) -> pd.Series: """Identify people reporting adult benefits that end QYP status.""" @@ -1505,9 +1524,6 @@ def determine_education_level(fted_val, typeed2_val, age_val): generator = np.random.default_rng(seed=100) # Load take-up rates from parameter files - child_benefit_rate = load_take_up_rate("child_benefit", year) - pension_credit_rate = load_take_up_rate("pension_credit", year) - universal_credit_rate = load_take_up_rate("universal_credit", year) marriage_allowance_rate = load_take_up_rate("marriage_allowance", year) child_benefit_opts_out_rate = load_take_up_rate("child_benefit_opts_out_rate", year) tfc_rate = load_take_up_rate("tax_free_childcare", year) @@ -1525,19 +1541,6 @@ def determine_education_level(fted_val, typeed2_val, age_val): # who report positive receipt of a benefit are assigned takeup=True with # certainty; the remaining non-reporters are filled probabilistically to # hit the aggregate target rate. See policyengine_uk_data/utils/takeup.py. - from policyengine_uk_data.utils.takeup import ( - assign_takeup_with_reported_anchors, - ) - - def _anchored_takeup(flag: str, rate: float) -> np.ndarray: - return assign_takeup_with_reported_anchors( - generator.random(len(pe_benunit)), - rate, - reported_mask=reported_benunit_mask( - pe_person, pe_benunit, REPORTED_TAKEUP_ANCHORS[flag][1] - ), - ) - # Person-level pe_person["would_claim_marriage_allowance"] = ( generator.random(len(pe_person)) < marriage_allowance_rate @@ -1545,17 +1548,21 @@ def _anchored_takeup(flag: str, rate: float) -> np.ndarray: # Benefit unit-level — anchor on any adult in the benefit unit having # reported positive receipt in the FRS benefits table. - pe_benunit["would_claim_child_benefit"] = _anchored_takeup( - "would_claim_child_benefit", child_benefit_rate + pe_benunit["would_claim_child_benefit"] = assign_reported_takeup( + pe_person, + pe_benunit, + "would_claim_child_benefit", + year, + generator.random(len(pe_benunit)), ) pe_benunit["child_benefit_opts_out"] = ( generator.random(len(pe_benunit)) < child_benefit_opts_out_rate ) - pe_benunit["would_claim_pc"] = _anchored_takeup( - "would_claim_pc", pension_credit_rate + pe_benunit["would_claim_pc"] = assign_reported_takeup( + pe_person, pe_benunit, "would_claim_pc", year, generator.random(len(pe_benunit)) ) - pe_benunit["would_claim_uc"] = _anchored_takeup( - "would_claim_uc", universal_credit_rate + pe_benunit["would_claim_uc"] = assign_reported_takeup( + pe_person, pe_benunit, "would_claim_uc", year, generator.random(len(pe_benunit)) ) pe_benunit["would_claim_tfc"] = generator.random(len(pe_benunit)) < tfc_rate diff --git a/policyengine_uk_data/datasets/imputations/frs_only.py b/policyengine_uk_data/datasets/imputations/frs_only.py index 036215650..26a39532f 100644 --- a/policyengine_uk_data/datasets/imputations/frs_only.py +++ b/policyengine_uk_data/datasets/imputations/frs_only.py @@ -43,10 +43,9 @@ from policyengine_uk_data.datasets.frs import ( BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS, REPORTED_TAKEUP_ANCHORS, - reported_benunit_mask, + assign_reported_takeup, + derive_receives_benefits_in_own_right, ) -from policyengine_uk_data.parameters import load_take_up_rate -from policyengine_uk_data.utils.takeup import assign_takeup_with_reported_anchors logger = logging.getLogger(__name__) @@ -109,33 +108,39 @@ "esa_income_reported", ] -# Benefit reports set to zero on SPI-donor rows once the QRF has drawn them. -# The QRF draws each person's reports from their age, gender, region and -# incomes; it sees nothing of their benefit unit (partner, children, rent, -# capital) or of their health. A report says the person is an existing -# claimant, and policyengine-uk acts on that, so these are zeroed: +# The QRF draws each person's benefit reports from their age, gender, region +# and incomes. It sees nothing of their benefit unit (partner, children, +# rent, capital), their health or their history, and policyengine-uk reads a +# positive report as an existing claim. On SPI-donor rows, after the draw: # -# - Income-related awards. Entitlement turns on the benefit unit's joint -# means and make-up, which on these rows come from the imputed incomes, so -# the model's own means test is the only coherent source. A report would -# instead be a continuing-award gate (housing benefit, income support, tax -# credits, income-related ESA and JSA), the paid amount (income-related -# ESA and JSA; council tax reduction where the model has no scheme) or a -# certain-claim anchor (UC, Pension Credit). Sure Start Maternity Grant -# needs one of these awards and is copied from the donor, not drawn. -# - Benefits paid only to people out of work or incapable of it: ESA and -# JSA (contributory), incapacity benefit and severe disablement allowance. -# On the 2024-25 build, 76% of SPI-row ESA (contributory) reporters earned -# more than ESA's permitted-work limit, against 2% of FRS reporters. -# - Child Benefit, whose only use is the take-up anchor. The draw ignores -# the children: on the 2024-25 build, 44% of SPI-row reports were in -# benefit units without a child under 16 (FRS rows: 8%). +# Zeroed (SPI_DONOR_ZEROED_PERSON_VARIABLES): +# - Income-related awards. Entitlement turns on the unit's joint means and +# make-up, which here come from the imputed incomes. UC and Pension Credit +# keep a route: their take-up flags are redrawn below. In +# policyengine-uk 2.93.0 the others (housing benefit, council tax +# reduction, income support, tax credits, income-related ESA and JSA) can +# only be claimed with a report, so these rows no longer receive them. +# Sure Start Maternity Grant needs one of these awards. +# - Benefits paid only to people out of work or incapable of it: ESA and JSA +# (contributory), incapacity benefit and severe disablement allowance. On +# the 2024-25 build, 41% of SPI-row ESA (contributory) reporters by weight +# earned more than ESA's permitted-work limit, against 0.4% on FRS rows. +# - Child Benefit, which the model reads only through the take-up flag. The +# draw ignores the children: 34% of SPI-row reports by weight were in +# benefit units with no child or qualifying young person (FRS rows: 0%). # -# State pension (driven by age), winter fuel payment (not read by the -# model), the disability benefits, carer's allowance (the draws respect its -# earnings limit) and IIDB, AFCS and bereavement support (payable at any -# income) keep their drawn values. The zeroed columns stay in the QRF chain -# above so that the values kept do not change. +# Restored to the donor's own value (SPI_DONOR_RESTORED_PERSON_VARIABLES): +# industrial injuries, armed forces compensation and bereavement support. +# These follow from an injury, service or a death, not income, and the QRF +# drew them at 6.2, 2.4 and 4.6 times the FRS rate by weight. +# +# Kept as drawn: state pension (paid as reported once over pension age), +# winter fuel payment (not read by the model), the disability benefits and +# carer's allowance, whose drawn rates sit below the FRS rates as the income +# gradient implies. +# +# Every column stays in the QRF chain above, so the values kept do not +# change. They were drawn alongside the values later zeroed or restored. SPI_DONOR_ZEROED_PERSON_VARIABLES = [ "universal_credit_reported", "pension_credit_reported", @@ -153,45 +158,59 @@ "sda_reported", "child_benefit_reported", ] +SPI_DONOR_RESTORED_PERSON_VARIABLES = [ + "iidb_reported", + "afcs_reported", + "bsp_reported", +] -# Seed for the take-up draws on SPI-donor rows; create_frs uses 100. +# Take-up flags redrawn on SPI-donor rows. Whether a synthetic family claims +# a means-tested benefit at its imputed income is unobserved, so these units +# draw at the take-up rate. The Child Benefit flag keeps the donor's value: +# the award doesn't depend on the replaced incomes, and the donor's claim is +# for the same children. +SPI_DONOR_REDRAWN_TAKEUP_FLAGS = ("would_claim_uc", "would_claim_pc") +# Seed for those draws; create_frs uses 100. SPI_DONOR_TAKEUP_SEED = 101 def apply_spi_donor_benefit_rules( dataset: UKSingleYearDataset, + donor_person: pd.DataFrame | None = None, ) -> UKSingleYearDataset: - """Zero SPI-donor benefit reports and re-derive the flags built from them. - - ``create_frs`` sets ``receives_benefits_in_own_right`` and the - report-anchored take-up flags from the donor's own reports, before the - SPI rows exist. Here they are rebuilt from the rows' own reports: a unit - reporting receipt claims, and the rest draw at the take-up rate. With the - anchoring reports zeroed, every SPI-donor unit draws at the rate, since - whether a synthetic family claims is unobserved. + """Apply the SPI-donor benefit rules above to ``dataset``. + + ``donor_person`` is the person table before the QRF draw, holding the + donor's own reports; without it the restored columns are left alone. + ``receives_benefits_in_own_right`` and the redrawn take-up flags, which + ``create_frs`` built from the donor's reports, are rebuilt from the rows' + own reports. """ dataset = dataset.copy() person, benunit = dataset.person, dataset.benunit for column in SPI_DONOR_ZEROED_PERSON_VARIABLES: if column in person.columns: person[column] = 0.0 + if donor_person is not None: + for column in SPI_DONOR_RESTORED_PERSON_VARIABLES: + if column in person.columns and column in donor_person.columns: + person[column] = donor_person[column].values - own_right = [c for c in BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS if c in person] if "receives_benefits_in_own_right" in person.columns: + own_right = person.reindex( + columns=list(BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS), fill_value=0.0 + ) person["receives_benefits_in_own_right"] = ( - person[own_right].fillna(0).sum(axis=1) > 0 + derive_receives_benefits_in_own_right(own_right).values ) year = int(str(dataset.time_period)[:4]) generator = np.random.default_rng(seed=SPI_DONOR_TAKEUP_SEED) - for flag, (rate_name, column) in REPORTED_TAKEUP_ANCHORS.items(): - if flag not in benunit.columns or column not in person.columns: - continue - benunit[flag] = assign_takeup_with_reported_anchors( - generator.random(len(benunit)), - load_take_up_rate(rate_name, year), - reported_mask=reported_benunit_mask(person, benunit, column), - ) + for flag in SPI_DONOR_REDRAWN_TAKEUP_FLAGS: + draws = generator.random(len(benunit)) + report_column = REPORTED_TAKEUP_ANCHORS[flag][1] + if flag in benunit.columns and report_column in person.columns: + benunit[flag] = assign_reported_takeup(person, benunit, flag, year, draws) return dataset @@ -270,13 +289,12 @@ def impute_frs_only_variables( replace the existing (donor-leaked) values in ``FRS_ONLY_PERSON_VARIABLES`` only. Variables absent from either frame are skipped silently. ``apply_spi_donor_benefit_rules`` then - zeroes ``SPI_DONOR_ZEROED_PERSON_VARIABLES`` and rebuilds the flags - derived from reports, before the disability categories and flags are - derived from the remaining reports. + zeroes or restores some reports and rebuilds the flags derived from + them, before the disability categories and flags are derived from the + final reports. """ - from policyengine_uk_data.utils.qrf import QRF - target_dataset = target_dataset.copy() + donor_person = target_dataset.person.copy() train_person = train_dataset.person target_person = target_dataset.person @@ -295,13 +313,32 @@ def impute_frs_only_variables( len(missing), sorted(missing), ) - if not outputs: + if outputs: + target_dataset = _impute_outputs(train_dataset, target_dataset, outputs) + else: logger.warning( "Stage-2 FRS-only imputation: no output variables available; " "applying only the SPI-donor benefit rules." ) - return apply_spi_donor_benefit_rules(target_dataset) + target_dataset = apply_spi_donor_benefit_rules(target_dataset, donor_person) + target_dataset.person = add_disability_benefit_categories_from_reported_amounts( + target_dataset.person, + int(str(target_dataset.time_period)[:4]), + ) + target_dataset.person = add_disability_benefit_flags_from_reported_amounts( + target_dataset.person, + int(str(target_dataset.time_period)[:4]), + ) + + return target_dataset + + +def _impute_outputs(train_dataset, target_dataset, outputs): + """Fit the stage-2 QRF and write its draws of ``outputs`` to the target.""" + from policyengine_uk_data.utils.qrf import QRF + + train_person = train_dataset.person train_inputs_raw = _build_predictor_frame(train_dataset) target_inputs_raw = _build_predictor_frame(target_dataset) @@ -335,15 +372,4 @@ def impute_frs_only_variables( # amounts or contributions and are non-negative by construction. values = np.maximum(predictions[column].values, 0.0) target_dataset.person[column] = values - - target_dataset = apply_spi_donor_benefit_rules(target_dataset) - target_dataset.person = add_disability_benefit_categories_from_reported_amounts( - target_dataset.person, - int(str(target_dataset.time_period)[:4]), - ) - target_dataset.person = add_disability_benefit_flags_from_reported_amounts( - target_dataset.person, - int(str(target_dataset.time_period)[:4]), - ) - return target_dataset diff --git a/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py b/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py index c221e54ed..36b87a7bc 100644 --- a/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py +++ b/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py @@ -2,13 +2,16 @@ Properties of ``apply_spi_donor_benefit_rules`` for any input: -1. Every column in ``SPI_DONOR_ZEROED_PERSON_VARIABLES`` is zero afterwards. +1. Every column in ``SPI_DONOR_ZEROED_PERSON_VARIABLES`` is zero afterwards, + and every column in ``SPI_DONOR_RESTORED_PERSON_VARIABLES`` equals the + donor's value. 2. Nothing else changes except ``receives_benefits_in_own_right`` and the - report-anchored take-up flags; the input is not mutated. -3. A benefit unit with a member reporting an anchoring benefit claims it. -4. ``receives_benefits_in_own_right`` is true exactly when the person - reports one of ``BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS``. -5. The rules are deterministic and idempotent. + redrawn take-up flags, and the input is not mutated. +3. A benefit unit with a member reporting UC or Pension Credit claims it. +4. ``receives_benefits_in_own_right`` holds exactly when the person reports + one of ``BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS``. +5. The rules are deterministic and idempotent, and no flag's draws depend on + which other columns are present. The fixtures are synthetic, not survey records. """ @@ -30,6 +33,8 @@ from policyengine_uk_data.datasets.imputations import frs_only from policyengine_uk_data.datasets.imputations.frs_only import ( FRS_ONLY_PERSON_VARIABLES, + SPI_DONOR_REDRAWN_TAKEUP_FLAGS, + SPI_DONOR_RESTORED_PERSON_VARIABLES, SPI_DONOR_ZEROED_PERSON_VARIABLES, apply_spi_donor_benefit_rules, ) @@ -39,15 +44,54 @@ REPORT_COLUMNS = sorted( set(FRS_ONLY_PERSON_VARIABLES) | set(SPI_DONOR_ZEROED_PERSON_VARIABLES) ) -DERIVED = {"receives_benefits_in_own_right", *REPORTED_TAKEUP_ANCHORS} +CHANGED = {"receives_benefits_in_own_right", *SPI_DONOR_REDRAWN_TAKEUP_FLAGS} +RULED = set(SPI_DONOR_ZEROED_PERSON_VARIABLES) | set( + SPI_DONOR_RESTORED_PERSON_VARIABLES +) +ANCHORING_REPORTS = { + REPORTED_TAKEUP_ANCHORS[flag][1] for flag in SPI_DONOR_REDRAWN_TAKEUP_FLAGS +} + + +def test_rule_sets_are_pinned(): + """Each column's treatment is a decision; changing one must be deliberate.""" + assert set(SPI_DONOR_ZEROED_PERSON_VARIABLES) == { + "universal_credit_reported", + "pension_credit_reported", + "housing_benefit_reported", + "council_tax_benefit_reported", + "income_support_reported", + "working_tax_credit_reported", + "child_tax_credit_reported", + "jsa_income_reported", + "esa_income_reported", + "ssmg_reported", + "jsa_contrib_reported", + "esa_contrib_reported", + "incapacity_benefit_reported", + "sda_reported", + "child_benefit_reported", + } + assert set(SPI_DONOR_RESTORED_PERSON_VARIABLES) == { + "iidb_reported", + "afcs_reported", + "bsp_reported", + } + assert set(SPI_DONOR_REDRAWN_TAKEUP_FLAGS) == {"would_claim_uc", "would_claim_pc"} + assert not set(SPI_DONOR_ZEROED_PERSON_VARIABLES) & set( + SPI_DONOR_RESTORED_PERSON_VARIABLES + ) -def _dataset(benunit_sizes, reports, flags) -> UKSingleYearDataset: +def _dataset(benunit_sizes, reports, flags, id_seed=0) -> UKSingleYearDataset: + """Benefit units of the given sizes, with shuffled, gapped ids.""" + rng = np.random.default_rng(id_seed) n_benunits, n_people = len(benunit_sizes), sum(benunit_sizes) - benunit_of_person = np.repeat(np.arange(n_benunits), benunit_sizes) + benunit_ids = rng.permutation(n_benunits) * 7 + 3 + benunit_of_person = np.repeat(benunit_ids, benunit_sizes) person = pd.DataFrame( { - "person_id": np.arange(n_people), + "person_id": rng.permutation(n_people) * 5 + 11, "person_benunit_id": benunit_of_person, "person_household_id": benunit_of_person, "age": np.full(n_people, 40), @@ -56,12 +100,11 @@ def _dataset(benunit_sizes, reports, flags) -> UKSingleYearDataset: ) for i, column in enumerate(REPORT_COLUMNS): person[column] = np.asarray(reports[i][:n_people], dtype=float) - benunit = pd.DataFrame({"benunit_id": np.arange(n_benunits)}) + person = person.sample(frac=1, random_state=id_seed).reset_index(drop=True) + benunit = pd.DataFrame({"benunit_id": benunit_ids}) for j, flag in enumerate(REPORTED_TAKEUP_ANCHORS): benunit[flag] = np.roll(np.asarray(flags[:n_benunits]), j) - household = pd.DataFrame( - {"household_id": np.arange(n_benunits), "household_weight": 0.0} - ) + household = pd.DataFrame({"household_id": benunit_ids, "household_weight": 0.0}) return UKSingleYearDataset( person=person, benunit=benunit, household=household, fiscal_year=YEAR ) @@ -74,7 +117,16 @@ def datasets(draw): amount = st.one_of(st.just(0.0), st.floats(0.01, 50_000)) reports = [draw(st.lists(amount, min_size=n, max_size=n)) for _ in REPORT_COLUMNS] flags = draw(st.lists(st.booleans(), min_size=n, max_size=n)) - return _dataset(sizes, reports, flags) + return _dataset(sizes, reports, flags, id_seed=draw(st.integers(0, 10**6))) + + +def _donor(dataset, seed=1): + """A donor person table: the same people with different report values.""" + donor = dataset.person.copy() + rng = np.random.default_rng(seed) + for column in SPI_DONOR_RESTORED_PERSON_VARIABLES: + donor[column] = np.where(rng.random(len(donor)) < 0.5, 0.0, 1_234.0) + return donor def _reporting_benunits(person, benunit, column): @@ -84,26 +136,30 @@ def _reporting_benunits(person, benunit, column): @settings(max_examples=60, deadline=None) @given(datasets()) -def test_rules_zero_listed_reports_and_touch_nothing_else(dataset): +def test_rules_zero_restore_and_touch_nothing_else(dataset): before = dataset.copy() - after = apply_spi_donor_benefit_rules(dataset) + donor = _donor(dataset) + after = apply_spi_donor_benefit_rules(dataset, donor) for column in SPI_DONOR_ZEROED_PERSON_VARIABLES: assert (after.person[column] == 0).all(), column - kept = [c for c in after.person.columns if c not in DERIVED] - unchanged = [c for c in kept if c not in SPI_DONOR_ZEROED_PERSON_VARIABLES] + for column in SPI_DONOR_RESTORED_PERSON_VARIABLES: + np.testing.assert_array_equal(after.person[column], donor[column]) + unchanged = [c for c in after.person.columns if c not in CHANGED | RULED] pd.testing.assert_frame_equal(after.person[unchanged], before.person[unchanged]) + kept_flags = [c for c in after.benunit.columns if c not in CHANGED] + pd.testing.assert_frame_equal(after.benunit[kept_flags], before.benunit[kept_flags]) pd.testing.assert_frame_equal(after.household, before.household) pd.testing.assert_frame_equal(dataset.person, before.person) pd.testing.assert_frame_equal(dataset.benunit, before.benunit) -@settings(max_examples=60, deadline=None) +@settings(max_examples=30, deadline=None) @given(datasets()) -def test_rules_rebuild_own_right_flag_from_own_reports(dataset): - after = apply_spi_donor_benefit_rules(dataset).person - own = after[list(BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS)].sum(axis=1) > 0 - assert (after.receives_benefits_in_own_right == own).all() +def test_restored_columns_untouched_without_donor(dataset): + after = apply_spi_donor_benefit_rules(dataset) + restored = list(SPI_DONOR_RESTORED_PERSON_VARIABLES) + pd.testing.assert_frame_equal(after.person[restored], dataset.person[restored]) @settings( @@ -112,17 +168,23 @@ def test_rules_rebuild_own_right_flag_from_own_reports(dataset): suppress_health_check=[HealthCheck.function_scoped_fixture], ) @given(dataset=datasets()) -def test_reporters_claim_whatever_is_zeroed(dataset, monkeypatch): - # With no anchoring report zeroed, reporters must still claim: the - # anchors are rebuilt from the rows' own reports, not left as copies. - anchoring = {column for _, column in REPORTED_TAKEUP_ANCHORS.values()} +def test_flags_follow_own_reports_whatever_is_zeroed(dataset, monkeypatch): + # With the own-right and anchoring reports left in place, the rebuilt + # flags must follow them, not the donor's flags. monkeypatch.setattr( frs_only, "SPI_DONOR_ZEROED_PERSON_VARIABLES", - [c for c in SPI_DONOR_ZEROED_PERSON_VARIABLES if c not in anchoring], + [ + c + for c in SPI_DONOR_ZEROED_PERSON_VARIABLES + if c not in ANCHORING_REPORTS | set(BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS) + ], ) after = apply_spi_donor_benefit_rules(dataset) - for flag, (_, column) in REPORTED_TAKEUP_ANCHORS.items(): + own = after.person[list(BENEFITS_IN_OWN_RIGHT_REPORTED_COLUMNS)].sum(axis=1) > 0 + assert (after.person.receives_benefits_in_own_right == own).all() + for flag in SPI_DONOR_REDRAWN_TAKEUP_FLAGS: + column = REPORTED_TAKEUP_ANCHORS[flag][1] reports = _reporting_benunits(after.person, after.benunit, column) assert after.benunit[flag].values[reports].all(), flag @@ -130,54 +192,70 @@ def test_reporters_claim_whatever_is_zeroed(dataset, monkeypatch): @settings(max_examples=40, deadline=None) @given(datasets()) def test_rules_are_deterministic_and_idempotent(dataset): - once = apply_spi_donor_benefit_rules(dataset) - again = apply_spi_donor_benefit_rules(dataset) - twice = apply_spi_donor_benefit_rules(once) + donor = _donor(dataset) + once = apply_spi_donor_benefit_rules(dataset, donor) + again = apply_spi_donor_benefit_rules(dataset, donor) + twice = apply_spi_donor_benefit_rules(once, donor) for frame in ("person", "benunit", "household"): pd.testing.assert_frame_equal(getattr(once, frame), getattr(again, frame)) pd.testing.assert_frame_equal(getattr(once, frame), getattr(twice, frame)) +@settings(max_examples=30, deadline=None) +@given(datasets()) +def test_draws_do_not_depend_on_other_columns(dataset): + full = apply_spi_donor_benefit_rules(dataset) + dataset.benunit = dataset.benunit.drop(columns=["would_claim_uc"]) + partial = apply_spi_donor_benefit_rules(dataset) + pd.testing.assert_series_equal( + full.benunit.would_claim_pc, partial.benunit.would_claim_pc + ) + + def test_unreported_units_claim_at_the_take_up_rate(): n = 40_000 rng = np.random.default_rng(0) reports = [rng.gamma(2, 1_000, n) for _ in REPORT_COLUMNS] dataset = _dataset([1] * n, reports, np.ones(n, dtype=bool)) after = apply_spi_donor_benefit_rules(dataset).benunit - for flag, (rate_name, _) in REPORTED_TAKEUP_ANCHORS.items(): - rate = load_take_up_rate(rate_name, YEAR) + for flag in SPI_DONOR_REDRAWN_TAKEUP_FLAGS: + rate = load_take_up_rate(REPORTED_TAKEUP_ANCHORS[flag][0], YEAR) assert after[flag].mean() == pytest.approx(rate, abs=0.01), flag + assert after.would_claim_child_benefit.all() -def test_stage_two_keeps_drawn_values_of_kept_reports(monkeypatch): - """Zeroing does not change the QRF draws of the reports that are kept.""" +def test_stage_two_applies_the_rules_and_keeps_drawn_values(monkeypatch): + """Through ``impute_frs_only_variables``: zeroed, restored, kept and flags.""" from policyengine_uk_data.tests.test_frs_only_imputation import _fake_dataset train = _fake_dataset(person_rows=400, seed=0) - for column in ("esa_contrib_reported", "state_pension_reported", "ssmg_reported"): - train.person[column] = np.where( - np.random.default_rng(1).random(400) < 0.3, 5_000.0, 0.0 - ) + rng = np.random.default_rng(1) + for column in ("esa_contrib_reported", "state_pension_reported", "iidb_reported"): + train.person[column] = np.where(rng.random(400) < 0.3, 5_000.0, 0.0) target = _fake_dataset(person_rows=80, seed=1) target.person["ssmg_reported"] = 600.0 + target.person["iidb_reported"] = np.where(rng.random(80) < 0.5, 0.0, 777.0) target.person["receives_benefits_in_own_right"] = True target.benunit["would_claim_uc"] = True + target.benunit["would_claim_child_benefit"] = False ruled = frs_only.impute_frs_only_variables(train, target) monkeypatch.setattr(frs_only, "SPI_DONOR_ZEROED_PERSON_VARIABLES", []) + monkeypatch.setattr(frs_only, "SPI_DONOR_RESTORED_PERSON_VARIABLES", []) unruled = frs_only.impute_frs_only_variables(train, target) for column in SPI_DONOR_ZEROED_PERSON_VARIABLES: assert (ruled.person[column] == 0).all(), column - kept = [ - c - for c in FRS_ONLY_PERSON_VARIABLES - if c not in SPI_DONOR_ZEROED_PERSON_VARIABLES - ] + np.testing.assert_array_equal( + ruled.person.iidb_reported, target.person.iidb_reported + ) + assert not np.array_equal(unruled.person.iidb_reported, target.person.iidb_reported) + kept = [c for c in FRS_ONLY_PERSON_VARIABLES if c not in RULED] pd.testing.assert_frame_equal(ruled.person[kept], unruled.person[kept]) assert not ruled.person.receives_benefits_in_own_right.any() + assert not ruled.benunit.would_claim_child_benefit.any() - # Disability flags come from the zeroed reports: ESA (contributory) was + # Disability flags come from the final reports: ESA (contributory) was # drawn for some people but no longer marks them disabled. assert (unruled.person.esa_contrib_reported > 0).any() flags = ["is_disabled_for_benefits", "is_severely_disabled_for_benefits"] From 90377e06575cd8f057ac3d449a8df25c875957a9 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 2 Oct 2026 05:14:11 -0400 Subject: [PATCH 3/4] Drop a small-cell comparator from the SPI rules comment; test the rate year and the UC redraw Round-2 review of eb14fd0 (subfleet 20261002-041419-spi-514-review-r2, APPROVE): the FRS comparison for ESA (contributory) rested on under 10 records; the take-up rates' year and the stage-two UC redraw had no direct test. Co-Authored-By: Claude Opus 5.5 --- .../datasets/imputations/frs_only.py | 2 +- .../tests/test_spi_donor_benefit_rules.py | 22 +++++++++++++++++++ 2 files changed, 23 insertions(+), 1 deletion(-) diff --git a/policyengine_uk_data/datasets/imputations/frs_only.py b/policyengine_uk_data/datasets/imputations/frs_only.py index 26a39532f..06b9b78b0 100644 --- a/policyengine_uk_data/datasets/imputations/frs_only.py +++ b/policyengine_uk_data/datasets/imputations/frs_only.py @@ -124,7 +124,7 @@ # - Benefits paid only to people out of work or incapable of it: ESA and JSA # (contributory), incapacity benefit and severe disablement allowance. On # the 2024-25 build, 41% of SPI-row ESA (contributory) reporters by weight -# earned more than ESA's permitted-work limit, against 0.4% on FRS rows. +# earned more than ESA's permitted-work limit. # - Child Benefit, which the model reads only through the take-up flag. The # draw ignores the children: 34% of SPI-row reports by weight were in # benefit units with no child or qualifying young person (FRS rows: 0%). diff --git a/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py b/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py index 36b87a7bc..799439038 100644 --- a/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py +++ b/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py @@ -224,6 +224,25 @@ def test_unreported_units_claim_at_the_take_up_rate(): assert after.would_claim_child_benefit.all() +def test_take_up_rates_are_read_for_the_dataset_year(monkeypatch): + years = [] + + def rate(name, year): + years.append(year) + return 0.5 + + monkeypatch.setattr("policyengine_uk_data.datasets.frs.load_take_up_rate", rate) + dataset = _dataset([1, 2], [[1.0] * 3 for _ in REPORT_COLUMNS], [True] * 3) + dataset = UKSingleYearDataset( + person=dataset.person, + benunit=dataset.benunit, + household=dataset.household, + fiscal_year=2031, + ) + apply_spi_donor_benefit_rules(dataset) + assert years == [2031] * len(SPI_DONOR_REDRAWN_TAKEUP_FLAGS) + + def test_stage_two_applies_the_rules_and_keeps_drawn_values(monkeypatch): """Through ``impute_frs_only_variables``: zeroed, restored, kept and flags.""" from policyengine_uk_data.tests.test_frs_only_imputation import _fake_dataset @@ -254,6 +273,9 @@ def test_stage_two_applies_the_rules_and_keeps_drawn_values(monkeypatch): pd.testing.assert_frame_equal(ruled.person[kept], unruled.person[kept]) assert not ruled.person.receives_benefits_in_own_right.any() assert not ruled.benunit.would_claim_child_benefit.any() + # Every unit entered claiming UC as the donor; with the reports zeroed, + # the redraw at the 55% rate leaves some 80 units out. + assert not ruled.benunit.would_claim_uc.all() # Disability flags come from the final reports: ESA (contributory) was # drawn for some people but no longer marks them disabled. From deb16e944e0ed00df43f6492344ebbc6b055657c Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Mon, 5 Oct 2026 14:23:51 -0400 Subject: [PATCH 4/4] Keep the QRF's council tax reduction draw on SPI rows (ruling d821) Max approved #514 into the batched release with one change: CTR is no longer zeroed on SPI-synthetic rows. In policyengine-uk 2.93.0 a CTR claim needs a report, so zeroing it left those rows no claim route (-21% CTR in the earlier build). SPI CTR stays the stage-two QRF draw, as on main, until it is zeroed together with a household-level CTR imputation (#499). Moves council_tax_benefit_reported out of SPI_DONOR_ZEROED_PERSON_VARIABLES, pins the set of reports kept as drawn, and tests that SPI CTR equals the stage-two draw (not zeroed, not the donor's value). Co-Authored-By: Claude Opus 5.5 --- .../spi-synthetic-reported-benefits.fixed.md | 2 +- .../datasets/imputations/frs_only.py | 14 ++++--- .../tests/test_spi_donor_benefit_rules.py | 40 ++++++++++++++++++- 3 files changed, 48 insertions(+), 8 deletions(-) diff --git a/changelog.d/spi-synthetic-reported-benefits.fixed.md b/changelog.d/spi-synthetic-reported-benefits.fixed.md index 1310079b8..a10ad0154 100644 --- a/changelog.d/spi-synthetic-reported-benefits.fixed.md +++ b/changelog.d/spi-synthetic-reported-benefits.fixed.md @@ -1 +1 @@ -SPI-synthetic rows no longer report income-related, out-of-work or Child Benefit receipt, take their industrial injuries, armed forces compensation and bereavement support from the FRS donor, and get UC, Pension Credit and `receives_benefits_in_own_right` flags from their own reports instead of the donor's. +SPI-synthetic rows no longer report income-related (except council tax reduction, which keeps its imputed value for now), out-of-work or Child Benefit receipt, take their industrial injuries, armed forces compensation and bereavement support from the FRS donor, and get UC, Pension Credit and `receives_benefits_in_own_right` flags from their own reports instead of the donor's. diff --git a/policyengine_uk_data/datasets/imputations/frs_only.py b/policyengine_uk_data/datasets/imputations/frs_only.py index 06b9b78b0..7c1d5cbea 100644 --- a/policyengine_uk_data/datasets/imputations/frs_only.py +++ b/policyengine_uk_data/datasets/imputations/frs_only.py @@ -117,10 +117,10 @@ # - Income-related awards. Entitlement turns on the unit's joint means and # make-up, which here come from the imputed incomes. UC and Pension Credit # keep a route: their take-up flags are redrawn below. In -# policyengine-uk 2.93.0 the others (housing benefit, council tax -# reduction, income support, tax credits, income-related ESA and JSA) can -# only be claimed with a report, so these rows no longer receive them. -# Sure Start Maternity Grant needs one of these awards. +# policyengine-uk 2.93.0 the others (housing benefit, income support, tax +# credits, income-related ESA and JSA) can only be claimed with a report, +# so these rows no longer receive them. Sure Start Maternity Grant needs +# one of these awards. Council tax reduction is kept (below). # - Benefits paid only to people out of work or incapable of it: ESA and JSA # (contributory), incapacity benefit and severe disablement allowance. On # the 2024-25 build, 41% of SPI-row ESA (contributory) reporters by weight @@ -137,7 +137,10 @@ # Kept as drawn: state pension (paid as reported once over pension age), # winter fuel payment (not read by the model), the disability benefits and # carer's allowance, whose drawn rates sit below the FRS rates as the income -# gradient implies. +# gradient implies. Council tax reduction is also kept as drawn for now. In +# 2.93.0 it too can only be claimed with a report, and zeroing it cut 2025 +# CTR by 21% on the 2024-25 build. It will be zeroed together with a +# household-level CTR imputation (#499). # # Every column stays in the QRF chain above, so the values kept do not # change. They were drawn alongside the values later zeroed or restored. @@ -145,7 +148,6 @@ "universal_credit_reported", "pension_credit_reported", "housing_benefit_reported", - "council_tax_benefit_reported", "income_support_reported", "working_tax_credit_reported", "child_tax_credit_reported", diff --git a/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py b/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py index 799439038..1a3633fc2 100644 --- a/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py +++ b/policyengine_uk_data/tests/test_spi_donor_benefit_rules.py @@ -59,7 +59,6 @@ def test_rule_sets_are_pinned(): "universal_credit_reported", "pension_credit_reported", "housing_benefit_reported", - "council_tax_benefit_reported", "income_support_reported", "working_tax_credit_reported", "child_tax_credit_reported", @@ -77,6 +76,20 @@ def test_rule_sets_are_pinned(): "afcs_reported", "bsp_reported", } + # Kept as drawn: every report the QRF draws that no rule above touches. + drawn_reports = {c for c in FRS_ONLY_PERSON_VARIABLES if c.endswith("_reported")} + assert drawn_reports - RULED == { + "state_pension_reported", + "winter_fuel_allowance_reported", + "attendance_allowance_reported", + "dla_sc_reported", + "dla_m_reported", + "pip_m_reported", + "pip_dl_reported", + "carers_allowance_reported", + "maternity_allowance_reported", + "council_tax_benefit_reported", + } assert set(SPI_DONOR_REDRAWN_TAKEUP_FLAGS) == {"would_claim_uc", "would_claim_pc"} assert not set(SPI_DONOR_ZEROED_PERSON_VARIABLES) & set( SPI_DONOR_RESTORED_PERSON_VARIABLES @@ -285,3 +298,28 @@ def test_stage_two_applies_the_rules_and_keeps_drawn_values(monkeypatch): ruled.person.drop(columns=flags), int(str(ruled.time_period)[:4]) ) pd.testing.assert_frame_equal(ruled.person[flags], recomputed[flags]) + + +def test_council_tax_reduction_keeps_the_stage_two_draw(): + """SPI rows carry the QRF's CTR draw through unchanged, as before the + rules: not zeroed, not the donor's. Zeroing waits for #499.""" + from policyengine_uk_data.tests.test_frs_only_imputation import _fake_dataset + + train = _fake_dataset(person_rows=400, seed=0) + rng = np.random.default_rng(2) + train.person["council_tax_benefit_reported"] = np.where( + rng.random(400) < 0.3, 1_200.0, 0.0 + ) + target = _fake_dataset(person_rows=80, seed=1) + target.person["council_tax_benefit_reported"] = 999.0 + outputs = [c for c in FRS_ONLY_PERSON_VARIABLES if c in target.person.columns] + + # The draw alone, which is what stage two returned before the rules. + drawn = frs_only._impute_outputs(train, target.copy(), outputs).person + ruled = frs_only.impute_frs_only_variables(train, target).person + + np.testing.assert_array_equal( + ruled.council_tax_benefit_reported, drawn.council_tax_benefit_reported + ) + assert (drawn.council_tax_benefit_reported > 0).any() + assert (drawn.council_tax_benefit_reported != 999.0).any()