From 2d8f56556602d3d7ba5861f8ead3dbfeed40bd9d Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Wed, 30 Sep 2026 17:12:04 -0400 Subject: [PATCH 1/2] Keep FRS-reported dividends instead of overwriting them with SPI draws The SPI income model predicts from age, gender and region only; applying it to the FRS half gave Universal Credit claimants six-figure dividends unrelated to anything they reported (policyengine-uk#1948). Only the SPI-donor copy is now imputed, as in the microcosm UK build. Co-Authored-By: Claude Opus 5.5 --- changelog.d/frs-reported-dividends.fixed.md | 1 + .../datasets/imputations/income.py | 15 +++-- .../tests/test_imputation_source_flags.py | 56 +++++++++++++++++++ 3 files changed, 66 insertions(+), 6 deletions(-) create mode 100644 changelog.d/frs-reported-dividends.fixed.md diff --git a/changelog.d/frs-reported-dividends.fixed.md b/changelog.d/frs-reported-dividends.fixed.md new file mode 100644 index 00000000..389b32f3 --- /dev/null +++ b/changelog.d/frs-reported-dividends.fixed.md @@ -0,0 +1 @@ +- The enhanced FRS keeps FRS respondents' reported dividends instead of replacing them with draws from the SPI income model, which predicts from age, gender and region alone and gave Universal Credit claimants six-figure dividends (policyengine-uk#1948). The SPI-donor half still carries the SPI income distribution. diff --git a/policyengine_uk_data/datasets/imputations/income.py b/policyengine_uk_data/datasets/imputations/income.py index feafcb27..9af0d7c4 100644 --- a/policyengine_uk_data/datasets/imputations/income.py +++ b/policyengine_uk_data/datasets/imputations/income.py @@ -266,7 +266,7 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset: model = create_income_model() - # Impute just dividends on the original, full variable set on the copy + # Impute the full income set on the SPI-donor copy only. zero_weight_copy = impute_over_incomes( zero_weight_copy, @@ -292,11 +292,14 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset: target_dataset=zero_weight_copy, ) - dataset = impute_over_incomes( - dataset, - model, - ["dividend_income"], - ) + # The FRS half keeps its reported dividends. Replacing them with a draw + # from the SPI model, whose only predictors are age, gender and region, + # gave dividends to FRS respondents without regard to their investments, + # earnings or benefits: in 2026 the FRS half carried £47bn of dividends + # against £17bn reported, and Universal Credit claimants received as + # many as anyone else (policyengine-uk#1948). The SPI-donor half carries + # the SPI dividend distribution, and calibration to the HMRC dividend + # targets reweights between the two, as in the microcosm UK build. zero_weight_copy.validate() dataset.validate() diff --git a/policyengine_uk_data/tests/test_imputation_source_flags.py b/policyengine_uk_data/tests/test_imputation_source_flags.py index 2df1f9d8..91f96db9 100644 --- a/policyengine_uk_data/tests/test_imputation_source_flags.py +++ b/policyengine_uk_data/tests/test_imputation_source_flags.py @@ -131,3 +131,59 @@ def test_impute_capital_gains_marks_capital_gains_clone_households(monkeypatch): True, ] assert result.household.loc[2:, "household_weight"].eq(1).all() + + +def test_impute_income_keeps_reported_dividends_on_the_frs_half(monkeypatch): + """SPI draws replace incomes only on the SPI-donor copy (policyengine-uk#1948). + + The SPI income model predicts from age, gender and region alone, so using + it to overwrite the FRS half's dividends gave Universal Credit claimants + dividends unrelated to anything they reported. + """ + from policyengine_uk_data.datasets.imputations import income as income_module + from policyengine_uk_data.datasets import disability_benefits + from policyengine_uk_data.datasets.imputations import frs_only + + imputed_halves = [] + + def fake_impute_over_incomes(dataset, _model, output_variables): + imputed_halves.append( + ( + bool(dataset.household["household_is_spi_synthetic"].all()), + tuple(output_variables), + ) + ) + dataset = dataset.copy() + for column in output_variables: + dataset.person[column] = 123_456.0 + return dataset + + monkeypatch.setattr(income_module, "create_income_model", lambda: object()) + monkeypatch.setattr( + income_module, + "subsample_dataset", + lambda dataset, _sample_size: dataset.copy(), + ) + monkeypatch.setattr(income_module, "impute_over_incomes", fake_impute_over_incomes) + monkeypatch.setattr( + frs_only, + "impute_frs_only_variables", + lambda train_dataset, target_dataset: target_dataset, + ) + monkeypatch.setattr( + disability_benefits, + "strip_internal_disability_reported_amounts", + lambda dataset: dataset, + ) + monkeypatch.setattr(income_module, "stack_datasets", _stack_without_remapping) + + dataset = _fake_dataset() + dataset.person["dividend_income"] = [150.0, 0.0] + result = income_module.impute_income(dataset) + + # Only the SPI-donor copy is imputed, with the full income set. + assert imputed_halves == [(True, tuple(income_module.IMPUTATIONS))] + frs_half = result.person.iloc[:2] + assert frs_half["dividend_income"].tolist() == [150.0, 0.0] + assert frs_half["employment_income"].tolist() == [20_000.0, 80_000.0] + assert result.person.iloc[2:]["dividend_income"].eq(123_456.0).all() From c9b203e34a79bfcee8908caf82100a52a70a1140 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Wed, 30 Sep 2026 17:24:42 -0400 Subject: [PATCH 2/2] Key FRS dividends on person_id, not row position MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit frs.py summed investment-account dividends to person.index while account.person_id holds household_id * 1e3 + person, so only 8 people got any (about £40m against £8bn). Extract the calculation into a tested helper and annualise with WEEKS_IN_YEAR like the savings lines beside it. Correct the income.py comments that relied on the near-zero FRS dividends. Co-Authored-By: Claude Opus 5.5 --- changelog.d/frs-reported-dividends.fixed.md | 3 +- policyengine_uk_data/datasets/frs.py | 37 +++++++++++------ .../datasets/imputations/income.py | 12 +++--- .../tests/test_frs_dividend_income.py | 40 +++++++++++++++++++ 4 files changed, 72 insertions(+), 20 deletions(-) create mode 100644 policyengine_uk_data/tests/test_frs_dividend_income.py diff --git a/changelog.d/frs-reported-dividends.fixed.md b/changelog.d/frs-reported-dividends.fixed.md index 389b32f3..cf31851c 100644 --- a/changelog.d/frs-reported-dividends.fixed.md +++ b/changelog.d/frs-reported-dividends.fixed.md @@ -1 +1,2 @@ -- The enhanced FRS keeps FRS respondents' reported dividends instead of replacing them with draws from the SPI income model, which predicts from age, gender and region alone and gave Universal Credit claimants six-figure dividends (policyengine-uk#1948). The SPI-donor half still carries the SPI income distribution. +- FRS investment-account dividends are now summed to the person who holds the account. They were keyed on each person's row position, so almost none matched and the base FRS carried about £40m of dividends instead of about £8bn in 2024-25. +- The enhanced FRS keeps those reported dividends instead of replacing them with draws from the SPI income model, which predicts from age, gender and region alone and gave Universal Credit claimants six-figure dividends (policyengine-uk#1948). The SPI-donor half still carries the SPI income distribution. diff --git a/policyengine_uk_data/datasets/frs.py b/policyengine_uk_data/datasets/frs.py index e440d486..3625f0f1 100644 --- a/policyengine_uk_data/datasets/frs.py +++ b/policyengine_uk_data/datasets/frs.py @@ -570,6 +570,29 @@ def validate_frs_survey_year(raw_frs_folder, year: int) -> None: ) +def frs_dividend_income(account: pd.DataFrame, person_ids) -> np.ndarray: + """Annual dividends each person reports on FRS investment accounts. + + Gilt-edged stock taxed at source (account type 6), unit and investment + trusts (7) and stocks and shares (8). Amounts taxed at source are grossed + up at the basic rate. Summed to ``person_ids``, which must be the same + ``household_id * 1e3 + person`` keys as ``account.person_id``: keying on + the positional index instead matched almost no accounts, leaving the FRS + with about £40m of dividends rather than about £8bn in 2024-25. + """ + INVERTED_BASIC_RATE = 1.25 + dividends = ( + account.accint * np.where(account.invtax == 1, INVERTED_BASIC_RATE, 1) + ) * ( + ((account.account == 6) & (account.invtax == 1)) # GGES + | account.account.isin((7, 8)) # Stocks/shares/UITs + ) + return np.maximum( + 0, + sum_to_entity(dividends, account.person_id, person_ids) * WEEKS_IN_YEAR, + ) + + def create_frs( raw_frs_folder: str, year: int, @@ -1070,19 +1093,7 @@ def determine_education_level(fted_val, typeed2_val, age_val): 0, taxable_savings_interest + pe_person["tax_free_savings_income"].values, ) - pe_person["dividend_income"] = np.maximum( - 0, - sum_to_entity( - (account.accint * np.where(account.invtax == 1, INVERTED_BASIC_RATE, 1)) - * ( - ((account.account == 6) & (account.invtax == 1)) # GGES - | account.account.isin((7, 8)) # Stocks/shares/UITs - ), - account.person_id, - person.index, - ) - * 52, - ) + pe_person["dividend_income"] = frs_dividend_income(account, person.person_id) is_head = person.hrpid == 1 household_property_income = ( household.tentyp2.isin((5, 6)) * household.subrent diff --git a/policyengine_uk_data/datasets/imputations/income.py b/policyengine_uk_data/datasets/imputations/income.py index 9af0d7c4..aef5e55d 100644 --- a/policyengine_uk_data/datasets/imputations/income.py +++ b/policyengine_uk_data/datasets/imputations/income.py @@ -223,8 +223,9 @@ def impute_over_incomes( # Housing costs (rent, mortgage interest, mortgage capital) used to be # rescaled here by new_income_total / original_income_total across - # INCOME_COMPONENTS. Because FRS dividend_income is near-zero and the - # SPI-trained QRF predicts materially larger dividends, the ratio + # INCOME_COMPONENTS. Because FRS dividend_income was then near-zero (a + # keying error in frs.py, since fixed) and the SPI-trained QRF predicts + # materially larger dividends, the ratio # inflated rent/mortgage by ~2.5× uniformly in the built enhanced FRS # — pushing AHC poverty rates 10–18 pp above HBAI for non-pensioners # (see issue #367). Housing costs now pass through unchanged; their @@ -295,11 +296,10 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset: # The FRS half keeps its reported dividends. Replacing them with a draw # from the SPI model, whose only predictors are age, gender and region, # gave dividends to FRS respondents without regard to their investments, - # earnings or benefits: in 2026 the FRS half carried £47bn of dividends - # against £17bn reported, and Universal Credit claimants received as - # many as anyone else (policyengine-uk#1948). The SPI-donor half carries + # earnings or benefits, so Universal Credit claimants received them as + # often as anyone else (policyengine-uk#1948). The SPI-donor half carries # the SPI dividend distribution, and calibration to the HMRC dividend - # targets reweights between the two, as in the microcosm UK build. + # targets reweights between the two. zero_weight_copy.validate() dataset.validate() diff --git a/policyengine_uk_data/tests/test_frs_dividend_income.py b/policyengine_uk_data/tests/test_frs_dividend_income.py new file mode 100644 index 00000000..77f37f6c --- /dev/null +++ b/policyengine_uk_data/tests/test_frs_dividend_income.py @@ -0,0 +1,40 @@ +"""FRS dividends land on the person who holds the account (policyengine-uk#1948).""" + +import numpy as np +import pandas as pd +import pytest + +from policyengine_uk_data.datasets.frs import WEEKS_IN_YEAR, frs_dividend_income + + +def test_dividends_are_keyed_on_person_id_not_row_position(): + # Person ids are household_id * 1000 + person number, so they never + # coincide with the positional index the rows happen to sit at. + person_ids = pd.Series([1_001, 1_002, 2_001, 2_002]) + account = pd.DataFrame( + { + "person_id": [1_002, 2_001, 2_001, 2_002, 1_001, 2_002], + # 8: stocks and shares; 7: unit/investment trusts; 6: gilts; 1: bank account + "account": [8, 7, 6, 6, 1, 8], + "accint": [10.0, 4.0, 2.0, 3.0, 50.0, 0.0], + "invtax": [2, 1, 1, 2, 2, 2], + } + ) + dividends = frs_dividend_income(account, person_ids) + weekly = np.array( + [ + 0.0, # 1,001 has only a bank account + 10.0, # 1,002: shares, not taxed at source + 4.0 * 1.25 + 2.0 * 1.25, # 2,001: trusts and gilts taxed at source + 0.0, # 2,002: gilts not taxed at source are excluded; shares pay 0 + ] + ) + assert dividends == pytest.approx(weekly * WEEKS_IN_YEAR) + + +def test_dividends_are_never_negative(): + person_ids = pd.Series([1_001]) + account = pd.DataFrame( + {"person_id": [1_001], "account": [8], "accint": [-5.0], "invtax": [2]} + ) + assert frs_dividend_income(account, person_ids).tolist() == [0.0]