diff --git a/changelog.d/frs-cvpay-not-property-income.fixed.md b/changelog.d/frs-cvpay-not-property-income.fixed.md new file mode 100644 index 00000000..55546420 --- /dev/null +++ b/changelog.d/frs-cvpay-not-property-income.fixed.md @@ -0,0 +1 @@ +Stop counting the rent a boarder or lodger pays (FRS CVPAY) as that boarder's or lodger's own property income. diff --git a/changelog.d/frs-property-losses-subrent.fixed.md b/changelog.d/frs-property-losses-subrent.fixed.md new file mode 100644 index 00000000..097b15f2 --- /dev/null +++ b/changelog.d/frs-property-losses-subrent.fixed.md @@ -0,0 +1 @@ +Stop counting FRS property losses (ROYYR1 with RENTPROF = 2) as property income, and count rent from sub-letting part of the home (SUBRENT) for every tenure, not only owner-occupiers. diff --git a/policyengine_uk_data/datasets/frs.py b/policyengine_uk_data/datasets/frs.py index e440d486..c7f8e179 100644 --- a/policyengine_uk_data/datasets/frs.py +++ b/policyengine_uk_data/datasets/frs.py @@ -84,6 +84,8 @@ # FRS government-training question variants use 10 or 13 for "None of these". FRS_APPROVED_TRAINING_CODES = tuple(range(1, 10)) UNKNOWN_QUALIFYING_EDUCATION_OR_TRAINING_ENTRY_AGE = 1000 +# FRS RENTPROF: whether ROYYR1 is a profit (1) or a loss (2). +FRS_RENTPROF_LOSS = 2 @lru_cache(maxsize=None) @@ -220,6 +222,47 @@ def derive_receives_benefits_in_own_right(pe_person: pd.DataFrame) -> pd.Series: ) +def frs_property_income(person: pd.DataFrame, household: pd.DataFrame) -> np.ndarray: + """Annual property income each person reports in the FRS. + + Two FRS amounts, both weekly in the released data: + + - SUBRENT, rent the household received for letting part of its home to + someone outside the household. The FRS asks every household (SubLet), + whatever its tenure, so renting and rent-free households count too. It + goes to the household reference person. ``household`` must be indexed + by ``household_id``. + - ROYYR1, the person's rent from other property, before tax and after + allowable expenses. The questionnaire cannot take a negative amount, + so a loss is entered as a positive amount with RENTPROF = 2 (question + RentProf, "Is that a profit or a loss from the property?"). A loss + counts as zero: it is not income, policyengine-uk has no property loss + input, and it is not set against the household's SUBRENT. + + SUBRENT is used as reported. SUBALLOW records whether it is before (1) + or after (2) allowable expenses, but the FRS collects no expense amount + to take off the before-expenses answers. + + Negative values are FRS missing-value codes (-1 to -9), not amounts, so + each amount is floored at zero before the two are added. + + CVPAY is not included. It is the rent that a boarder or lodger pays the + householder, and it sits on the boarder's or lodger's own adult record. + The FRS question (CvPay) asks how much rent [name] paid for board and + lodging, after deducting any state benefits to help with rent. + """ + is_head = person.hrpid == 1 + persons_household_subrent = ( + household.subrent.clip(lower=0).reindex(person.household_id).fillna(0).values + ) + rent_from_other_property = person.royyr1.clip(lower=0).where( + person.rentprof != FRS_RENTPROF_LOSS, 0 + ) + return ( + (is_head * persons_household_subrent + rent_from_other_property) * WEEKS_IN_YEAR + ).values + + def derive_is_in_non_advanced_education( current_education, is_apprentice=None, @@ -1083,25 +1126,7 @@ def determine_education_level(fted_val, typeed2_val, age_val): ) * 52, ) - is_head = person.hrpid == 1 - household_property_income = ( - household.tentyp2.isin((5, 6)) * household.subrent - ) # Owned and subletting - persons_household_property_income = ( - pd.Series( - household_property_income[person.household_id].values, - index=person.person_id, - ) - .fillna(0) - .values - ) - pe_person["property_income"] = ( - np.maximum( - 0, - is_head * persons_household_property_income + person.cvpay + person.royyr1, - ) - * WEEKS_IN_YEAR - ) + pe_person["property_income"] = frs_property_income(person, household) maintenance_to_self = np.maximum( pd.Series(np.where(person.mntus1 == 2, person.mntusam1, person.mntamt1)).fillna( 0 diff --git a/policyengine_uk_data/tests/test_frs_property_income.py b/policyengine_uk_data/tests/test_frs_property_income.py new file mode 100644 index 00000000..5fb2982c --- /dev/null +++ b/policyengine_uk_data/tests/test_frs_property_income.py @@ -0,0 +1,301 @@ +import numpy as np +import pandas as pd +import pytest + +from policyengine_uk_data.datasets.frs import ( + FRS_RENTPROF_LOSS, + WEEKS_IN_YEAR, + frs_property_income, +) +from policyengine_uk_data.datasets.frs_release import CURRENT_FRS_RELEASE +from policyengine_uk_data.storage import STORAGE_FOLDER + +HRP, NOT_HRP = 1, 2 +NOT_ASKED, PROFIT, LOSS = 0, 1, FRS_RENTPROF_LOSS +# FRS missing-value codes run from -1 to -9. +MISSING, REFUSED = -1, -9 +OWNED_WITH_MORTGAGE, OWNED_OUTRIGHT = 5, 6 +COUNCIL_RENTED, PRIVATE_RENTED_FURNISHED = 1, 4 +TENURES = range(1, 9) +BEFORE_EXPENSES, AFTER_EXPENSES = 1, 2 + + +def make_tables(people, households): + person = pd.DataFrame( + people, + columns=["household_id", "person_id", "hrpid", "royyr1", "rentprof", "cvpay"], + ) + household = pd.DataFrame( + households, columns=["household_id", "tentyp2", "subrent", "suballow"] + ).set_index("household_id") + return person, household + + +def test_rent_paid_by_a_lodger_is_not_their_property_income(): + # Owner-occupier household with a lodger who pays £100 a week (CVPAY). + person, household = make_tables( + [(1, 1_001, HRP, 0, NOT_ASKED, 0), (1, 1_002, NOT_HRP, 0, NOT_ASKED, 100)], + [(1, OWNED_OUTRIGHT, 0, 0)], + ) + assert frs_property_income(person, household).tolist() == [0, 0] + + +def test_rent_from_other_property_counts_for_any_adult(): + person, household = make_tables( + [(1, 1_001, HRP, 0, NOT_ASKED, 0), (1, 1_002, NOT_HRP, 50, PROFIT, 0)], + [(1, COUNCIL_RENTED, 0, 0)], + ) + np.testing.assert_allclose( + frs_property_income(person, household), [0, 50 * WEEKS_IN_YEAR] + ) + + +def test_a_loss_on_other_property_is_not_income(): + # ROYYR1 holds the size of the loss as a positive amount; RENTPROF flags it. + person, household = make_tables( + [(1, 1_001, HRP, 90, LOSS, 0), (1, 1_002, NOT_HRP, 50, PROFIT, 0)], + [(1, OWNED_WITH_MORTGAGE, 0, 0)], + ) + np.testing.assert_allclose( + frs_property_income(person, household), [0, 50 * WEEKS_IN_YEAR] + ) + + +def test_a_loss_on_other_property_does_not_reduce_subletting_rent(): + person, household = make_tables( + [(1, 1_001, HRP, 30, LOSS, 0)], + [(1, OWNED_OUTRIGHT, 80, AFTER_EXPENSES)], + ) + np.testing.assert_allclose( + frs_property_income(person, household), [80 * WEEKS_IN_YEAR] + ) + + +def test_rent_with_no_profit_or_loss_answer_still_counts(): + # Only an explicit loss removes the amount. + person, household = make_tables( + [(1, 1_001, HRP, 50, NOT_ASKED, 0)], [(1, OWNED_OUTRIGHT, 0, 0)] + ) + np.testing.assert_allclose( + frs_property_income(person, household), [50 * WEEKS_IN_YEAR] + ) + + +@pytest.mark.parametrize("tenure", TENURES) +def test_subletting_rent_goes_to_the_household_reference_person(tenure): + # Every tenure is asked SubLet, so renting households count too. + person, household = make_tables( + [(1, 1_001, NOT_HRP, 0, NOT_ASKED, 0), (1, 1_002, HRP, 0, NOT_ASKED, 0)], + [(1, tenure, 80, BEFORE_EXPENSES)], + ) + np.testing.assert_allclose( + frs_property_income(person, household), [0, 80 * WEEKS_IN_YEAR] + ) + + +@pytest.mark.parametrize("suballow", [0, BEFORE_EXPENSES, AFTER_EXPENSES]) +def test_subletting_rent_is_used_as_reported_whatever_its_expenses_basis(suballow): + person, household = make_tables( + [(1, 1_001, HRP, 0, NOT_ASKED, 0)], + [(1, PRIVATE_RENTED_FURNISHED, 80, suballow)], + ) + np.testing.assert_allclose( + frs_property_income(person, household), [80 * WEEKS_IN_YEAR] + ) + + +def test_missing_value_codes_count_as_zero(): + # A missing code in one amount neither becomes income nor cancels the other. + person, household = make_tables( + [ + (1, 1_001, HRP, MISSING, MISSING, 0), + (1, 1_002, NOT_HRP, REFUSED, MISSING, 0), + (2, 2_001, HRP, 50, PROFIT, 0), + ], + [(1, COUNCIL_RENTED, 10, AFTER_EXPENSES), (2, OWNED_OUTRIGHT, MISSING, 0)], + ) + np.testing.assert_allclose( + frs_property_income(person, household), + [10 * WEEKS_IN_YEAR, 0, 50 * WEEKS_IN_YEAR], + ) + + +def test_adult_and_child_rows_sharing_index_labels(): + # create_frs stacks the adult and child tables, so index labels repeat. + adults, household = make_tables( + [(1, 1_001, HRP, 20, PROFIT, 0), (2, 2_001, HRP, 40, LOSS, 60)], + [ + (1, OWNED_OUTRIGHT, 80, BEFORE_EXPENSES), + (2, PRIVATE_RENTED_FURNISHED, 30, AFTER_EXPENSES), + ], + ) + children, _ = make_tables([(1, 1_002, 0, 0, NOT_ASKED, 0)], []) + person = pd.concat([adults, children]).sort_index(kind="stable") + assert person.index.tolist() == [0, 0, 1] + np.testing.assert_allclose( + frs_property_income(person, household), + [100 * WEEKS_IN_YEAR, 0, 30 * WEEKS_IN_YEAR], + ) + + +def random_tables(seed: int): + """Random households of one to four adults; the first is the HRP.""" + rng = np.random.default_rng(seed) + people, households = [], [] + for household_id in range(1, rng.integers(1, 30) + 1): + tenure = int(rng.integers(1, 9)) + subrent = float(rng.choice([0, MISSING, rng.uniform(0, 500)])) + suballow = int(rng.integers(1, 3)) if subrent > 0 else 0 + households.append((household_id, tenure, subrent, suballow)) + for person in range(1, rng.integers(1, 5) + 1): + royyr1 = float(rng.choice([0, MISSING, rng.uniform(0, 2_000)])) + if royyr1 > 0: + rentprof = int(rng.choice([PROFIT, PROFIT, LOSS])) + else: + rentprof = MISSING if royyr1 < 0 else NOT_ASKED + people.append( + ( + household_id, + household_id * 1_000 + person, + HRP if person == 1 else NOT_HRP, + royyr1, + rentprof, + float(rng.choice([0, rng.uniform(0, 400)])), + ) + ) + return make_tables(people, households) + + +def property_income_one_person_at_a_time(person, household): + """The same rules written as a loop, to check the vectorised helper.""" + result = [] + for row in person.itertuples(): + weekly = 0.0 + if row.rentprof != LOSS: + weekly += max(0.0, row.royyr1) + if row.hrpid == HRP and row.household_id in household.index: + weekly += max(0.0, household.subrent[row.household_id]) + result.append(weekly * WEEKS_IN_YEAR) + return np.array(result) + + +SEEDS = range(200) + + +@pytest.mark.parametrize("seed", SEEDS) +def test_property_income_matches_the_loop_version(seed): + person, household = random_tables(seed) + np.testing.assert_allclose( + frs_property_income(person, household), + property_income_one_person_at_a_time(person, household), + ) + + +@pytest.mark.parametrize("seed", SEEDS) +def test_property_income_does_not_depend_on_cvpay_tenure_or_expenses_basis(seed): + person, household = random_tables(seed) + rng = np.random.default_rng(seed) + other_person = person.assign(cvpay=rng.uniform(0, 400, len(person))) + other_household = household.assign( + tentyp2=rng.integers(1, 9, len(household)), + suballow=rng.integers(1, 3, len(household)), + ) + np.testing.assert_array_equal( + frs_property_income(person, household), + frs_property_income(other_person, other_household), + ) + + +@pytest.mark.parametrize("seed", SEEDS) +def test_property_income_conserves_reported_rent(seed): + # Each household's SUBRENT is counted once (on its HRP) and every ROYYR1 + # that is not a loss is counted on its own record, so the totals match. + person, household = random_tables(seed) + result = frs_property_income(person, household) + assert (result >= 0).all() + profits = person.royyr1.clip(lower=0)[person.rentprof != LOSS] + subrent = household.subrent.clip(lower=0) + expected = (subrent.sum() + profits.sum()) * WEEKS_IN_YEAR + assert result.sum() == pytest.approx(expected) + + +@pytest.mark.parametrize("seed", SEEDS) +def test_marking_rent_as_a_loss_removes_it_from_that_person_only(seed): + person, household = random_tables(seed) + before = frs_property_income(person, household) + row = seed % len(person) + was_counted = person.loc[row, "rentprof"] != LOSS + person.loc[row, "rentprof"] = LOSS + after = frs_property_income(person, household) + change = np.zeros(len(person)) + change[row] = -max(0, person.loc[row, "royyr1"]) * WEEKS_IN_YEAR * was_counted + np.testing.assert_allclose(after - before, change, atol=1e-6) + + +@pytest.mark.parametrize("seed", SEEDS) +def test_extra_rent_from_other_property_moves_only_that_person(seed): + person, household = random_tables(seed) + before = frs_property_income(person, household) + row = seed % len(person) + person.loc[row, "royyr1"] = max(0, person.loc[row, "royyr1"]) + 10 + after = frs_property_income(person, household) + change = np.zeros(len(person)) + change[row] = 10 * WEEKS_IN_YEAR * (person.loc[row, "rentprof"] != LOSS) + np.testing.assert_allclose(after - before, change, atol=1e-6) + + +@pytest.mark.parametrize("seed", SEEDS) +def test_extra_subletting_rent_moves_only_that_households_reference_person(seed): + person, household = random_tables(seed) + before = frs_property_income(person, household) + household_id = household.index[seed % len(household)] + household.loc[household_id, "subrent"] = ( + max(0, household.loc[household_id, "subrent"]) + 10 + ) + after = frs_property_income(person, household) + is_its_hrp = (person.household_id == household_id) & (person.hrpid == HRP) + np.testing.assert_allclose( + after - before, is_its_hrp * 10 * WEEKS_IN_YEAR, atol=1e-6 + ) + + +def test_built_frs_matches_the_raw_tables(frs): + """The built base FRS against a merge-based reading of the raw tables. + + Covers the call site in ``create_frs`` and the raw column names and codes. + The assertion carries no values, because the dataset is licensed microdata. + """ + raw_folder = STORAGE_FOLDER / CURRENT_FRS_RELEASE.name + if not (raw_folder / "adult.tab").exists(): + pytest.skip("Raw FRS tables not available") + adult = pd.read_csv( + raw_folder / "adult.tab", + sep="\t", + usecols=lambda c: ( + c.upper() in ("SERNUM", "PERSON", "HRPID", "ROYYR1", "RENTPROF") + ), + ) + househol = pd.read_csv( + raw_folder / "househol.tab", + sep="\t", + usecols=lambda c: c.upper() in ("SERNUM", "SUBRENT"), + ) + adult.columns = adult.columns.str.upper() + househol.columns = househol.columns.str.upper() + raw = adult.merge(househol, on="SERNUM", how="left").apply( + pd.to_numeric, errors="coerce" + ) + profit = raw.ROYYR1.clip(lower=0).where(raw.RENTPROF != 2, 0).fillna(0) + subrent = raw.SUBRENT.clip(lower=0).where(raw.HRPID == 1, 0).fillna(0) + expected = pd.Series( + ((profit + subrent) * WEEKS_IN_YEAR).values, + index=(raw.SERNUM * 1_000 + raw.PERSON).astype(int), + ) + built = pd.Series( + frs.person.property_income.values, index=frs.person.person_id.values + ) + adults_match = np.allclose(built.reindex(expected.index), expected, atol=0.01) + children_have_none = (built.drop(expected.index) == 0).all() + assert adults_match and children_have_none, ( + "property_income in the built FRS differs from the raw FRS tables" + ) diff --git a/policyengine_uk_data/tests/test_legacy_benefit_proxies.py b/policyengine_uk_data/tests/test_legacy_benefit_proxies.py index 5f1acd85..5beb95ab 100644 --- a/policyengine_uk_data/tests/test_legacy_benefit_proxies.py +++ b/policyengine_uk_data/tests/test_legacy_benefit_proxies.py @@ -442,6 +442,7 @@ def fake_read_csv(path, *args, **kwargs): "mntus1": 0, "mntusam1": 0, "redamt": 0, + "rentprof": 0, "royyr1": 0, "seincam2": 0, "sex": 1,