Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/frs-cvpay-not-property-income.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Stop counting the rent a boarder or lodger pays (FRS CVPAY) as that boarder's or lodger's own property income.
48 changes: 29 additions & 19 deletions policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -220,6 +220,34 @@ def derive_receives_benefits_in_own_right(pe_person: pd.DataFrame) -> pd.Series:
)


def frs_property_income(person: pd.DataFrame, household: pd.DataFrame) -> np.ndarray:
"""Annual property income each person reports in the FRS.

Two FRS amounts, both weekly in the released data:

- SUBRENT, rent the household received for letting part of its home to
someone outside the household. It goes to the household reference
person, and only in owner-occupied households (TENTYP2 5 or 6).
``household`` must be indexed by ``household_id``.
- ROYYR1, the person's rent from other property, before tax and after
allowable expenses.

CVPAY is not included. It is the rent that a boarder or lodger pays the
householder, and it sits on the boarder's or lodger's own adult record.
The FRS question (CvPay) asks how much rent [name] paid for board and
lodging, after deducting any state benefits to help with rent.
"""
is_head = person.hrpid == 1
household_property_income = household.tentyp2.isin((5, 6)) * household.subrent
persons_household_property_income = (
household_property_income.reindex(person.household_id).fillna(0).values
)
return (
np.maximum(0, is_head * persons_household_property_income + person.royyr1)
* WEEKS_IN_YEAR
).values


def derive_is_in_non_advanced_education(
current_education,
is_apprentice=None,
Expand Down Expand Up @@ -1083,25 +1111,7 @@ def determine_education_level(fted_val, typeed2_val, age_val):
)
* 52,
)
is_head = person.hrpid == 1
household_property_income = (
household.tentyp2.isin((5, 6)) * household.subrent
) # Owned and subletting
persons_household_property_income = (
pd.Series(
household_property_income[person.household_id].values,
index=person.person_id,
)
.fillna(0)
.values
)
pe_person["property_income"] = (
np.maximum(
0,
is_head * persons_household_property_income + person.cvpay + person.royyr1,
)
* WEEKS_IN_YEAR
)
pe_person["property_income"] = frs_property_income(person, household)
maintenance_to_self = np.maximum(
pd.Series(np.where(person.mntus1 == 2, person.mntusam1, person.mntamt1)).fillna(
0
Expand Down
129 changes: 129 additions & 0 deletions policyengine_uk_data/tests/test_frs_property_income.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,129 @@
import numpy as np
import pandas as pd
import pytest

from policyengine_uk_data.datasets.frs import WEEKS_IN_YEAR, frs_property_income

HRP, NOT_HRP = 1, 2
OWNED_WITH_MORTGAGE, OWNED_OUTRIGHT = 5, 6
COUNCIL_RENTED, PRIVATE_RENTED_FURNISHED = 1, 4


def make_tables(people, households):
person = pd.DataFrame(
people, columns=["household_id", "person_id", "hrpid", "royyr1", "cvpay"]
)
household = pd.DataFrame(
households, columns=["household_id", "tentyp2", "subrent"]
).set_index("household_id")
return person, household


def test_rent_paid_by_a_lodger_is_not_their_property_income():
# Owner-occupier household with a lodger who pays £100 a week (CVPAY).
person, household = make_tables(
[(1, 1_001, HRP, 0, 0), (1, 1_002, NOT_HRP, 0, 100)],
[(1, OWNED_OUTRIGHT, 0)],
)
assert frs_property_income(person, household).tolist() == [0, 0]


def test_rent_from_other_property_counts_for_any_adult():
person, household = make_tables(
[(1, 1_001, HRP, 0, 0), (1, 1_002, NOT_HRP, 50, 0)],
[(1, COUNCIL_RENTED, 0)],
)
np.testing.assert_allclose(
frs_property_income(person, household), [0, 50 * WEEKS_IN_YEAR]
)


def test_subletting_rent_goes_to_the_owner_household_reference_person():
person, household = make_tables(
[(1, 1_001, NOT_HRP, 0, 0), (1, 1_002, HRP, 0, 0)],
[(1, OWNED_WITH_MORTGAGE, 80)],
)
np.testing.assert_allclose(
frs_property_income(person, household), [0, 80 * WEEKS_IN_YEAR]
)


def test_negative_amounts_do_not_become_negative_income():
person, household = make_tables(
[(1, 1_001, HRP, -30, 0), (1, 1_002, NOT_HRP, -5, 0)],
[(1, OWNED_OUTRIGHT, 10)],
)
assert frs_property_income(person, household).tolist() == [0, 0]


def test_adult_and_child_rows_sharing_index_labels():
# create_frs stacks the adult and child tables, so index labels repeat.
adults, household = make_tables(
[(1, 1_001, HRP, 20, 0), (2, 2_001, HRP, 0, 60)],
[(1, OWNED_OUTRIGHT, 80), (2, PRIVATE_RENTED_FURNISHED, 0)],
)
children, _ = make_tables([(1, 1_002, 0, 0, 0)], [])
person = pd.concat([adults, children]).sort_index(kind="stable")
assert person.index.tolist() == [0, 0, 1]
np.testing.assert_allclose(
frs_property_income(person, household),
[100 * WEEKS_IN_YEAR, 0, 0],
)


def random_tables(seed: int):
"""Random households of one to four adults; the first is the HRP."""
rng = np.random.default_rng(seed)
people, households = [], []
for household_id in range(1, rng.integers(1, 30) + 1):
tenure = int(rng.integers(1, 9))
subrent = float(rng.choice([0, rng.uniform(0, 500)]))
households.append((household_id, tenure, subrent))
for person in range(1, rng.integers(1, 5) + 1):
people.append(
(
household_id,
household_id * 1_000 + person,
HRP if person == 1 else NOT_HRP,
float(rng.choice([0, rng.uniform(0, 2_000)])),
float(rng.choice([0, rng.uniform(0, 400)])),
)
)
return make_tables(people, households)


SEEDS = range(200)


@pytest.mark.parametrize("seed", SEEDS)
def test_property_income_does_not_depend_on_cvpay(seed):
person, household = random_tables(seed)
without_cvpay = person.assign(cvpay=0.0)
np.testing.assert_array_equal(
frs_property_income(person, household),
frs_property_income(without_cvpay, household),
)


@pytest.mark.parametrize("seed", SEEDS)
def test_property_income_conserves_reported_rent(seed):
# Each owner household's SUBRENT is counted once (on its HRP) and every
# ROYYR1 is counted on its own record, so the totals match.
person, household = random_tables(seed)
result = frs_property_income(person, household)
assert (result >= 0).all()
owner = household.tentyp2.isin((OWNED_WITH_MORTGAGE, OWNED_OUTRIGHT))
expected = (household.subrent[owner].sum() + person.royyr1.sum()) * WEEKS_IN_YEAR
assert result.sum() == pytest.approx(expected)


@pytest.mark.parametrize("seed", SEEDS)
def test_extra_rent_from_other_property_moves_only_that_person(seed):
person, household = random_tables(seed)
before = frs_property_income(person, household)
row = seed % len(person)
person.loc[row, "royyr1"] += 10
after = frs_property_income(person, household)
change = np.zeros(len(person))
change[row] = 10 * WEEKS_IN_YEAR
np.testing.assert_allclose(after - before, change)
Loading