Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/frs-boarder-lodger-rent.added.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Carry the rent that boarders and lodgers pay the householder (FRS CVPAY, split by CONVBL) as the person inputs `rent_paid_as_boarder` and `rent_paid_as_lodger`.
39 changes: 39 additions & 0 deletions policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -248,6 +248,41 @@ def frs_property_income(person: pd.DataFrame, household: pd.DataFrame) -> np.nda
).values


def frs_boarder_and_lodger_rent(person: pd.DataFrame) -> tuple[np.ndarray, np.ndarray]:
"""Annual rent each person pays the householder as a boarder and as a lodger.

CVPAY is the weekly rent a boarder or lodger pays the householder, held
on the payer's own adult record. The FRS asks it about each person not
related to the household reference person in the second and later
benefit units of a conventional household (question CvPay), and CONVBL
says which the person is:

- 1, a boarder, "someone who pays you a rent for board and lodging";
- 2, a lodger, "someone who pays you a rent for lodging, but not food".

A positive CVPAY with any other CONVBL is classed as a lodger's.

CVPAY is the amount after deducting any state benefits to help with rent.
The FRS derived variables BOARDER and LODGER add housing benefit paid for
the benefit unit (HBOTHAMT); they equal the unit's summed CVPAY in every
paying unit of the raw 2023-24 and 2024-25 data.

The released data hold this rent only on the payer's record; there is
no separate variable for what the householder receives, and the FRS
gross-income derivation leaves it out. policyengine-uk works out the
householder's receipt from these two inputs.

Weekly amounts are annualised with ``WEEKS_IN_YEAR`` (365.25 / 7), as for
every other weekly FRS amount. policyengine-uk converts annual amounts
back to weekly with 52 weeks, so a weekly amount reaches its weekly
disregards about 0.34% higher; that convention gap is not specific to
these columns.
"""
rent_paid = np.maximum(0, person.cvpay.fillna(0).values) * WEEKS_IN_YEAR
is_boarder = person.convbl.values == 1
return rent_paid * is_boarder, rent_paid * ~is_boarder


def derive_is_in_non_advanced_education(
current_education,
is_apprentice=None,
Expand Down Expand Up @@ -1112,6 +1147,10 @@ def determine_education_level(fted_val, typeed2_val, age_val):
* 52,
)
pe_person["property_income"] = frs_property_income(person, household)
(
pe_person["rent_paid_as_boarder"],
pe_person["rent_paid_as_lodger"],
) = frs_boarder_and_lodger_rent(person)
maintenance_to_self = np.maximum(
pd.Series(np.where(person.mntus1 == 2, person.mntusam1, person.mntamt1)).fillna(
0
Expand Down
2 changes: 2 additions & 0 deletions policyengine_uk_data/storage/uprating_factors.csv
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,8 @@ private_pension_income,1.0,1.003,1.053,1.106,1.161,1.216,1.261,1.288,1.315,1.346
private_transfer_income,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
property_income,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
recreation_consumption,1.0,1.04,1.144,1.209,1.237,1.277,1.301,1.327,1.353,1.38,1.38,1.38,1.38,1.38,1.38
rent_paid_as_boarder,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
rent_paid_as_lodger,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
restaurants_and_hotels_consumption,1.0,1.04,1.144,1.209,1.237,1.277,1.301,1.327,1.353,1.38,1.38,1.38,1.38,1.38,1.38
savings,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
savings_interest_income,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
Expand Down
2 changes: 2 additions & 0 deletions policyengine_uk_data/storage/uprating_growth_factors.csv
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,8 @@ private_pension_income,0,0.003,0.05,0.05,0.05,0.047,0.037,0.021,0.021,0.024,0.0,
private_transfer_income,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
property_income,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
recreation_consumption,0,0.04,0.1,0.057,0.023,0.032,0.019,0.02,0.02,0.02,0.0,0.0,0.0,0.0,0.0
rent_paid_as_boarder,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
rent_paid_as_lodger,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
restaurants_and_hotels_consumption,0,0.04,0.1,0.057,0.023,0.032,0.019,0.02,0.02,0.02,0.0,0.0,0.0,0.0,0.0
savings,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
savings_interest_income,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
Expand Down
183 changes: 183 additions & 0 deletions policyengine_uk_data/tests/test_frs_boarder_lodger_rent.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,183 @@
import numpy as np
import pandas as pd
import pytest

from policyengine_uk_data.datasets.frs import (
WEEKS_IN_YEAR,
frs_boarder_and_lodger_rent,
frs_property_income,
)

HRP, NOT_HRP = 1, 2
BOARDER, LODGER, NEITHER, NOT_ASKED = 1, 2, 3, 0
OWNED_OUTRIGHT = 6


def make_person(people):
return pd.DataFrame(people, columns=["person_id", "convbl", "cvpay"])


def test_a_boarder_pays_rent_as_a_boarder():
boarder, lodger = frs_boarder_and_lodger_rent(
make_person([(1_001, NOT_ASKED, 0), (1_002, BOARDER, 120)])
)
np.testing.assert_allclose(boarder, [0, 120 * WEEKS_IN_YEAR])
assert lodger.tolist() == [0, 0]


def test_a_lodger_pays_rent_as_a_lodger():
boarder, lodger = frs_boarder_and_lodger_rent(
make_person([(1_001, NOT_ASKED, 0), (1_002, LODGER, 100)])
)
assert boarder.tolist() == [0, 0]
np.testing.assert_allclose(lodger, [0, 100 * WEEKS_IN_YEAR])


@pytest.mark.parametrize("convbl", [NEITHER, NOT_ASKED, -1, np.nan])
def test_rent_without_a_boarder_code_is_a_lodgers(convbl):
boarder, lodger = frs_boarder_and_lodger_rent(make_person([(1_001, convbl, 90)]))
assert boarder.tolist() == [0]
np.testing.assert_allclose(lodger, [90 * WEEKS_IN_YEAR])


@pytest.mark.parametrize("cvpay", [0, -1, -30, np.nan])
@pytest.mark.parametrize("convbl", [BOARDER, LODGER, NEITHER])
def test_no_positive_amount_means_no_rent(convbl, cvpay):
boarder, lodger = frs_boarder_and_lodger_rent(make_person([(1_001, convbl, cvpay)]))
assert boarder.tolist() == [0]
assert lodger.tolist() == [0]


def test_adult_and_child_rows_sharing_index_labels():
# create_frs stacks the adult and child tables and fills the gaps with
# zero, so index labels repeat and child rows carry CVPAY 0 and CONVBL 0.
adults = make_person([(1_001, NOT_ASKED, 0), (1_002, BOARDER, 70)])
children = pd.DataFrame({"person_id": [1_003]})
person = pd.concat([adults, children]).sort_index(kind="stable").fillna(0)
assert person.index.tolist() == [0, 0, 1]
boarder, lodger = frs_boarder_and_lodger_rent(person)
np.testing.assert_allclose(boarder, [0, 0, 70 * WEEKS_IN_YEAR])
assert lodger.tolist() == [0, 0, 0]


def random_tables(seed: int):
"""Random households of one to four adults; the first is the HRP."""
rng = np.random.default_rng(seed)
people, households = [], []
for household_id in range(1, rng.integers(1, 30) + 1):
households.append((household_id, int(rng.integers(1, 9)), 0.0))
for person in range(1, rng.integers(1, 5) + 1):
people.append(
(
household_id,
household_id * 1_000 + person,
HRP if person == 1 else NOT_HRP,
float(rng.choice([0, rng.uniform(0, 2_000)])),
rng.choice([BOARDER, LODGER, NEITHER, NOT_ASKED, -1, np.nan]),
rng.choice([0, rng.uniform(0, 400), -1, np.nan]),
)
)
person = pd.DataFrame(
people,
columns=["household_id", "person_id", "hrpid", "royyr1", "convbl", "cvpay"],
)
household = pd.DataFrame(
households, columns=["household_id", "tentyp2", "subrent"]
).set_index("household_id")
return person, household


SEEDS = range(200)


@pytest.mark.parametrize("seed", SEEDS)
def test_boarder_and_lodger_rent_conserves_reported_rent(seed):
# Every positive CVPAY is counted once, in exactly one of the two classes.
person, _ = random_tables(seed)
boarder, lodger = frs_boarder_and_lodger_rent(person)
assert (boarder >= 0).all() and (lodger >= 0).all()
assert not ((boarder > 0) & (lodger > 0)).any()
paid = person.cvpay.where(person.cvpay > 0, 0)
np.testing.assert_allclose(boarder + lodger, paid * WEEKS_IN_YEAR)
np.testing.assert_allclose(
boarder.sum(), paid[person.convbl == BOARDER].sum() * WEEKS_IN_YEAR
)


@pytest.mark.parametrize("seed", SEEDS)
def test_extra_rent_moves_only_the_person_who_pays_it(seed):
person, _ = random_tables(seed)
person["cvpay"] = person.cvpay.where(person.cvpay > 0, 0)
before = sum(frs_boarder_and_lodger_rent(person))
row = seed % len(person)
person.loc[row, "cvpay"] += 10
after = sum(frs_boarder_and_lodger_rent(person))
change = np.zeros(len(person))
change[row] = 10 * WEEKS_IN_YEAR
np.testing.assert_allclose(after - before, change)


@pytest.mark.parametrize("seed", SEEDS)
def test_rent_paid_and_property_income_do_not_affect_each_other(seed):
person, household = random_tables(seed)
property_income = frs_property_income(person.fillna(0), household)
rent_paid = frs_boarder_and_lodger_rent(person)
np.testing.assert_array_equal(
frs_property_income(
person.fillna(0).assign(cvpay=0.0, convbl=NOT_ASKED), household
),
property_income,
)
other_property = person.assign(royyr1=person.royyr1 + 50)
for changed, original in zip(
frs_boarder_and_lodger_rent(other_property), rent_paid
):
np.testing.assert_array_equal(changed, original)


RENT_COLUMNS = ["rent_paid_as_boarder", "rent_paid_as_lodger"]


@pytest.mark.parametrize(
"table", ["uprating_factors.csv", "uprating_growth_factors.csv"]
)
def test_rent_paid_is_uprated_like_sublet_income(table):
# policyengine-uk uprates both inputs with the per capita GDP index it
# uses for sublet_income, so their rows must match.
from policyengine_uk_data.storage import STORAGE_FOLDER

factors = pd.read_csv(STORAGE_FOLDER / table).set_index("Variable")
for column in RENT_COLUMNS:
pd.testing.assert_series_equal(
factors.loc[column], factors.loc["sublet_income"], check_names=False
)


def test_rent_paid_is_uprated_to_the_calibration_year():
# The build materialises the survey-year dataset at the calibration year
# before calibrating; the rent paid must move with it.
from policyengine_uk.data import UKSingleYearDataset

from policyengine_uk_data.storage import STORAGE_FOLDER
from policyengine_uk_data.utils.uprating import uprate_dataset

person = pd.DataFrame(
{
"person_id": [1_001, 1_002],
"person_benunit_id": [101, 102],
"person_household_id": [1, 1],
"rent_paid_as_boarder": [0.0, 100 * WEEKS_IN_YEAR],
"rent_paid_as_lodger": [80 * WEEKS_IN_YEAR, 0.0],
}
)
benunit = pd.DataFrame({"benunit_id": [101, 102]})
household = pd.DataFrame({"household_id": [1], "household_weight": [1.0]})
dataset = UKSingleYearDataset(
person=person, benunit=benunit, household=household, fiscal_year=2024
)
uprated = uprate_dataset(dataset, 2025)
factors = pd.read_csv(STORAGE_FOLDER / "uprating_factors.csv").set_index("Variable")
growth = factors.loc["sublet_income", "2025"] / factors.loc["sublet_income", "2024"]
assert growth > 1
for column in RENT_COLUMNS:
np.testing.assert_allclose(uprated.person[column], person[column] * growth)
20 changes: 18 additions & 2 deletions policyengine_uk_data/tests/test_legacy_benefit_proxies.py
Original file line number Diff line number Diff line change
@@ -1,9 +1,11 @@
import numpy as np
import pandas as pd
import policyengine_uk
import pytest
import policyengine_uk_data.datasets.frs as frs_module

from policyengine_uk_data.datasets.frs import (
WEEKS_IN_YEAR,
add_legacy_benefit_proxies,
attach_legacy_benefit_proxies_from_frs_person,
apply_legacy_benefit_proxies,
Expand Down Expand Up @@ -373,7 +375,13 @@ def calculate(self, variable, year=None):
raise KeyError(variable)


def test_create_frs_smoke_includes_legacy_proxy_columns(tmp_path, monkeypatch):
@pytest.mark.parametrize(
"convbl, cvpay, boarder_weekly, lodger_weekly",
[(0, 0, 0, 0), (1, 100, 100, 0), (2, 80, 0, 80)],
)
def test_create_frs_smoke_includes_legacy_proxy_columns(
tmp_path, monkeypatch, convbl, cvpay, boarder_weekly, lodger_weekly
):
original_read_csv = frs_module.pd.read_csv

def fake_read_csv(path, *args, **kwargs):
Expand Down Expand Up @@ -421,7 +429,8 @@ def fake_read_csv(path, *args, **kwargs):
"ademaamt": 0,
"age": 30,
"age80": 30,
"cvpay": 0,
"convbl": convbl,
"cvpay": cvpay,
"educft": 0,
"educqual": 0,
"eduma": 0,
Expand Down Expand Up @@ -561,3 +570,10 @@ def fake_read_csv(path, *args, **kwargs):
].iloc[0]
assert dataset.person["education_grants"].iloc[0] == 100
assert dataset.person["disabled_students_allowance_eligible_expenses"].iloc[0] == 0
# create_frs carries the rent this person pays the householder, by class.
assert dataset.person["rent_paid_as_boarder"].iloc[0] == pytest.approx(
boarder_weekly * WEEKS_IN_YEAR
)
assert dataset.person["rent_paid_as_lodger"].iloc[0] == pytest.approx(
lodger_weekly * WEEKS_IN_YEAR
)