Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/frs-uc-gainful-self-employment.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Set policyengine-uk's Universal Credit input uc_is_in_gainful_self_employment from the FRS, as a survey proxy for UC Regs 2013 reg 64 that overrides the model's income-based default: true for every adult whose main job (EMPSTATI) is self-employment, including traders who break even or make a loss, and for anyone whose self-employment profit is above their employment income (ADM H4034); false otherwise. The SPI copy re-derives it from its own imputed incomes.
46 changes: 46 additions & 0 deletions policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,10 @@
EmploymentStatus.LONG_TERM_DISABLED.name,
EmploymentStatus.SHORT_TERM_DISABLED.name,
)
SELF_EMPLOYED_STATUSES = (
EmploymentStatus.FT_SELF_EMPLOYED.name,
EmploymentStatus.PT_SELF_EMPLOYED.name,
)
FORMULA_MODELED_EDUCATION_GRANT_VARIABLES = (
"childcare_grant",
"parents_learning_allowance",
Expand Down Expand Up @@ -392,6 +396,39 @@ def derive_is_parent_from_frs_microdata(
return is_adult_record & has_dependent_children


def derive_uc_is_in_gainful_self_employment(
employment_status, self_employment_income, employment_income
) -> np.ndarray:
"""Whether each person is in gainful self-employment for Universal Credit.

UC Regs 2013 reg 64(a) asks whether the person carries on a trade as their
main employment. DWP's Advice for Decision Making starts from hours
(H4031) but lets earnings outweigh them: someone who works more hours as
an employee but earns more from self-employment is likely to be gainfully
self-employed (H4034). So the flag is true for:

- a self-employed main job (FRS EMPSTATI, the job the respondent names as
their dominant activity, else the one with more hours), whatever its
profit: a trade can make a loss or break even and still be carried on
in expectation of profit (ADM H4013, H4054, H4503);
- a side trade whose profit is above the person's employment income.

policyengine-uk's default reads any self-employment income other than
zero as gainful self-employment, and none as none.

The flag is fixed from survey-year (or SPI-imputed) incomes. Uprating
reprices the two incomes by different indices, so in a projected year a
flagged side trade can earn less than the job, and the reverse; the flag
does not follow.
"""
self_employed_main_job = np.isin(
np.asarray(employment_status, dtype=object), SELF_EMPLOYED_STATUSES
)
profit = np.asarray(self_employment_income, dtype=float)
pay = np.asarray(employment_income, dtype=float)
return self_employed_main_job | ((profit > 0) & (profit > pay))


def _as_non_negative_array(values) -> np.ndarray:
values = np.asarray(values, dtype=float)
return np.maximum(np.nan_to_num(values, nan=0.0), 0.0)
Expand Down Expand Up @@ -1045,6 +1082,15 @@ def determine_education_level(fted_val, typeed2_val, age_val):
) * WEEKS_IN_YEAR

pe_person["self_employment_income"] = np.maximum(0, person.seincam2) * WEEKS_IN_YEAR
# policyengine-uk releases without this input skip the column and keep
# their formula.
pe_person["uc_is_in_gainful_self_employment"] = (
derive_uc_is_in_gainful_self_employment(
pe_person.employment_status,
pe_person.self_employment_income,
pe_person.employment_income,
)
)

INVERTED_BASIC_RATE = 1.25

Expand Down
17 changes: 17 additions & 0 deletions policyengine_uk_data/datasets/imputations/income.py
Original file line number Diff line number Diff line change
Expand Up @@ -292,6 +292,23 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset:
target_dataset=zero_weight_copy,
)

# The copy keeps its FRS donor's employment status but now has SPI
# incomes, so derive its gainful self-employment flag again from the
# copy's own values.
if "uc_is_in_gainful_self_employment" in zero_weight_copy.person.columns:
from policyengine_uk_data.datasets.frs import (
derive_uc_is_in_gainful_self_employment,
)

person = zero_weight_copy.person
person["uc_is_in_gainful_self_employment"] = (
derive_uc_is_in_gainful_self_employment(
person.employment_status,
person.self_employment_income,
person.employment_income,
)
)

dataset = impute_over_incomes(
dataset,
model,
Expand Down
233 changes: 233 additions & 0 deletions policyengine_uk_data/tests/test_uc_gainful_self_employment.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,233 @@
import numpy as np
import pandas as pd
import pytest
from hypothesis import given, settings
from hypothesis import strategies as st
from policyengine_uk.variables.household.income.employment_status import (
EmploymentStatus,
)

from policyengine_uk_data.datasets.frs import (
SELF_EMPLOYED_STATUSES,
derive_uc_is_in_gainful_self_employment,
)
from policyengine_uk_data.tests.test_imputation_source_flags import (
_FakeDataset,
_stack_without_remapping,
)

STATUSES = [status.name for status in EmploymentStatus]
GAINFUL = "uc_is_in_gainful_self_employment"
incomes = st.floats(0, 1e6, allow_nan=False)


def derive(status, profit, pay):
return derive_uc_is_in_gainful_self_employment([status], [profit], [pay])[0]


def oracle(status, profit, pay):
# Written out per person, independently of the vectorised helper.
if status == "FT_SELF_EMPLOYED" or status == "PT_SELF_EMPLOYED":
return True
return profit > 0 and profit > pay


def test_self_employed_statuses_are_model_enum_members():
assert set(SELF_EMPLOYED_STATUSES) <= set(STATUSES)


@pytest.mark.parametrize("status", STATUSES)
@pytest.mark.parametrize("profit", [0.0, 5_000.0])
def test_every_status_without_a_side_trade_that_out_earns_pay(status, profit):
# Pay at or above profit: only the main-job status decides.
expected = status in ("FT_SELF_EMPLOYED", "PT_SELF_EMPLOYED")
assert derive(status, profit, 10_000.0) == expected


@pytest.mark.parametrize("profit", [0.0, 1.0, 20_000.0])
def test_self_employed_main_job_is_gainful_whatever_the_profit(profit):
# A loss reaches this build floored at zero, so zero covers losses too.
for status in SELF_EMPLOYED_STATUSES:
assert derive(status, profit, 0.0)
assert derive(status, profit, 50_000.0)


def test_side_trade_counts_only_when_it_out_earns_pay():
# ADM H4034 example 1 (Jos): more employed hours, more self-employed pay.
assert derive("PT_EMPLOYED", 140 * 52, 80 * 52)
# ADM H4035 example 2 (Ann): the job earns more.
assert not derive("PT_EMPLOYED", 40 * 52, 49.6 * 52)
assert not derive("FT_EMPLOYED", 10_000.0, 10_000.0)
assert not derive("FT_EMPLOYED", 0.0, 0.0)
assert derive("UNEMPLOYED", 1.0, 0.0)


@pytest.mark.parametrize("status", STATUSES)
@pytest.mark.parametrize(
"profit, pay",
[(0.0, 0.0), (0.0, 1.0), (1.0, 0.0), (1.0, 1.0), (2.0, 1.0), (1.0, 2.0)],
)
def test_truth_table_against_oracle(status, profit, pay):
assert derive(status, profit, pay) == oracle(status, profit, pay)


@settings(max_examples=500, deadline=None)
@given(st.sampled_from(STATUSES), incomes, incomes)
def test_invariants(status, profit, pay):
gainful = derive(status, profit, pay)
# Every self-employed main job is gainful.
if status in SELF_EMPLOYED_STATUSES:
assert gainful
# Nobody else is gainful without a profit above their pay.
if gainful:
assert status in SELF_EMPLOYED_STATUSES or profit > max(pay, 0)
# More profit never removes the flag; more pay never adds it.
assert derive(status, profit * 2 + 1, pay) >= gainful
assert derive(status, profit, pay * 2 + 1) <= gainful


@settings(max_examples=100, deadline=None)
@given(st.lists(st.tuples(st.sampled_from(STATUSES), incomes, incomes), max_size=40))
def test_vectorised_matches_elementwise(rows):
statuses = [r[0] for r in rows]
profits = [r[1] for r in rows]
pays = [r[2] for r in rows]
result = derive_uc_is_in_gainful_self_employment(statuses, profits, pays)
assert result.dtype == bool
assert result.tolist() == [oracle(*r) for r in rows]
np.testing.assert_array_equal(
derive_uc_is_in_gainful_self_employment(
pd.Series(statuses, dtype="category"),
pd.Series(profits, dtype=float),
pd.Series(pays, dtype=float),
),
result,
)


def _frs_like_dataset(statuses, profits, pays):
n = len(statuses)
person = pd.DataFrame(
{
"person_id": np.arange(1, n + 1),
"person_household_id": np.arange(1, n + 1),
"person_benunit_id": np.arange(1, n + 1),
"employment_status": statuses,
"employment_income": pays,
"self_employment_income": profits,
"savings_interest_income": 0.0,
"dividend_income": 0.0,
"private_pension_income": 0.0,
"property_income": 0.0,
}
)
person[GAINFUL] = derive_uc_is_in_gainful_self_employment(
person.employment_status,
person.self_employment_income,
person.employment_income,
)
household = pd.DataFrame(
{
"household_id": np.arange(1, n + 1),
"household_weight": 1.0,
"region": "LONDON",
}
)
return _FakeDataset(person=person, household=household)


@settings(max_examples=50, deadline=None)
@given(
st.lists(
st.tuples(st.sampled_from(STATUSES), incomes, incomes, incomes, incomes),
min_size=1,
max_size=10,
)
)
def test_spi_copy_flag_follows_its_own_imputed_incomes(rows):
from policyengine_uk_data.datasets import disability_benefits
from policyengine_uk_data.datasets.imputations import frs_only
from policyengine_uk_data.datasets.imputations import income as income_module

imputed_profit = [r[3] for r in rows]
imputed_pay = [r[4] for r in rows]

def impute_over_incomes(dataset, _model, output_variables):
dataset = dataset.copy()
if "self_employment_income" in output_variables:
dataset.person["self_employment_income"] = imputed_profit
dataset.person["employment_income"] = imputed_pay
return dataset

with pytest.MonkeyPatch.context() as m:
m.setattr(income_module, "create_income_model", lambda: object())
m.setattr(income_module, "subsample_dataset", lambda d, _n: d.copy())
m.setattr(income_module, "impute_over_incomes", impute_over_incomes)
m.setattr(
frs_only,
"impute_frs_only_variables",
lambda train_dataset, target_dataset: target_dataset,
)
m.setattr(
disability_benefits,
"strip_internal_disability_reported_amounts",
lambda dataset: dataset,
)
m.setattr(income_module, "stack_datasets", _stack_without_remapping)
result = income_module.impute_income(
_frs_like_dataset(
[r[0] for r in rows], [r[1] for r in rows], [r[2] for r in rows]
)
)

person = result.person
n = len(rows)
assert len(person) == 2 * n
# The SPI copy kept the donors' statuses and took the imputed incomes.
assert person.self_employment_income.iloc[n:].tolist() == imputed_profit
assert person[GAINFUL].tolist() == [
oracle(*values)
for values in zip(
person.employment_status,
person.self_employment_income,
person.employment_income,
)
]


def test_frs_only_stage_leaves_the_rule_inputs_alone():
# The second-stage QRF rewrites these columns on the SPI copy after the
# flag's inputs are set; none of them may be an input to the flag.
from policyengine_uk_data.datasets.imputations.frs_only import (
FRS_ONLY_PERSON_VARIABLES,
)

rule_columns = {
GAINFUL,
"employment_status",
"employment_income",
"self_employment_income",
}
assert rule_columns.isdisjoint(FRS_ONLY_PERSON_VARIABLES)


@pytest.mark.parametrize("fixture", ["frs", "enhanced_frs"])
def test_built_dataset_flag(fixture, request):
# Skips only when no built dataset exists; a build without the column fails.
dataset = request.getfixturevalue(fixture)
person = dataset.person
assert GAINFUL in person.columns, f"{fixture} lacks {GAINFUL}: rebuild it"
gainful = person[GAINFUL].to_numpy(dtype=bool)
self_employed = np.isin(person.employment_status, SELF_EMPLOYED_STATUSES)
assert gainful[self_employed].all()
profit = person.self_employment_income.to_numpy()
assert (profit[gainful & ~self_employed] > 0).all()
if fixture == "frs":
# Later enhanced-FRS stages reprice incomes, so the exact rule holds on
# the base build only.
np.testing.assert_array_equal(
gainful,
derive_uc_is_in_gainful_self_employment(
person.employment_status, profit, person.employment_income
),
)
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,7 @@ dependencies = [
dev = [
"ruff>=0.9.0",
"pytest",
"hypothesis",
"torch",
"l0-python>=0.4.0",
"tables",
Expand Down
Loading
Loading