Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/frs-trading-loss.added.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
FRS self-employment losses (negative SEINCAM2) now go into policyengine-uk's `trading_loss` input as a positive amount, instead of being floored away; `self_employment_income` stays the non-negative profit. SPI-donor rows of the enhanced FRS carry no trading loss.
22 changes: 21 additions & 1 deletion policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -397,6 +397,23 @@ def _as_non_negative_array(values) -> np.ndarray:
return np.maximum(np.nan_to_num(values, nan=0.0), 0.0)


def split_self_employment_profit(weekly_profit) -> tuple[np.ndarray, np.ndarray]:
"""Split FRS weekly self-employment profit (SEINCAM2) into annual
``self_employment_income`` and ``trading_loss``.

SEINCAM2 keeps losses as negative values ("Any losses are recorded as
such", FRS methodology glossary). policyengine-uk wants profits and losses
as two non-negative inputs, because its programmes treat a loss
differently: Income Tax and tax credits set it against other income,
means-tested benefits do not, and HBAI counts it as negative income. So the
profit goes to ``self_employment_income`` and the loss, as a positive
amount, to ``trading_loss``. Their difference is the reported profit.
"""
annual = np.nan_to_num(np.asarray(weekly_profit, dtype=float), nan=0.0)
annual = annual * WEEKS_IN_YEAR
return np.maximum(annual, 0.0), np.maximum(-annual, 0.0)


def allocate_reported_education_grants(
reported_grants, grant_capacities: dict[str, np.ndarray]
) -> dict[str, np.ndarray]:
Expand Down Expand Up @@ -1044,7 +1061,10 @@ def determine_education_level(fted_val, typeed2_val, age_val):
pension_payment + pension_tax_paid + pension_deductions_removed
) * WEEKS_IN_YEAR

pe_person["self_employment_income"] = np.maximum(0, person.seincam2) * WEEKS_IN_YEAR
(
pe_person["self_employment_income"],
pe_person["trading_loss"],
) = split_self_employment_profit(person.seincam2)

INVERTED_BASIC_RATE = 1.25

Expand Down
3 changes: 3 additions & 0 deletions policyengine_uk_data/datasets/imputations/frs_only.py
Original file line number Diff line number Diff line change
Expand Up @@ -100,6 +100,9 @@
"jsa_income_reported",
"esa_contrib_reported",
"esa_income_reported",
# Self-employment losses. Last, so the QRF's sequential imputation of
# every variable above is unchanged by it.
"trading_loss",
]


Expand Down
10 changes: 10 additions & 0 deletions policyengine_uk_data/datasets/imputations/income.py
Original file line number Diff line number Diff line change
Expand Up @@ -292,6 +292,16 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset:
target_dataset=zero_weight_copy,
)

# The second stage imputes trading_loss from FRS respondents with similar
# demographics and imputed incomes (the SPI has no current-year loss).
# SEINCAM2 nets a person's trades, so an FRS respondent has a profit or a
# loss, never both; keep the SPI-donor rows the same.
if "trading_loss" in zero_weight_copy.person.columns:
person = zero_weight_copy.person
person["trading_loss"] = np.where(
person["self_employment_income"] > 0, 0.0, person["trading_loss"]
)

dataset = impute_over_incomes(
dataset,
model,
Expand Down
1 change: 1 addition & 0 deletions policyengine_uk_data/storage/uprating_factors.csv
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,7 @@ statutory_paternity_pay,1.0,1.04,1.144,1.209,1.237,1.277,1.301,1.327,1.353,1.38,
statutory_sick_pay,1.0,1.04,1.144,1.209,1.237,1.277,1.301,1.327,1.353,1.38,1.38,1.38,1.38,1.38,1.38
student_loan_repayments,1.0,1.059,1.127,1.205,1.261,1.308,1.337,1.365,1.396,1.431,1.431,1.431,1.431,1.431,1.431
sublet_income,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
trading_loss,1.0,1.0,1.063,1.089,1.141,1.194,1.231,1.27,1.315,1.365,1.365,1.365,1.365,1.365,1.365
transport_consumption,1.0,1.04,1.144,1.209,1.237,1.277,1.301,1.327,1.353,1.38,1.38,1.38,1.38,1.38,1.38
universal_credit_reported,1.0,1.04,1.144,1.209,1.237,1.277,1.301,1.327,1.353,1.38,1.38,1.38,1.38,1.38,1.38
water_and_sewerage_charges,1.0,1.0,1.0,1.092,1.14,1.21,1.283,1.349,1.4,1.46,1.46,1.46,1.46,1.46,1.46
Expand Down
1 change: 1 addition & 0 deletions policyengine_uk_data/storage/uprating_growth_factors.csv
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,7 @@ statutory_paternity_pay,0,0.04,0.1,0.057,0.023,0.032,0.019,0.02,0.02,0.02,0.0,0.
statutory_sick_pay,0,0.04,0.1,0.057,0.023,0.032,0.019,0.02,0.02,0.02,0.0,0.0,0.0,0.0,0.0
student_loan_repayments,0,0.059,0.064,0.069,0.046,0.037,0.022,0.021,0.023,0.025,0.0,0.0,0.0,0.0,0.0
sublet_income,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
trading_loss,0,0.0,0.063,0.024,0.048,0.046,0.031,0.032,0.035,0.038,0.0,0.0,0.0,0.0,0.0
transport_consumption,0,0.04,0.1,0.057,0.023,0.032,0.019,0.02,0.02,0.02,0.0,0.0,0.0,0.0,0.0
universal_credit_reported,0,0.04,0.1,0.057,0.023,0.032,0.019,0.02,0.02,0.02,0.0,0.0,0.0,0.0,0.0
water_and_sewerage_charges,0,0.0,0.0,0.092,0.044,0.061,0.06,0.051,0.038,0.043,0.0,0.0,0.0,0.0,0.0
Expand Down
1 change: 1 addition & 0 deletions policyengine_uk_data/tests/test_non_negative_incomes.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
INCOME_VARIABLES = [
"employment_income",
"self_employment_income",
"trading_loss",
"tax_free_savings_income",
"savings_interest_income",
"dividend_income",
Expand Down
95 changes: 95 additions & 0 deletions policyengine_uk_data/tests/test_trading_loss.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
"""FRS self-employment profit is split into a profit and a trading loss.

SEINCAM2 records a loss as a negative profit. The build puts the profit in
``self_employment_income`` and the loss, as a positive amount, in
``trading_loss``, so policyengine-uk can apply each programme's own loss rule.

Invariants, for any weekly profit (including missing values):

1. Both outputs are non-negative and at most one is positive.
2. Conservation: profit less loss is the annualised reported profit.
3. Profit never falls and loss never rises as the reported profit rises.

And for the built datasets:

4. No FRS person has both a profit and a loss, and no loss is negative.
5. The same holds on the SPI-donor rows of the enhanced FRS, whose losses the
second-stage QRF imputes from FRS respondents with similar incomes.

The data checks assert on counts only, so a failure never prints a record's
amounts (the FRS is licensed microdata).
"""

import numpy as np
import pytest
from hypothesis import given, settings
from hypothesis import strategies as st
from hypothesis.extra.numpy import arrays

from policyengine_uk_data.datasets.frs import (
WEEKS_IN_YEAR,
split_self_employment_profit,
)

weekly_profits = arrays(
np.float64,
st.integers(1, 50),
elements=st.one_of(
st.floats(-50_000, 50_000, allow_nan=False, allow_infinity=False),
st.just(0.0),
st.just(np.nan),
),
)


@settings(max_examples=200, deadline=None, derandomize=True)
@given(weekly_profits)
def test_split_is_non_negative_exclusive_and_conserves_profit(weekly):
income, loss = split_self_employment_profit(weekly)
assert np.all(income >= 0) and np.all(loss >= 0)
assert np.all((income == 0) | (loss == 0))
reported = np.nan_to_num(weekly, nan=0.0) * WEEKS_IN_YEAR
np.testing.assert_allclose(income - loss, reported, rtol=0, atol=1e-6)


@settings(max_examples=200, deadline=None, derandomize=True)
@given(weekly_profits, st.floats(0, 10_000, allow_nan=False))
def test_split_is_monotone_in_reported_profit(weekly, rise):
income, loss = split_self_employment_profit(weekly)
higher_income, higher_loss = split_self_employment_profit(
np.nan_to_num(weekly, nan=0.0) + rise
)
assert np.all(higher_income >= income)
assert np.all(higher_loss <= loss)


def _violations(person, mask=None) -> dict:
income = person["self_employment_income"].to_numpy()
loss = person["trading_loss"].to_numpy()
keep = np.ones(len(person), dtype=bool) if mask is None else mask
return {
"negative_loss": int((loss[keep] < 0).sum()),
"profit_and_loss": int(((income > 0) & (loss > 0))[keep].sum()),
}


def test_frs_profit_and_loss_never_both_positive(frs):
if "trading_loss" not in frs.person.columns:
pytest.skip("Dataset built before trading_loss was added")
counts = _violations(frs.person)
assert counts == {"negative_loss": 0, "profit_and_loss": 0}, counts


def test_spi_donor_rows_never_have_both(enhanced_frs):
person = enhanced_frs.person
if "trading_loss" not in person.columns:
pytest.skip("Dataset built before trading_loss was added")
household = enhanced_frs.household
synthetic_households = household.household_id[
household.household_is_spi_synthetic.astype(bool)
]
synthetic = person.person_household_id.isin(synthetic_households).to_numpy()
assert int(synthetic.sum()) > 0, "expected SPI-donor rows in the enhanced FRS"
counts = _violations(person, synthetic)
assert counts == {"negative_loss": 0, "profit_and_loss": 0}, counts
assert _violations(person)["negative_loss"] == 0
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,7 @@ dependencies = [
dev = [
"ruff>=0.9.0",
"pytest",
"hypothesis",
"torch",
"l0-python>=0.4.0",
"tables",
Expand Down
Loading
Loading