Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions changelog.d/frs-reported-dividends.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
- FRS investment-account dividends are now summed to the person who holds the account. They were keyed on each person's row position, so almost none matched and the base FRS carried about £40m of dividends instead of about £8bn in 2024-25.
- The enhanced FRS keeps those reported dividends instead of replacing them with draws from the SPI income model, which predicts from age, gender and region alone and gave Universal Credit claimants six-figure dividends (policyengine-uk#1948). The SPI-donor half still carries the SPI income distribution.
37 changes: 24 additions & 13 deletions policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -570,6 +570,29 @@ def validate_frs_survey_year(raw_frs_folder, year: int) -> None:
)


def frs_dividend_income(account: pd.DataFrame, person_ids) -> np.ndarray:
"""Annual dividends each person reports on FRS investment accounts.

Gilt-edged stock taxed at source (account type 6), unit and investment
trusts (7) and stocks and shares (8). Amounts taxed at source are grossed
up at the basic rate. Summed to ``person_ids``, which must be the same
``household_id * 1e3 + person`` keys as ``account.person_id``: keying on
the positional index instead matched almost no accounts, leaving the FRS
with about £40m of dividends rather than about £8bn in 2024-25.
"""
INVERTED_BASIC_RATE = 1.25
dividends = (
account.accint * np.where(account.invtax == 1, INVERTED_BASIC_RATE, 1)
) * (
((account.account == 6) & (account.invtax == 1)) # GGES
| account.account.isin((7, 8)) # Stocks/shares/UITs
)
return np.maximum(
0,
sum_to_entity(dividends, account.person_id, person_ids) * WEEKS_IN_YEAR,
)


def create_frs(
raw_frs_folder: str,
year: int,
Expand Down Expand Up @@ -1070,19 +1093,7 @@ def determine_education_level(fted_val, typeed2_val, age_val):
0,
taxable_savings_interest + pe_person["tax_free_savings_income"].values,
)
pe_person["dividend_income"] = np.maximum(
0,
sum_to_entity(
(account.accint * np.where(account.invtax == 1, INVERTED_BASIC_RATE, 1))
* (
((account.account == 6) & (account.invtax == 1)) # GGES
| account.account.isin((7, 8)) # Stocks/shares/UITs
),
account.person_id,
person.index,
)
* 52,
)
pe_person["dividend_income"] = frs_dividend_income(account, person.person_id)
is_head = person.hrpid == 1
household_property_income = (
household.tentyp2.isin((5, 6)) * household.subrent
Expand Down
19 changes: 11 additions & 8 deletions policyengine_uk_data/datasets/imputations/income.py
Original file line number Diff line number Diff line change
Expand Up @@ -223,8 +223,9 @@ def impute_over_incomes(

# Housing costs (rent, mortgage interest, mortgage capital) used to be
# rescaled here by new_income_total / original_income_total across
# INCOME_COMPONENTS. Because FRS dividend_income is near-zero and the
# SPI-trained QRF predicts materially larger dividends, the ratio
# INCOME_COMPONENTS. Because FRS dividend_income was then near-zero (a
# keying error in frs.py, since fixed) and the SPI-trained QRF predicts
# materially larger dividends, the ratio
# inflated rent/mortgage by ~2.5× uniformly in the built enhanced FRS
# — pushing AHC poverty rates 10–18 pp above HBAI for non-pensioners
# (see issue #367). Housing costs now pass through unchanged; their
Expand Down Expand Up @@ -266,7 +267,7 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset:

model = create_income_model()

# Impute just dividends on the original, full variable set on the copy
# Impute the full income set on the SPI-donor copy only.

zero_weight_copy = impute_over_incomes(
zero_weight_copy,
Expand All @@ -292,11 +293,13 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset:
target_dataset=zero_weight_copy,
)

dataset = impute_over_incomes(
dataset,
model,
["dividend_income"],
)
# The FRS half keeps its reported dividends. Replacing them with a draw
# from the SPI model, whose only predictors are age, gender and region,
# gave dividends to FRS respondents without regard to their investments,
# earnings or benefits, so Universal Credit claimants received them as
# often as anyone else (policyengine-uk#1948). The SPI-donor half carries
# the SPI dividend distribution, and calibration to the HMRC dividend
# targets reweights between the two.

zero_weight_copy.validate()
dataset.validate()
Expand Down
40 changes: 40 additions & 0 deletions policyengine_uk_data/tests/test_frs_dividend_income.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,40 @@
"""FRS dividends land on the person who holds the account (policyengine-uk#1948)."""

import numpy as np
import pandas as pd
import pytest

from policyengine_uk_data.datasets.frs import WEEKS_IN_YEAR, frs_dividend_income


def test_dividends_are_keyed_on_person_id_not_row_position():
# Person ids are household_id * 1000 + person number, so they never
# coincide with the positional index the rows happen to sit at.
person_ids = pd.Series([1_001, 1_002, 2_001, 2_002])
account = pd.DataFrame(
{
"person_id": [1_002, 2_001, 2_001, 2_002, 1_001, 2_002],
# 8: stocks and shares; 7: unit/investment trusts; 6: gilts; 1: bank account
"account": [8, 7, 6, 6, 1, 8],
"accint": [10.0, 4.0, 2.0, 3.0, 50.0, 0.0],
"invtax": [2, 1, 1, 2, 2, 2],
}
)
dividends = frs_dividend_income(account, person_ids)
weekly = np.array(
[
0.0, # 1,001 has only a bank account
10.0, # 1,002: shares, not taxed at source
4.0 * 1.25 + 2.0 * 1.25, # 2,001: trusts and gilts taxed at source
0.0, # 2,002: gilts not taxed at source are excluded; shares pay 0
]
)
assert dividends == pytest.approx(weekly * WEEKS_IN_YEAR)


def test_dividends_are_never_negative():
person_ids = pd.Series([1_001])
account = pd.DataFrame(
{"person_id": [1_001], "account": [8], "accint": [-5.0], "invtax": [2]}
)
assert frs_dividend_income(account, person_ids).tolist() == [0.0]
56 changes: 56 additions & 0 deletions policyengine_uk_data/tests/test_imputation_source_flags.py
Original file line number Diff line number Diff line change
Expand Up @@ -131,3 +131,59 @@ def test_impute_capital_gains_marks_capital_gains_clone_households(monkeypatch):
True,
]
assert result.household.loc[2:, "household_weight"].eq(1).all()


def test_impute_income_keeps_reported_dividends_on_the_frs_half(monkeypatch):
"""SPI draws replace incomes only on the SPI-donor copy (policyengine-uk#1948).

The SPI income model predicts from age, gender and region alone, so using
it to overwrite the FRS half's dividends gave Universal Credit claimants
dividends unrelated to anything they reported.
"""
from policyengine_uk_data.datasets.imputations import income as income_module
from policyengine_uk_data.datasets import disability_benefits
from policyengine_uk_data.datasets.imputations import frs_only

imputed_halves = []

def fake_impute_over_incomes(dataset, _model, output_variables):
imputed_halves.append(
(
bool(dataset.household["household_is_spi_synthetic"].all()),
tuple(output_variables),
)
)
dataset = dataset.copy()
for column in output_variables:
dataset.person[column] = 123_456.0
return dataset

monkeypatch.setattr(income_module, "create_income_model", lambda: object())
monkeypatch.setattr(
income_module,
"subsample_dataset",
lambda dataset, _sample_size: dataset.copy(),
)
monkeypatch.setattr(income_module, "impute_over_incomes", fake_impute_over_incomes)
monkeypatch.setattr(
frs_only,
"impute_frs_only_variables",
lambda train_dataset, target_dataset: target_dataset,
)
monkeypatch.setattr(
disability_benefits,
"strip_internal_disability_reported_amounts",
lambda dataset: dataset,
)
monkeypatch.setattr(income_module, "stack_datasets", _stack_without_remapping)

dataset = _fake_dataset()
dataset.person["dividend_income"] = [150.0, 0.0]
result = income_module.impute_income(dataset)

# Only the SPI-donor copy is imputed, with the full income set.
assert imputed_halves == [(True, tuple(income_module.IMPUTATIONS))]
frs_half = result.person.iloc[:2]
assert frs_half["dividend_income"].tolist() == [150.0, 0.0]
assert frs_half["employment_income"].tolist() == [20_000.0, 80_000.0]
assert result.person.iloc[2:]["dividend_income"].eq(123_456.0).all()
Loading