Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/frs-pc-reported-capital.changed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Carry each benefit unit's FRS total capital (TOTCAPB4, DWP's current benefit-unit savings and investments measure; TOTCAPB3 for earlier survey years) into policyengine-uk's `pension_credit_reported_capital`, so Pension Credit's capital test uses the survey's own benefit-unit capital instead of imputed household wealth.
37 changes: 37 additions & 0 deletions policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -570,6 +570,36 @@ def validate_frs_survey_year(raw_frs_folder, year: int) -> None:
)


def derive_pension_credit_reported_capital(benunit: pd.DataFrame) -> np.ndarray:
"""Each benefit unit's capital as the FRS records it, for Pension Credit.

Uses ``TOTCAPB4``, DWP's derived benefit-unit total of the adults' savings
and investments, which its below-average-resources statistics use in place
of ``TOTCAPB3`` since it became available in 2019/20; ``TOTCAPB3`` is the
fallback for earlier survey years. Pension Credit counts the claimant's
capital and, under the State Pension Credit Act 2002 s. 5, the partner's,
and this is a benefit-unit measure. It is an approximation of Pension
Credit capital, not the assessed figure: it covers financial assets only
(second homes and land, which Pension Credit also counts, are not in it),
and no Schedule V disregard or reg. 19 valuation is applied to it. The
household wealth imputation instead draws a household's wealth from Wealth
and Assets Survey households with similar income, composition, tenure and
region, with no information on means-tested receipt, and policyengine-uk
spreads it over the household's pension-age adults.

A missing or negative value gives -1, so policyengine-uk falls back to the
household proxy.
"""
capital = pd.Series(np.nan, index=benunit.index, dtype=float)
for column in ("totcapb3", "totcapb4"): # later columns take precedence
if column in benunit.columns:
values = pd.to_numeric(benunit[column], errors="coerce")
valid = np.isfinite(values) & (values >= 0)
capital = capital.where(~valid, values)
values = capital.to_numpy(dtype=float)
return np.where(np.isfinite(values) & (values >= 0), values, -1.0)


def create_frs(
raw_frs_folder: str,
year: int,
Expand Down Expand Up @@ -1627,6 +1657,13 @@ def _reported_benunit_mask(person_column: str) -> np.ndarray:

pe_benunit["is_married"] = benunit.famtypb2.isin([5, 7])

# Pension Credit capital as the FRS records it for the benefit unit, in
# place of the household wealth proxy (policyengine-uk
# `pension_credit_reported_capital`).
pe_benunit["pension_credit_reported_capital"] = (
derive_pension_credit_reported_capital(benunit)
)

# Assign property_purchased to a share of households matching the UK
# housing transaction rate, so only genuine purchasers are charged SDLT.
#
Expand Down
14 changes: 14 additions & 0 deletions policyengine_uk_data/datasets/imputations/income.py
Original file line number Diff line number Diff line change
Expand Up @@ -234,6 +234,19 @@ def impute_over_incomes(
return dataset


def clear_frs_reported_capital(dataset: UKSingleYearDataset) -> UKSingleYearDataset:
"""Set ``pension_credit_reported_capital`` to -1 (none recorded).

Used on the SPI-synthetic copy. The FRS benefit-unit capital belongs to the
FRS donor, whose incomes the SPI imputation replaces; keeping it would
assess an SPI-income unit on the donor's capital. With -1, policyengine-uk
uses the household capital proxy for these rows.
"""
if "pension_credit_reported_capital" in dataset.benunit.columns:
dataset.benunit["pension_credit_reported_capital"] = -1.0
return dataset


def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset:
"""
Impute detailed income components using trained model.
Expand Down Expand Up @@ -262,6 +275,7 @@ def impute_income(dataset: UKSingleYearDataset) -> UKSingleYearDataset:
zero_weight_copy = dataset.copy()
zero_weight_copy.household.household_weight = 0
zero_weight_copy.household["household_is_spi_synthetic"] = True
zero_weight_copy = clear_frs_reported_capital(zero_weight_copy)
zero_weight_copy = subsample_dataset(zero_weight_copy, 10_000)

model = create_income_model()
Expand Down
1 change: 1 addition & 0 deletions policyengine_uk_data/storage/uprating_factors.csv
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,7 @@ other_investment_income,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.
other_residential_property_value,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
owned_land,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
pension_credit_reported,1.0,1.04,1.144,1.209,1.237,1.277,1.301,1.327,1.353,1.38,1.38,1.38,1.38,1.38,1.38
pension_credit_reported_capital,1.0,1.0,1.102,1.161,1.204,1.256,1.293,1.335,1.376,1.417,1.462,1.508,1.555,1.604,1.655
pension_income,1.0,1.0,1.092,1.147,1.19,1.223,1.258,1.297,1.34,1.384,1.384,1.384,1.384,1.384,1.384
personal_pension_contributions,1.0,1.059,1.127,1.205,1.261,1.308,1.337,1.365,1.396,1.431,1.431,1.431,1.431,1.431,1.431
petrol_spending,1.0,1.104,1.635,1.796,1.531,1.483,1.452,1.406,1.363,1.305,1.237,1.237,1.237,1.237,1.237
Expand Down
1 change: 1 addition & 0 deletions policyengine_uk_data/storage/uprating_growth_factors.csv
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,7 @@ other_investment_income,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0
other_residential_property_value,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
owned_land,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
pension_credit_reported,0,0.04,0.1,0.057,0.023,0.032,0.019,0.02,0.02,0.02,0.0,0.0,0.0,0.0,0.0
pension_credit_reported_capital,0,0.0,0.102,0.054,0.037,0.043,0.029,0.032,0.031,0.03,0.032,0.031,0.031,0.032,0.032
pension_income,0,0.0,0.092,0.05,0.037,0.028,0.029,0.031,0.033,0.033,0.0,0.0,0.0,0.0,0.0
personal_pension_contributions,0,0.059,0.064,0.069,0.046,0.037,0.022,0.021,0.023,0.025,0.0,0.0,0.0,0.0,0.0
petrol_spending,0,0.104,0.481,0.099,-0.147,-0.031,-0.021,-0.032,-0.031,-0.043,-0.052,0.0,0.0,0.0,0.0
Expand Down
106 changes: 106 additions & 0 deletions policyengine_uk_data/tests/test_pension_credit_reported_capital.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
"""`pension_credit_reported_capital` from the FRS benefit-unit capital
measure (TOTCAPB4, falling back to TOTCAPB3).

Invariants:
1. A finite, non-negative TOTCAPB4 is carried over unchanged; where TOTCAPB4
is absent or invalid, a valid TOTCAPB3 is used.
2. A missing, non-numeric or negative value in both gives -1
(policyengine-uk's "none recorded" sentinel), so the household proxy
applies.
3. Without either column every benefit unit gets -1.
4. The output is always -1 or a non-negative number, one per benefit unit.
"""

import numpy as np
import pandas as pd

from policyengine_uk_data.datasets.frs import derive_pension_credit_reported_capital


def test_values_carry_over_and_invalid_values_fall_back():
benunit = pd.DataFrame(
{"totcapb3": [0.0, 300.0, 2_900.0, 1_250_000.0, np.nan, -5.0, "x"]}
)
result = derive_pension_credit_reported_capital(benunit)
assert result.tolist() == [0.0, 300.0, 2_900.0, 1_250_000.0, -1.0, -1.0, -1.0]


def test_totcapb4_takes_precedence_with_totcapb3_fallback():
benunit = pd.DataFrame(
{
"totcapb3": [100.0, 200.0, 300.0, np.nan, -1.0],
"totcapb4": [150.0, np.nan, -5.0, 400.0, np.nan],
}
)
result = derive_pension_credit_reported_capital(benunit)
assert result.tolist() == [150.0, 200.0, 300.0, 400.0, -1.0]


def test_missing_column_gives_sentinel():
benunit = pd.DataFrame({"benunit_id": [101, 102, 201]})
assert derive_pension_credit_reported_capital(benunit).tolist() == [-1.0] * 3


def test_output_is_sentinel_or_non_negative_for_random_inputs():
rng = np.random.default_rng(1_792)
for _ in range(50):
n = int(rng.integers(1, 200))
values = rng.normal(5_000, 20_000, n)
values[rng.random(n) < 0.1] = np.nan
result = derive_pension_credit_reported_capital(
pd.DataFrame({"totcapb4": values})
)
assert len(result) == n
assert np.all((result == -1) | (result >= 0))
keep = np.isfinite(values) & (values >= 0)
np.testing.assert_array_equal(result[keep], values[keep])
assert np.all(result[~keep] == -1)


def test_uprating_rows_match_the_model():
"""The build uprates the column with these rows (calibration materialises
the calibration year from them), and policyengine-uk projects the saved
dataset with the variable's own uprating index at runtime. The rows must
equal what ``create_policyengine_uprating_factors_table`` derives from the
locked policyengine-uk, or a unit near the 10,000 pound deemed-income
disregard can be assessed differently in calibration and at runtime."""
from policyengine_uk.system import system

from policyengine_uk_data.storage import STORAGE_FOLDER
from policyengine_uk_data.utils.uprating import END_YEAR, START_YEAR

variable = system.variables["pension_credit_reported_capital"]
index = system.parameters.get_child(variable.uprating)
years = range(START_YEAR, END_YEAR + 1)
expected = {y: round(index(y) / index(START_YEAR), 3) for y in years}
factors = pd.read_csv(STORAGE_FOLDER / "uprating_factors.csv").set_index("Variable")
growth = pd.read_csv(STORAGE_FOLDER / "uprating_growth_factors.csv").set_index(
"Variable"
)
for y in years:
assert factors.loc["pension_credit_reported_capital", str(y)] == expected[y]
expected_growth = (
0 if y == START_YEAR else round(expected[y] / expected[y - 1] - 1, 3)
)
assert growth.loc["pension_credit_reported_capital", str(y)] == expected_growth


def test_spi_copy_records_no_capital():
"""SPI-synthetic copies carry SPI-imputed incomes, so the FRS donor's
capital is cleared to -1 (household proxy) on them."""
from types import SimpleNamespace

from policyengine_uk_data.datasets.imputations.income import (
clear_frs_reported_capital,
)

copy = SimpleNamespace(
benunit=pd.DataFrame({"pension_credit_reported_capital": [0.0, 300.0, -1.0]})
)
assert clear_frs_reported_capital(copy).benunit[
"pension_credit_reported_capital"
].tolist() == [-1.0, -1.0, -1.0]
without = SimpleNamespace(benunit=pd.DataFrame({"benunit_id": [1, 2]}))
assert "pension_credit_reported_capital" not in (
clear_frs_reported_capital(without).benunit.columns
)
10 changes: 9 additions & 1 deletion policyengine_uk_data/tests/test_policybench_transfer.py
Original file line number Diff line number Diff line change
Expand Up @@ -205,6 +205,9 @@ def test_policybench_transfer_family_structure_matches_person_membership(
person_benunit_ids = sim.calculate("person_benunit_id", map_to="person").values
is_adult = sim.calculate("is_adult", map_to="person").values
is_child = sim.calculate("is_child", map_to="person").values
is_claimant_or_partner = sim.calculate(
"is_claimant_or_partner", map_to="person"
).values
is_married = sim.calculate("is_married", map_to="benunit").values
family_type = sim.calculate("family_type", map_to="benunit").values

Expand All @@ -213,7 +216,12 @@ def test_policybench_transfer_family_structure_matches_person_membership(
adults = int(is_adult[member_mask].sum())
children = int(is_child[member_mask].sum())

assert bool(married) == (adults == 2)
# policyengine-uk (from 2.107, #1896) presumes a couple married when
# the dataset does not say, and a couple is a claimant and partner,
# not any two adults: a member under 20 and 16+ years younger than
# the claimant is presumed to be their child.
claimant_and_partner = int(is_claimant_or_partner[member_mask].sum())
assert bool(married) == (claimant_and_partner == 2)

if adults == 2 and children > 0:
expected = "COUPLE_WITH_CHILDREN"
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ dependencies = [
"policyengine",
"google-cloud-storage",
"google-auth",
"policyengine-uk>=2.93.0",
"policyengine-uk>=2.107.0",
"microcalibrate>=0.18.0",
"microimpute>=1.0.1",
"ruff>=0.9.0",
Expand Down
16 changes: 8 additions & 8 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading