Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/frs-empstati-other-inactive.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Map FRS EMPSTATI code 11 ("Other inactive") to `employment_status` OTHER_INACTIVE; it had fallen through to LONG_TERM_DISABLED, which also put other-inactive adults into the ESA health-condition and support-group proxies. An adult EMPSTATI code the mapping does not know now fails the build instead of defaulting.
63 changes: 46 additions & 17 deletions policyengine_uk_data/datasets/frs.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,22 @@
LEGACY_JOBSEEKER_MIN_AGE = 18
HOURS_WORKED_WEEKS_PER_YEAR = 52
ESA_MIN_AGE = 16
# Adult-table EMPSTATI ("Adult - Employment Status - ILO definition") value
# labels from the UKDS FRS 2024-25 data dictionary, mapped to PolicyEngine
# statuses. Children have no EMPSTATI and are CHILD.
FRS_EMPSTATI_EMPLOYMENT_STATUS = {
1: EmploymentStatus.FT_EMPLOYED.name, # Full-time employee
2: EmploymentStatus.PT_EMPLOYED.name, # Part-time employee
3: EmploymentStatus.FT_SELF_EMPLOYED.name, # Full-time self-employed
4: EmploymentStatus.PT_SELF_EMPLOYED.name, # Part-time self-employed
5: EmploymentStatus.UNEMPLOYED.name, # Unemployed
6: EmploymentStatus.RETIRED.name, # Retired
7: EmploymentStatus.STUDENT.name, # Student
8: EmploymentStatus.CARER.name, # Looking after family/home
9: EmploymentStatus.LONG_TERM_DISABLED.name, # Permanently sick/disabled
10: EmploymentStatus.SHORT_TERM_DISABLED.name, # Temporarily sick/injured
11: EmploymentStatus.OTHER_INACTIVE.name, # Other inactive
}
ESA_HEALTH_EMPLOYMENT_STATUSES = (
EmploymentStatus.LONG_TERM_DISABLED.name,
EmploymentStatus.SHORT_TERM_DISABLED.name,
Expand Down Expand Up @@ -112,6 +128,33 @@ def require_variable(name: str, description: str) -> None:
)


def derive_employment_status_from_frs(empstati, is_adult_record) -> np.ndarray:
"""Map FRS EMPSTATI codes to ``employment_status``.

Rows from the child table are CHILD. Every adult must carry a code in
``FRS_EMPSTATI_EMPLOYMENT_STATUS``: a missing or unknown adult code fails
the build rather than falling back to a guessed status, as the old
fallback silently made code 11 (Other inactive) LONG_TERM_DISABLED.
"""

codes = pd.Series(np.asarray(empstati, dtype=float))
is_adult_record = np.asarray(is_adult_record, dtype=bool)
adult_status = codes.map(FRS_EMPSTATI_EMPLOYMENT_STATUS).to_numpy()
unknown = is_adult_record & pd.isna(adult_status)
if unknown.any():
# Build logs are public, and adults are not survey households, so
# name the codes and never a count.
bad = codes[unknown]
listed = [f"{code:g}" for code in sorted(bad.dropna().unique())]
listed += ["blank"] if bad.isna().any() else []
raise ValueError(
"FRS adults have EMPSTATI codes missing from "
f"FRS_EMPSTATI_EMPLOYMENT_STATUS: {', '.join(listed)}. Map them "
"from the release's data dictionary."
)
return np.where(is_adult_record, adult_status, EmploymentStatus.CHILD.name)


def derive_legacy_jobseeker_proxy(
age,
employment_status,
Expand Down Expand Up @@ -854,23 +897,9 @@ def determine_education_level(fted_val, typeed2_val, age_val):
"UPPER_SECONDARY"
)

# Add employment status
EMPLOYMENTS = [
"CHILD",
"FT_EMPLOYED",
"PT_EMPLOYED",
"FT_SELF_EMPLOYED",
"PT_SELF_EMPLOYED",
"UNEMPLOYED",
"RETIRED",
"STUDENT",
"CARER",
"LONG_TERM_DISABLED",
"SHORT_TERM_DISABLED",
]
pe_person["employment_status"] = categorical(
person.empstati, 1, range(12), EMPLOYMENTS
).fillna("LONG_TERM_DISABLED")
pe_person["employment_status"] = derive_employment_status_from_frs(
person.empstati, person.person_id.isin(frs["adult"].person_id)
)

# Add employer sector of the main job from FRS `mjobsect`
# (1 = private, 2 = public; missing/blank = not in paid work).
Expand Down
211 changes: 211 additions & 0 deletions policyengine_uk_data/tests/test_frs_employment_status.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,211 @@
"""FRS EMPSTATI codes to ``employment_status``, and what the ESA proxies read."""

import numpy as np
import pytest
from hypothesis import given, strategies as st
from policyengine_uk.variables.household.income.employment_status import (
EmploymentStatus,
)

from policyengine_uk_data.datasets.frs import (
ESA_HEALTH_EMPLOYMENT_STATUSES,
ESA_MIN_AGE,
FRS_EMPSTATI_EMPLOYMENT_STATUS,
derive_employment_status_from_frs,
derive_esa_health_condition_proxy,
derive_esa_support_group_proxy,
)

# EMPSTATI ("Adult - Employment Status - ILO definition") value labels in the
# UKDS FRS 2024-25 data dictionary (SN 9563, adult table), with the status each
# label means. Every adult in the 2020-21, 2022-23 and 2023-24 releases also
# has a code from 1 to 11.
DATA_DICTIONARY = {
1: ("Full-time Employee", "FT_EMPLOYED"),
2: ("Part-time Employee", "PT_EMPLOYED"),
3: ("Full-time Self-Employed", "FT_SELF_EMPLOYED"),
4: ("Part-time Self-Employed", "PT_SELF_EMPLOYED"),
5: ("Unemployed", "UNEMPLOYED"),
6: ("Retired", "RETIRED"),
7: ("Student", "STUDENT"),
8: ("Looking after family/home", "CARER"),
9: ("Permanently sick/disabled", "LONG_TERM_DISABLED"),
10: ("Temporarily sick/injured", "SHORT_TERM_DISABLED"),
11: ("Other Inactive", "OTHER_INACTIVE"),
}
ADULT_CODES = sorted(DATA_DICTIONARY)
HEALTH_CODES = (9, 10)

adult_rows = st.tuples(st.just(True), st.sampled_from(ADULT_CODES))
# Child-table rows have no EMPSTATI (0 once the person table fills blanks); the
# code must not matter for them.
child_rows = st.tuples(
st.just(False),
st.one_of(st.just(0), st.just(np.nan), st.integers(-9, 99)),
)
people = st.lists(st.one_of(adult_rows, child_rows), max_size=60)
unknown_adult_codes = st.one_of(
st.just(np.nan),
st.integers(-99, 0),
st.integers(12, 999),
st.floats(0.5, 11.5).filter(lambda x: not float(x).is_integer()),
)


def derive(rows):
is_adult = [adult for adult, _ in rows]
codes = [code for _, code in rows]
return derive_employment_status_from_frs(codes, is_adult)


def expected(adult, code):
return DATA_DICTIONARY[code][1] if adult else "CHILD"


@pytest.mark.parametrize("code", range(12))
def test_every_code_from_0_to_11(code):
assert derive_employment_status_from_frs([code], [False]).tolist() == ["CHILD"]
if code == 0:
with pytest.raises(ValueError, match="EMPSTATI"):
derive_employment_status_from_frs([code], [True])
else:
label, status = DATA_DICTIONARY[code]
assert derive_employment_status_from_frs([code], [True]).tolist() == [status], (
label
)


def test_other_inactive_is_not_long_term_disabled():
result = derive_employment_status_from_frs([9, 11], [True, True])
assert result.tolist() == ["LONG_TERM_DISABLED", "OTHER_INACTIVE"]


def test_code_table_is_the_data_dictionary():
assert FRS_EMPSTATI_EMPLOYMENT_STATUS == {
code: status for code, (_, status) in DATA_DICTIONARY.items()
}


def test_adult_codes_and_child_rows_cover_each_status_once():
statuses = [*FRS_EMPSTATI_EMPLOYMENT_STATUS.values(), "CHILD"]
assert len(set(statuses)) == len(statuses)
assert set(statuses) == set(EmploymentStatus.__members__)


@pytest.mark.parametrize("code", [0, -1, 12, 11.5, np.nan])
def test_unknown_adult_code_fails_the_build(code):
with pytest.raises(ValueError, match="EMPSTATI"):
derive_employment_status_from_frs([1, code], [True, True])


def listed_codes(message):
return message.split("FRS_EMPSTATI_EMPLOYMENT_STATUS: ")[1].split(". Map")[0]


def test_failure_message_lists_codes_in_numeric_order_with_blank_last():
# Formatted from the float codes, not Series.astype(str), whose NaN
# handling differs between pandas 2 and 3. A string sort would put 100
# before 11.5.
codes = [100, 12, np.nan, -1, 12, 11.5, 1]
with pytest.raises(ValueError) as error:
derive_employment_status_from_frs(codes, [True] * len(codes))
assert listed_codes(str(error.value)) == "-1, 11.5, 12, 100, blank"


@given(st.lists(unknown_adult_codes, min_size=1, max_size=20))
def test_failure_message_lists_each_unknown_code_once_in_order(bad_codes):
with pytest.raises(ValueError) as error:
derive_employment_status_from_frs(bad_codes, [True] * len(bad_codes))
listed = listed_codes(str(error.value)).split(", ")
blank = any(np.isnan(code) for code in bad_codes)
assert (listed[-1] == "blank") == blank
numbers = [float(code) for code in listed[: len(listed) - blank]]
assert numbers == sorted(numbers)
assert set(listed) - {"blank"} == {
f"{code:g}" for code in bad_codes if not np.isnan(code)
}


@given(st.integers(1, 30), st.integers(1, 30))
def test_failure_message_discloses_no_count(n_twelve, n_thirteen):
# Adults are not survey households, so no count of them is safe to print
# in a public build log: the message depends only on which codes occur.
codes = [12] * n_twelve + [13] * n_thirteen
with pytest.raises(ValueError) as error:
derive_employment_status_from_frs(codes, [True] * len(codes))
message = str(error.value)
assert listed_codes(message) == "12, 13"
assert not any(character.isdigit() for character in message.replace("12, 13", ""))


@given(people)
def test_each_row_maps_on_its_own(rows):
result = derive(rows)
assert result.tolist() == [expected(adult, code) for adult, code in rows]


@given(people, st.randoms(use_true_random=False))
def test_mapping_commutes_with_row_order(rows, rng):
order = list(range(len(rows)))
rng.shuffle(order)
statuses = derive(rows)
assert derive([rows[i] for i in order]).tolist() == [statuses[i] for i in order]


@given(people, unknown_adult_codes, st.integers(0, 60))
def test_any_unknown_adult_code_fails_the_build(rows, bad_code, position):
rows = list(rows)
rows.insert(min(position, len(rows)), (True, bad_code))
with pytest.raises(ValueError, match="EMPSTATI"):
derive(rows)


@given(
st.lists(
st.tuples(
st.integers(0, 100), # age
st.sampled_from(ADULT_CODES),
st.integers(60, 68), # State Pension age
st.integers(0, 3_000), # annual hours worked
st.booleans(), # EMPSTATI reported
),
min_size=1,
max_size=60,
)
)
def test_esa_proxies_read_only_the_sick_or_disabled_codes(rows):
age, codes, spa, hours, reported = map(np.array, zip(*rows))
status = derive_employment_status_from_frs(codes, np.ones(len(rows), bool))
health = derive_esa_health_condition_proxy(
age=age,
employment_status=status,
employment_status_reported=reported,
state_pension_age=spa,
)
support = derive_esa_support_group_proxy(
age=age,
employment_status=status,
hours_worked=hours,
esa_health_condition_proxy=health,
employment_status_reported=reported,
state_pension_age=spa,
)
working_age = (age >= ESA_MIN_AGE) & (age < spa)
assert (
health.tolist()
== (reported & working_age & np.isin(codes, HEALTH_CODES)).tolist()
)
assert not (support & ~health).any()
assert not support[codes != 9].any()
assert not health[codes == 11].any()


@pytest.mark.parametrize("dataset_name", ["frs", "enhanced_frs"])
def test_built_dataset_statuses(dataset_name, request):
person = request.getfixturevalue(dataset_name).person
status = person["employment_status"].astype(str)
assert set(status) <= set(EmploymentStatus.__members__)
assert (status == "OTHER_INACTIVE").any()
not_sick = ~status.isin(ESA_HEALTH_EMPLOYMENT_STATUSES)
assert not person.loc[not_sick, "esa_health_condition_proxy"].any()
assert not person.loc[not_sick, "esa_support_group_proxy"].any()
50 changes: 45 additions & 5 deletions policyengine_uk_data/tests/test_legacy_benefit_proxies.py
Original file line number Diff line number Diff line change
@@ -1,9 +1,11 @@
import numpy as np
import pandas as pd
import pytest
import policyengine_uk
import policyengine_uk_data.datasets.frs as frs_module

from policyengine_uk_data.datasets.frs import (
FRS_EMPSTATI_EMPLOYMENT_STATUS,
add_legacy_benefit_proxies,
attach_legacy_benefit_proxies_from_frs_person,
apply_legacy_benefit_proxies,
Expand Down Expand Up @@ -359,7 +361,7 @@ def calculate(self, variable, year=None):
if variable == "household_id":
return np.array([100])
if variable == "state_pension_age":
return pd.Series([66])
return pd.Series([66] * len(self.dataset.person))
if variable in (
"childcare_grant",
"parents_learning_allowance",
Expand All @@ -373,7 +375,7 @@ def calculate(self, variable, year=None):
raise KeyError(variable)


def test_create_frs_smoke_includes_legacy_proxy_columns(tmp_path, monkeypatch):
def create_single_adult_frs(tmp_path, monkeypatch, empstati=8, with_child=False):
original_read_csv = frs_module.pd.read_csv

def fake_read_csv(path, *args, **kwargs):
Expand Down Expand Up @@ -426,7 +428,7 @@ def fake_read_csv(path, *args, **kwargs):
"educqual": 0,
"eduma": 0,
"edumaamt": 0,
"empstati": 8,
"empstati": empstati,
"mjobsect": 0,
"sic": 0,
"fsbval": 0,
Expand Down Expand Up @@ -468,7 +470,13 @@ def fake_read_csv(path, *args, **kwargs):
}
]
)
child = pd.DataFrame(columns=adult.columns)
# The FRS child table has no EMPSTATI column.
child_columns = adult.columns.drop("empstati")
child = pd.DataFrame(columns=child_columns)
if with_child:
child = pd.DataFrame(
[{**dict.fromkeys(child_columns, 0), "sernum": 100, "benunit": 1}]
).assign(person=2, age=5, uperson=2)
benunit = pd.DataFrame([{"sernum": 100, "benunit": 1, "famtypb2": 1}])
househol = pd.DataFrame(
[
Expand Down Expand Up @@ -536,7 +544,11 @@ def fake_read_csv(path, *args, **kwargs):
for name, table in raw_tables.items():
table.to_csv(tmp_path / f"{name}.tab", sep="\t", index=False)

dataset = create_frs(tmp_path, 2025)
return create_frs(tmp_path, 2025)


def test_create_frs_smoke_includes_legacy_proxy_columns(tmp_path, monkeypatch):
dataset = create_single_adult_frs(tmp_path, monkeypatch)

assert {
"legacy_jobseeker_proxy",
Expand All @@ -561,3 +573,31 @@ def fake_read_csv(path, *args, **kwargs):
].iloc[0]
assert dataset.person["education_grants"].iloc[0] == 100
assert dataset.person["disabled_students_allowance_eligible_expenses"].iloc[0] == 0


@pytest.mark.parametrize("empstati", sorted(FRS_EMPSTATI_EMPLOYMENT_STATUS))
def test_create_frs_maps_every_empstati_code(tmp_path, monkeypatch, empstati):
person = create_single_adult_frs(tmp_path, monkeypatch, empstati).person

status = person["employment_status"].iloc[0]
assert status == FRS_EMPSTATI_EMPLOYMENT_STATUS[empstati]
# A working-age adult reporting no hours: only the sick/disabled codes
# (9 permanently, 10 temporarily) are ESA health states.
assert person["esa_health_condition_proxy"].iloc[0] == (empstati in (9, 10))
assert person["esa_support_group_proxy"].iloc[0] == (empstati == 9)


def test_create_frs_child_rows_are_child(tmp_path, monkeypatch):
person = create_single_adult_frs(
tmp_path, monkeypatch, empstati=11, with_child=True
).person.set_index("person_id")

assert person["employment_status"].to_dict() == {
100_001: "OTHER_INACTIVE",
100_002: "CHILD",
}


def test_create_frs_rejects_unknown_adult_empstati(tmp_path, monkeypatch):
with pytest.raises(ValueError, match="EMPSTATI"):
create_single_adult_frs(tmp_path, monkeypatch, empstati=12)
Loading
Loading