diff --git a/changelog.d/543.fixed.md b/changelog.d/543.fixed.md new file mode 100644 index 00000000..3483b81b --- /dev/null +++ b/changelog.d/543.fixed.md @@ -0,0 +1 @@ +US household calculations without `county_fips` now measure SPM poverty nationally instead of raising `SPM_GEOGRAPHY_REQUIRED`, and report it as `provenance["spm_geography_source"] == "national_fallback"`. Households with `county_fips` keep their county's SPM estimation area, and a geography chosen with `spm=` is still used as given. diff --git a/docs/countries.md b/docs/countries.md index 4e2227a5..7493898b 100644 --- a/docs/countries.md +++ b/docs/countries.md @@ -39,8 +39,9 @@ Override in any output with `income_variable=`. US: Populace row scoping uses `state_fips` and `congressional_district_geoid`. `state_code` remains the human-readable state input for custom households. -US resource and poverty calculations also require observed `county_fips` or an -explicit national or SPM-area choice; see [Households](households.md#spm-geography-and-measurement-selection). +US SPM thresholds use a household's observed `county_fips` to find its SPM +estimation area; a household calculation without one is measured nationally and +says so in its provenance. See [Households](households.md#spm-geography-and-measurement-selection). UK: constituency code and local authority code on every household where available. diff --git a/docs/households.md b/docs/households.md index bac0225e..2370587c 100644 --- a/docs/households.md +++ b/docs/households.md @@ -27,7 +27,7 @@ result = pe.us.calculate_household( | `people` | List of person dicts. Keys are any person-level variable on the model. | | `tax_unit` | Tax-unit inputs (e.g. `filing_status`). | | `spm_unit` | SPM-unit inputs. | -| `household` | Household inputs, including `state_code` and observed five-digit `county_fips` for the default SPM geography selection. | +| `household` | Household inputs, including `state_code` and, for area-adjusted SPM measurement, the observed five-digit `county_fips`. | | `family` | Family-level inputs. | | `marital_unit` | Marital-unit inputs. | @@ -35,8 +35,11 @@ All adults default to one shared tax unit and household. For separate tax units ### SPM geography and measurement selection -US household results include SPM resources and poverty by default. Provide the -household's county FIPS, as above, or explicitly select national measurement: +US household results include SPM resources and poverty by default. With the +household's county FIPS, as above, SPM thresholds use that county's Census SPM +estimation area. A household that gives only its state is measured nationally, +with no geographic adjustment, and the result says so. Missing values (`None`, +`""`, NaN) and a `county` or `county_str` of `"UNKNOWN"` count as no county: ```python result = pe.us.calculate_household( @@ -44,15 +47,27 @@ result = pe.us.calculate_household( tax_unit={"filing_status": "SINGLE"}, household={"state_code": "CA"}, year=2026, - spm={"geography_kind": "national"}, ) +result.provenance["spm_config"]["geography_kind"] # "national" +result.provenance["spm_geography_source"] # "national_fallback" receipt = result.to_dict()["provenance"]["spm"] result.write("household-result.json") # Includes the JSON-compatible receipt. ``` +National measurement applies to everything that uses the SPM measurement: the +thresholds and poverty status, and, for a unit allocated housing assistance, the +capped SPM housing subsidy and the SPM resources built on it. + +`spm_geography_source` is `"national_fallback"` when national measurement +replaced the default county selection because the household named no county, +`"default"` when the bundle default applied as is, and `"selection"` when you +chose the geography with `spm`. The same national result, recorded as your +selection, comes from `spm={"geography_kind": "national"}`. + County FIPS assigns a household to the selected year's Census SPM estimation area; it does not select a separately estimated county rent factor. National -measurement is a conscious analytical choice and is recorded in provenance. +measurement, whether chosen or used because no county was given, is recorded in +provenance. A fixed SPM area can instead be selected with `spm={"geography_kind": "metro", "geography_id": area_id}`, using an area ID available in the pinned artifact for the requested year. @@ -64,7 +79,7 @@ set of keys is: |---|---| | `forecast_content_sha256` | Optional assertion of the bundle's independently pinned artifact content hash. A different hash is rejected. | | `scenario` | Scenario within that artifact: `ce_trend` by default or the `zero_real` sensitivity. | -| `geography_kind` | `county` by default; `national` or `metro` require an explicit choice. | +| `geography_kind` | `county` by default. If you leave it out and the household has no `county_fips`, the calculation falls back to `national`. `national` or `metro` can be chosen explicitly. | | `geography_id` | Required only for a fixed `metro` SPM area. | | `county_vintage` | County assignment vintage, `"2020"` by default. | | `as_of` | Optional information-date cutoff accepted by the pinned artifact. | @@ -75,14 +90,21 @@ national inputs from forecast components and research geography; an unsupported year fails instead of being extrapolated by the wrapper. Scenario forecasts are conditional research estimates, not agency forecasts or uncertainty bounds. -State alone is insufficient for SPM and raises `SPM_GEOGRAPHY_REQUIRED`. -Unknown counties or selected areas raise `SPM_GEOGRAPHY_UNAVAILABLE`; a measured -unit with no classified adult raises `SPM_COMPOSITION_REQUIRED`. These are -calculator `SPMInputError` exceptions with `code` and `to_dict()` attributes. -Geography is checked when an SPM-dependent formula runs, and only for the units -whose result depends on the measurement. SPM measurement itself — thresholds, -the geographic factor and SPM poverty — always requires this choice. The -ordinary resource outputs — household net income, benefits, income decile and +A state alone does not identify a Census SPM estimation area, so a household +without `county_fips` is measured nationally unless you choose a geography. A +geography you choose is used as given: `spm={"geography_kind": "county"}` for a +household without `county_fips` raises `SPM_GEOGRAPHY_REQUIRED`. So does a +household that names its county only as `county` or `county_str`, because county +measurement reads `county_fips`; it keeps the county selection rather than being +measured nationally. Unknown counties or selected areas raise +`SPM_GEOGRAPHY_UNAVAILABLE`; a measured unit with no classified adult raises +`SPM_COMPOSITION_REQUIRED`. These are calculator `SPMInputError` exceptions with +`code` and `to_dict()` attributes. + +Under a county selection, geography is checked when an SPM-dependent formula +runs, and only for the units whose result depends on the measurement. SPM +measurement itself — thresholds, the geographic factor and SPM poverty — always +needs the county. The ordinary resource outputs — household net income, benefits, income decile and equivalized net income, and each person's marginal tax rate — take the household's housing assistance amount rather than the capped SPM subsidy, so they succeed on state alone and record an empty measurement receipt whether or @@ -93,9 +115,7 @@ without a geography the cap and everything downstream of it — `spm_unit_benefits`, `spm_unit_net_income` and in turn `spm_unit_oecd_equiv_net_income` and `spm_unit_income_decile` — raise `SPM_GEOGRAPHY_REQUIRED`, while a unit allocated none has a capped subsidy of -zero by construction and consults no measurement. A genuinely independent -tax-only country-model calculation can use state alone; the wrapper's default -outputs include SPM poverty and therefore always require the choice. +zero by construction and consults no measurement. Supply observed inputs such as age, tenure, county and the source-backed `is_spm_independent_minor_role`. Computed SPM thresholds, geographic factors, diff --git a/examples/household_impact_example.py b/examples/household_impact_example.py index 0133dce9..583fc2b9 100644 --- a/examples/household_impact_example.py +++ b/examples/household_impact_example.py @@ -73,9 +73,10 @@ def us_example() -> None: print(f" Income tax: ${single.tax_unit.income_tax:,.0f}") print(f" Payroll tax: ${single.tax_unit.employee_payroll_tax:,.0f}") - # Married couple with two kids, Texas, lower income. Explicit national - # measurement is recorded in the result; state alone does not select an - # SPM area. Use an observed county_fips for local SPM measurement instead. + # Married couple with two kids, Texas, lower income. A state alone does not + # select an SPM area, so this household would be measured nationally even + # without the spm argument; choosing it explicitly records it as a + # selection. Use an observed county_fips for local SPM measurement instead. family = pe.us.calculate_household( people=[ {"age": 35, "employment_income": 40_000}, diff --git a/src/policyengine/core/spm.py b/src/policyengine/core/spm.py index 2d14b86e..f3ecdd4a 100644 --- a/src/policyengine/core/spm.py +++ b/src/policyengine/core/spm.py @@ -21,12 +21,15 @@ def _selection_schema(schema: dict[str, Any]) -> None: class SPMSelection(BaseModel): - """Select from the bundle's pinned artifact; national geography is explicit. + """Select from the bundle's pinned artifact. County mode reads the household's observed ``county_fips``. A state alone - does not identify an SPM area. These settings contain no provider or path. - Serialization preserves omitted options so they still inherit bundle defaults - after a round trip. A resolved selection explicitly contains all six fields. + does not identify an SPM area, so national measurement is either selected + explicitly or, for a household calculation that chose no geography and + names no county, a fallback recorded as ``spm_geography_source``. These + settings contain no provider or path. Serialization preserves omitted + options so they still inherit bundle defaults after a round trip. A resolved + selection explicitly contains all six fields. """ model_config = ConfigDict( diff --git a/src/policyengine/tax_benefit_models/us/household.py b/src/policyengine/tax_benefit_models/us/household.py index 17f828a4..05371d08 100644 --- a/src/policyengine/tax_benefit_models/us/household.py +++ b/src/policyengine/tax_benefit_models/us/household.py @@ -37,6 +37,8 @@ from __future__ import annotations +import math +import numbers from collections.abc import Mapping from typing import Any, Optional @@ -52,10 +54,61 @@ from policyengine.utils.household_validation import validate_household_input from .model import us_latest -from .spm import SPMSelection, calculation_provenance, resolve_spm_selection +from .spm import ( + SPMSelection, + calculation_provenance, + resolve_household_spm_selection, +) _GROUP_ENTITIES = ("marital_unit", "family", "spm_unit", "tax_unit", "household") +# Household inputs that name a county. SPM county measurement reads only +# ``county_fips``, but a household that names its county another way still +# asked for a county measurement, so it keeps the county selection and gets the +# error that asks for ``county_fips`` instead of a national result. +_COUNTY_INPUTS = ("county_fips", "county", "county_str") + + +def _is_absent(name: str, value: Any) -> bool: + """Whether a county input's value means that no county was given. + + Missing values (``None``, empty text, NaN or the text ``"nan"``, + ``pd.NA``) are absent, as the country model's county check also treats + them, and so is ``"UNKNOWN"``, the ``county`` enum's default, for + ``county`` and ``county_str`` only. Anything else, including a malformed + code such as ``6037``, names a county and so reaches the county + selection's typed error. This decides only the SPM selection: the country + model itself rejects some of these as inputs (``pd.NA`` and bytes fail to + serialize whatever the geography). + """ + if value is None: + return True + if isinstance(value, bytes): + value = value.decode(errors="replace") + if isinstance(value, str): + if value == "" or value.lower() == "nan": + return True + return name != "county_fips" and value == "UNKNOWN" + if isinstance(value, numbers.Number) and not isinstance(value, bool): + try: + return math.isnan(value) + except TypeError: + return False + import pandas as pd + + return value is pd.NA + + +def _names_county( + household: Mapping[str, Any], axes: Optional[list[list[dict[str, Any]]]] +) -> bool: + if any( + name in household and not _is_absent(name, household[name]) + for name in _COUNTY_INPUTS + ): + return True + return any(axis["name"] in _COUNTY_INPUTS for group in axes or [] for axis in group) + def _raise_unexpected_kwargs(unexpected: Mapping[str, Any]) -> None: from difflib import get_close_matches @@ -188,24 +241,28 @@ def calculate_household( values default to ``year``. When axes are present, result values are lists ordered by the axis grid instead of scalars. spm: SPMSelection or mapping selecting a scenario and geography from - the bundle's independently pinned artifact. Geography is demanded - only by the results that actually use the measurement. SPM - measurement itself — thresholds, the geographic factor and SPM - poverty — always requires household county_fips or an explicit - metro/national selection, and so do the default outputs, which - include SPM poverty. The ordinary resource outputs take the - household's housing assistance amount rather than the capped SPM - subsidy, so they compute on state alone whether or not the unit is - allocated assistance. The country's cap - (``spm_unit_capped_housing_subsidy``) is what consults the - canonical housing portion, and only for units allocated - assistance, so for an assisted unit without a geography the cap - and everything downstream of it — ``spm_unit_benefits``, - ``spm_unit_net_income`` and in turn - ``spm_unit_oecd_equiv_net_income`` and - ``spm_unit_income_decile`` — raise ``SPM_GEOGRAPHY_REQUIRED``; - a unit allocated none has a capped subsidy of zero by - construction. + the bundle's independently pinned artifact. When it chooses no + ``geography_kind``, a household with ``county_fips`` is measured + in its county's Census SPM estimation area, and a household that + names no county (no county input, only missing values, or + ``county``/``county_str`` of ``"UNKNOWN"``) is measured + nationally, with no geographic + adjustment, in its thresholds and in the capped SPM housing + subsidy; ``provenance["spm_geography_source"]`` is then + ``"national_fallback"`` (otherwise ``"default"``, or + ``"selection"`` when you chose the geography). A household that + names its county only as ``county`` or ``county_str`` keeps county + measurement and raises ``SPM_GEOGRAPHY_REQUIRED``, asking for + ``county_fips``. A ``geography_kind`` you choose is used as given, + so an explicit county selection without ``county_fips`` also + raises ``SPM_GEOGRAPHY_REQUIRED``. Under that selection only the + results that use the measurement need the county: SPM thresholds + and poverty, and, for a unit allocated housing assistance, the + capped SPM subsidy (``spm_unit_capped_housing_subsidy``) and the + SPM resources built on it (``spm_unit_benefits``, + ``spm_unit_net_income``, ``spm_unit_oecd_equiv_net_income``, + ``spm_unit_income_decile``). Ordinary resource outputs use the + actual housing assistance amount and never need one. Formula-owned SPM amounts and measurement counts cannot be supplied as inputs/axes. @@ -213,6 +270,8 @@ def calculate_household( :class:`HouseholdResult` with dot-accessible per-entity variables. Singleton entities (``tax_unit``, ``household``, ...) return :class:`EntityResult`; ``person`` returns a list of them. + ``provenance`` holds the resolved ``spm_config``, the + ``spm_geography_source`` and the SPM calculation receipt (``spm``). Raises: ValueError: if any input dict uses an unknown variable name, @@ -272,7 +331,10 @@ def calculate_household( } ) axes_active = normalized_axes is not None - spm_config = resolve_spm_selection(spm) + spm_config, spm_geography_source = resolve_household_spm_selection( + spm, + household_names_county=_names_county(entities["household"], normalized_axes), + ) simulation = Simulation( situation=_build_situation( @@ -330,6 +392,7 @@ def calculate_household( ) result["provenance"] = { "spm_config": dict(simulation.spm_config), + "spm_geography_source": spm_geography_source, "spm": calculation_provenance(simulation), } return result diff --git a/src/policyengine/tax_benefit_models/us/spm.py b/src/policyengine/tax_benefit_models/us/spm.py index 568fa58f..51b6ca50 100644 --- a/src/policyengine/tax_benefit_models/us/spm.py +++ b/src/policyengine/tax_benefit_models/us/spm.py @@ -5,10 +5,18 @@ __all__ = [ "SPMSelection", "SPMProvenance", + "SPM_GEOGRAPHY_SOURCES", "resolve_spm_selection", + "resolve_household_spm_selection", "calculation_provenance", ] +# How a household calculation's SPM geography was chosen, reported as +# ``provenance["spm_geography_source"]``: the caller's own ``geography_kind``, +# the bundle default as given, or national measurement in place of the default +# county selection because the household names no county. +SPM_GEOGRAPHY_SOURCES = ("selection", "default", "national_fallback") + def resolve_spm_selection(selection=None) -> dict: """Resolve public options against an independently pinned bundle artifact.""" @@ -43,6 +51,38 @@ def resolve_spm_selection(selection=None) -> dict: return SPMSelection.model_validate(values).model_dump() +def resolve_household_spm_selection( + selection=None, *, household_names_county: bool +) -> tuple[dict, str]: + """Resolve one household calculation's SPM selection and how it was chosen. + + A ``geography_kind`` the caller chose is honoured as given, so an explicit + county selection for a household with no county still raises + ``SPM_GEOGRAPHY_REQUIRED``. Otherwise the bundle default applies, except + that the default county selection becomes national measurement when the + household names no county. A state alone does not identify a Census SPM + estimation area, and national measurement is the one geography that needs + no area. The substitution is reported in the returned source, never made + silently. + + Every other setting, such as ``scenario``, is kept. Population simulations + do not use this function: their data must supply observed counties. + + Returns the resolved configuration and one of ``SPM_GEOGRAPHY_SOURCES``. + """ + chosen = SPMSelection.model_validate({} if selection is None else selection) + # Resolve before deciding: this also rejects a selection that asserts a + # different artifact hash, whichever geography is finally used. + config = resolve_spm_selection(chosen) + if "geography_kind" in chosen.model_fields_set: + return config, "selection" + if config["geography_kind"] != "county" or household_names_county: + return config, "default" + # ``model_dump`` keeps only the settings the caller chose. + national = {**chosen.model_dump(), "geography_kind": "national"} + return resolve_spm_selection(national), "national_fallback" + + def calculation_provenance(simulation) -> dict: return SPMProvenance.model_validate(simulation.spm_provenance()).model_dump( mode="json" diff --git a/tests/test_spm_household.py b/tests/test_spm_household.py index 5e63ca8e..410b01ad 100644 --- a/tests/test_spm_household.py +++ b/tests/test_spm_household.py @@ -128,7 +128,22 @@ def test_default_county_and_explicit_metro_resolve_the_same_area(): @pytest.mark.parametrize( "location,settings,code", [ - ({"state_code": "CA"}, {}, "SPM_GEOGRAPHY_REQUIRED"), + # A household that names no county falls back to national measurement + # only when no geography is chosen; an explicit county selection is + # used as given. + ({"state_code": "CA"}, {"geography_kind": "county"}, "SPM_GEOGRAPHY_REQUIRED"), + # A county named without its FIPS code keeps county measurement, and + # the error asks for county_fips rather than measuring nationally. + ( + {"state_code": "CA", "county": "LOS_ANGELES_COUNTY_CA"}, + {}, + "SPM_GEOGRAPHY_REQUIRED", + ), + ( + {"state_code": "CA", "county_str": "LOS_ANGELES_COUNTY_CA"}, + {}, + "SPM_GEOGRAPHY_REQUIRED", + ), ( {"state_code": "CA", "county_fips": "99999"}, {}, @@ -148,6 +163,8 @@ def test_default_resource_outputs_require_a_real_geography(location, settings, c assert caught.value.to_dict()["code"] == code +COUNTY = {"geography_kind": "county"} + RESOURCE_VARIABLES = ( "household_net_income", "household_benefits", @@ -162,10 +179,12 @@ def test_state_only_graphs_require_geography_only_where_measurement_is_used( ): """Geography is demanded by exactly the results that use the measurement. - Ordinary benefits and income use the actual housing award independently of - SPM geography. The country's cap consults the canonical housing portion only - for units allocated assistance, so assisted SPM resources require geography. - The threshold and SPM poverty status require geography either way. + Under an explicit county selection, ordinary benefits and income use the + actual housing award independently of SPM geography. The country's cap + consults the canonical housing portion only for units allocated assistance, + so assisted SPM resources require geography. The threshold and SPM poverty + status require geography either way. Without a chosen geography the same + households measure nationally instead (tests/test_spm_household_geography.py). """ from policyengine.tax_benefit_models.us.model import us_latest @@ -215,7 +234,8 @@ def test_state_only_graphs_require_geography_only_where_measurement_is_used( assert math.isfinite(entity[variable]) assert computed.to_dict()["provenance"]["spm"]["years"] == {} - # Only the assisted SPM resource graph needs geography for its housing cap: + # Under an explicit county selection, only the assisted SPM resource graph + # needs geography for its housing cap: # the five variables the shipped contract names, and nothing household-level. for variable in ( "spm_unit_capped_housing_subsidy", @@ -225,14 +245,18 @@ def test_state_only_graphs_require_geography_only_where_measurement_is_used( "spm_unit_income_decile", ): with pytest.raises(SPMInputError) as caught: - pe.us.calculate_household(**assisted, extra_variables=[variable]) + pe.us.calculate_household( + **assisted, spm=COUNTY, extra_variables=[variable] + ) assert caught.value.code == "SPM_GEOGRAPHY_REQUIRED" - # SPM measurement always requires geography, assisted or not. + # County SPM measurement always requires the county, assisted or not. for situation in (inputs, assisted): for variable in ("spm_unit_spm_threshold", "spm_unit_is_in_spm_poverty"): with pytest.raises(SPMInputError) as caught: - pe.us.calculate_household(**situation, extra_variables=[variable]) + pe.us.calculate_household( + **situation, spm=COUNTY, extra_variables=[variable] + ) assert caught.value.code == "SPM_GEOGRAPHY_REQUIRED" # Preserve the unassisted national and county controls. diff --git a/tests/test_spm_household_geography.py b/tests/test_spm_household_geography.py new file mode 100644 index 00000000..f03508af --- /dev/null +++ b/tests/test_spm_household_geography.py @@ -0,0 +1,368 @@ +"""Default SPM geography for household calculations. + +With no ``geography_kind`` chosen, ``calculate_household`` measures a household +in its county's SPM estimation area when it has ``county_fips``, and nationally +when it names no county, reporting the substitution in provenance. The unit +tests check the selection logic against a stub bundle; the household tests run +the real pinned country model and calculator. + +Invariants checked here, for every household that names no county: + +- the default calculation never raises ``SPMInputError``; +- it equals the explicit national calculation, output for output, apart from + ``spm_geography_source``; +- its geographic adjustment is exactly 1 and its threshold equals the + unadjusted threshold; +- SPM poverty is exactly ``spm_unit_net_income < spm_unit_spm_threshold`` + (the country formula, rechecked on fallback results as a consistency check). + +A household with ``county_fips`` equals the explicit county calculation. +Missing values (``None``, ``""``, NaN) and ``"UNKNOWN"`` count as no county. +""" + +import math + +import pandas as pd +import pytest +from hypothesis import HealthCheck, given, settings +from hypothesis import strategies as st + +# The household cases run the real pinned country model and calculator, never a +# substitute; skip only if they are absent from the environment. +pytest.importorskip("policyengine_us") +pytest.importorskip("spm_calculator.policyengine_adapter") + +from policyengine_us.variables.household.demographic.geographic.state_code import ( # noqa: E402 + StateCode, +) +from spm_calculator.errors import SPMInputError # noqa: E402 + +import policyengine as pe # noqa: E402 +from policyengine.tax_benefit_models.us.household import _names_county # noqa: E402 +from policyengine.tax_benefit_models.us.spm import ( # noqa: E402 + SPM_GEOGRAPHY_SOURCES, + SPMSelection, + resolve_household_spm_selection, + resolve_spm_selection, +) + +ARTIFACT = "a" * 64 + + +@pytest.fixture +def county_bundle(monkeypatch): + value = { + "measurements": { + "spm": { + "forecast_content_sha256": ARTIFACT, + "scenario": "ce_trend", + "geography_kind": "county", + } + } + } + monkeypatch.setattr("policyengine.bundle.get_current_bundle", lambda: value) + return value + + +def test_household_without_county_falls_back_to_national(county_bundle): + config, source = resolve_household_spm_selection(None, household_names_county=False) + assert source == "national_fallback" + assert config == resolve_spm_selection({"geography_kind": "national"}) + assert config["geography_kind"] == "national" + assert config["forecast_content_sha256"] == ARTIFACT + + +def test_household_with_county_keeps_the_default(county_bundle): + config, source = resolve_household_spm_selection(None, household_names_county=True) + assert source == "default" + assert config == resolve_spm_selection() + assert config["geography_kind"] == "county" + + +@pytest.mark.parametrize("names_county", [False, True]) +@pytest.mark.parametrize( + "selection", + [ + {"geography_kind": "county"}, + {"geography_kind": "national"}, + {"geography_kind": "metro", "geography_id": "31080"}, + SPMSelection(geography_kind="county"), + ], +) +def test_chosen_geography_is_used_as_given(county_bundle, selection, names_county): + config, source = resolve_household_spm_selection( + selection, household_names_county=names_county + ) + assert source == "selection" + assert config == resolve_spm_selection(selection) + + +def test_fallback_keeps_the_other_chosen_settings(county_bundle): + selection = {"scenario": "zero_real", "as_of": "2026-09-09"} + config, source = resolve_household_spm_selection( + selection, household_names_county=False + ) + assert source == "national_fallback" + assert config == resolve_spm_selection({**selection, "geography_kind": "national"}) + assert config["scenario"] == "zero_real" + assert config["as_of"] == "2026-09-09" + + +@pytest.mark.parametrize( + "selection", + [ + SPMSelection(scenario="zero_real"), + {"forecast_content_sha256": ARTIFACT}, + {"forecast_content_sha256": None}, + {"county_vintage": "2020"}, + ], +) +def test_selections_without_a_geography_fall_back(county_bundle, selection): + config, source = resolve_household_spm_selection( + selection, household_names_county=False + ) + assert source == "national_fallback" + chosen = SPMSelection.model_validate(selection).model_dump() + expected = resolve_spm_selection({**chosen, "geography_kind": "national"}) + assert config == expected + assert config["forecast_content_sha256"] == ARTIFACT + + +def test_fallback_still_rejects_a_different_artifact(county_bundle): + with pytest.raises(ValueError, match="artifact hash"): + resolve_household_spm_selection( + {"forecast_content_sha256": "b" * 64}, household_names_county=False + ) + + +def test_fallback_is_only_for_the_county_default(county_bundle): + county_bundle["measurements"]["spm"]["geography_kind"] = "national" + config, source = resolve_household_spm_selection(None, household_names_county=False) + assert source == "default" + assert config == resolve_spm_selection() + + +def test_sources_are_the_documented_set(county_bundle): + seen = { + resolve_household_spm_selection(selection, household_names_county=names)[1] + for selection in (None, {"geography_kind": "national"}) + for names in (False, True) + } + assert seen == set(SPM_GEOGRAPHY_SOURCES) + + +@pytest.mark.parametrize( + "household,axes,expected", + [ + ({"state_code": "CA"}, None, False), + ({"state_code": "CA", "county_fips": None}, None, False), + ({"state_code": "CA", "county_fips": ""}, None, False), + ({"state_code": "CA", "county_fips": b""}, None, False), + ({"state_code": "CA", "county_fips": float("nan")}, None, False), + ({"state_code": "CA", "county_fips": "nan"}, None, False), + # "UNKNOWN" is the county enum's default, not a county FIPS code. + ({"state_code": "CA", "county_fips": "UNKNOWN"}, None, True), + ({"state_code": "CA", "county_fips": pd.NA}, None, False), + ({"state_code": "CA", "county": "UNKNOWN"}, None, False), + ({"state_code": "CA", "county_str": "UNKNOWN"}, None, False), + ({"state_code": "CA", "county": None, "county_str": ""}, None, False), + ({"state_code": "CA", "county_fips": "06037"}, None, True), + # Malformed codes still name a county, so they reach the typed error. + ({"state_code": "CA", "county_fips": 6037}, None, True), + ({"state_code": "CA", "county_fips": "6037"}, None, True), + ({"state_code": "CA", "county": "LOS_ANGELES_COUNTY_CA"}, None, True), + ({"state_code": "CA", "county_str": "LOS_ANGELES_COUNTY_CA"}, None, True), + ( + {"state_code": "CA"}, + [[{"name": "employment_income", "min": 0, "max": 1, "count": 2}]], + False, + ), + ( + {"state_code": "CA"}, + [[{"name": "county_fips", "min": 0, "max": 1, "count": 2}]], + True, + ), + ], +) +def test_names_county(household, axes, expected): + assert _names_county(household, axes) is expected + + +# Household calculations against the pinned country model and calculator. + +MEASUREMENT = [ + "spm_unit_spm_threshold", + "spm_unit_unadjusted_spm_threshold", + "spm_unit_geographic_adjustment", +] +STATE_CODES = [state.name for state in StateCode] +# policyengine-us 2.2.1 has no SNAP region for Puerto Rico, so every Puerto +# Rico household calculation fails before any SPM formula runs, whatever the +# SPM geography. +FAILS_OUTSIDE_SPM = {"PR": ValueError} + + +def household(state_code="CA", *, earnings=50_000, children=1, **location): + return { + "people": [ + {"age": 40, "employment_income": earnings}, + *({"age": 8} for _ in range(children)), + ], + "tax_unit": {"filing_status": "HEAD_OF_HOUSEHOLD" if children else "SINGLE"}, + "household": {"state_code": state_code, **location}, + "year": 2026, + } + + +def calculate(inputs, spm=None, *, tenure=None): + inputs = dict(inputs) + if tenure is not None: + inputs["spm_unit"] = {"spm_unit_tenure_type": tenure} + return pe.us.calculate_household( + **inputs, spm=spm, extra_variables=MEASUREMENT + ).to_dict() + + +def without_source(result): + provenance = dict(result["provenance"]) + provenance.pop("spm_geography_source") + return {**result, "provenance": provenance} + + +def check_national_fallback(result): + provenance = result["provenance"] + assert provenance["spm_geography_source"] == "national_fallback" + assert provenance["spm_config"]["geography_kind"] == "national" + assert provenance["spm"]["geography_kind"] == "national" + unit = result["spm_unit"] + assert unit["spm_unit_geographic_adjustment"] == 1.0 + assert unit["spm_unit_spm_threshold"] == unit["spm_unit_unadjusted_spm_threshold"] + assert unit["spm_unit_spm_threshold"] > 0 + assert math.isfinite(unit["spm_unit_net_income"]) + in_poverty = unit["spm_unit_net_income"] < unit["spm_unit_spm_threshold"] + assert bool(unit["spm_unit_is_in_spm_poverty"]) is in_poverty + + +def test_state_only_household_returns_a_national_result(): + """The regression: a state-only household used to raise SPMInputError.""" + result = calculate(household("CA")) + check_national_fallback(result) + assert result["household"]["household_net_income"] > 0 + assert math.isfinite(result["tax_unit"]["income_tax"]) + + +@pytest.mark.parametrize("state_code", ["CA", "MS", "NY"]) +def test_state_only_default_equals_explicit_national(state_code): + inputs = household(state_code) + default = calculate(inputs) + explicit = calculate(inputs, {"geography_kind": "national"}) + assert explicit["provenance"]["spm_geography_source"] == "selection" + assert without_source(default) == without_source(explicit) + + +@pytest.mark.parametrize( + "location", + [ + {"county_fips": ""}, + {"county_fips": None}, + {"county_fips": float("nan")}, + {"county_fips": "nan"}, + {"county_str": "UNKNOWN"}, + ], +) +def test_missing_county_values_are_no_county(location): + # Compare with explicit national for the same inputs: a county input can + # move non-SPM outputs slightly (a county-name input shifts one float32 + # Medicaid value in the last digit), which is not what this tests. + inputs = household("CA", **location) + result = calculate(inputs) + check_national_fallback(result) + explicit = calculate(inputs, {"geography_kind": "national"}) + assert without_source(result) == without_source(explicit) + + +def test_unknown_county_enum_is_no_county(): + inputs = household("CA", county="UNKNOWN") + result = calculate(inputs) + check_national_fallback(result) + explicit = calculate(inputs, {"geography_kind": "national"}) + assert without_source(result) == without_source(explicit) + + +def test_fallback_keeps_a_chosen_scenario(): + inputs = household("TX") + default = calculate(inputs, {"scenario": "zero_real"}) + check_national_fallback(default) + assert default["provenance"]["spm_config"]["scenario"] == "zero_real" + explicit = calculate( + inputs, {"geography_kind": "national", "scenario": "zero_real"} + ) + assert without_source(default) == without_source(explicit) + + +@pytest.mark.parametrize("county_fips", ["06037", "28001", "36061"]) +def test_county_household_default_equals_explicit_county(county_fips): + state_code = {"06": "CA", "28": "MS", "36": "NY"}[county_fips[:2]] + inputs = household(state_code, county_fips=county_fips) + default = calculate(inputs) + explicit = calculate(inputs, {"geography_kind": "county"}) + assert default["provenance"]["spm_geography_source"] == "default" + assert default["provenance"]["spm_config"]["geography_kind"] == "county" + assert without_source(default) == without_source(explicit) + assignment = default["provenance"]["spm"]["geographies"][0]["county_assignment"] + assert assignment is not None + + +def test_county_adjustment_differs_from_national_where_rents_do(): + """Los Angeles County's area adjustment is used, not the national one.""" + county = calculate(household("CA", county_fips="06037")) + national = calculate(household("CA")) + county_unit, national_unit = county["spm_unit"], national["spm_unit"] + assert county_unit["spm_unit_geographic_adjustment"] > 1 + assert ( + county_unit["spm_unit_spm_threshold"] > national_unit["spm_unit_spm_threshold"] + ) + assert ( + county_unit["spm_unit_unadjusted_spm_threshold"] + == national_unit["spm_unit_unadjusted_spm_threshold"] + ) + + +@pytest.mark.parametrize("state_code", STATE_CODES) +def test_every_state_code_without_county_computes(state_code): + inputs = household(state_code) + expected_error = FAILS_OUTSIDE_SPM.get(state_code) + if expected_error is not None: + # The same failure with or without the fallback, and never an SPM one. + for spm in (None, {"geography_kind": "national"}): + with pytest.raises(expected_error) as caught: + calculate(inputs, spm) + assert not isinstance(caught.value, SPMInputError) + return + check_national_fallback(calculate(inputs)) + + +@settings( + max_examples=20, + derandomize=True, + deadline=None, + suppress_health_check=[HealthCheck.too_slow], +) +@given( + state_code=st.sampled_from( + [code for code in STATE_CODES if code not in FAILS_OUTSIDE_SPM] + ), + earnings=st.integers(min_value=0, max_value=250_000), + children=st.integers(min_value=0, max_value=3), + tenure=st.sampled_from( + [None, "RENTER", "OWNER_WITH_MORTGAGE", "OWNER_WITHOUT_MORTGAGE"] + ), +) +def test_no_county_is_always_the_explicit_national_result( + state_code, earnings, children, tenure +): + inputs = household(state_code, earnings=earnings, children=children) + default = calculate(inputs, tenure=tenure) + check_national_fallback(default) + explicit = calculate(inputs, {"geography_kind": "national"}, tenure=tenure) + assert without_source(default) == without_source(explicit)