From d4bf360fbd0dca8b837ed8f7db37d0ec44610945 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 25 Sep 2026 18:25:28 -0400 Subject: [PATCH 1/3] Map the stored WIC take-up draw onto takes_up_wic_if_eligible policyengine-us 2.x renamed the WIC take-up input from would_claim_wic to takes_up_wic_if_eligible. The certified US default release stores the draw under the old name only, so every load path skipped it and every WIC-eligible person took WIC up (PolicyEngine/microcosm#1026). Add a register of renamed stored inputs, LEGACY_INPUT_RENAMES in tax_benefit_models/us/legacy_inputs.py, applied by one function. A rename applies only when the data stores the old name, the engine lacks it but defines the new one, and the data does not already store the new one. It then sets the new input from the stored draw for every month of every dataset year, after checking that each stored table is in the simulation's order. Simulation.run(), managed_microsimulation() and create_datasets() apply it, including a reform's baseline branch. Record the renames applied in output metadata, release_bundle, saved US output files, run records, policyengine_bundle and created datasets. Saved US outputs without the record predate the fix, so load() refuses them and ensure() recomputes them. Fixes #530 Co-Authored-By: Claude Opus 5.5 --- .gitignore | 1 + changelog.d/530.fixed.md | 1 + docs/microsim.md | 37 ++ docs/run-records.md | 2 +- pyproject.toml | 1 + src/policyengine/core/run_record.py | 9 + src/policyengine/core/simulation.py | 10 +- .../common/model_version.py | 44 +- .../tax_benefit_models/us/datasets.py | 21 +- .../tax_benefit_models/us/legacy_inputs.py | 345 +++++++++++ .../tax_benefit_models/us/model.py | 46 +- tests/test_us_legacy_inputs.py | 578 ++++++++++++++++++ tests/test_us_legacy_inputs_integration.py | 528 ++++++++++++++++ tests/test_us_native_alignment.py | 3 +- uv.lock | 90 +++ 15 files changed, 1705 insertions(+), 11 deletions(-) create mode 100644 changelog.d/530.fixed.md create mode 100644 src/policyengine/tax_benefit_models/us/legacy_inputs.py create mode 100644 tests/test_us_legacy_inputs.py create mode 100644 tests/test_us_legacy_inputs_integration.py diff --git a/.gitignore b/.gitignore index 3c351eab..3cf0cd51 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,4 @@ _build/ .env **/.DS_Store build/ +.hypothesis/ diff --git a/changelog.d/530.fixed.md b/changelog.d/530.fixed.md new file mode 100644 index 00000000..d285face --- /dev/null +++ b/changelog.d/530.fixed.md @@ -0,0 +1 @@ +Map the stored US WIC take-up draw `would_claim_wic` onto `takes_up_wic_if_eligible` when loading US data, so policyengine-us 2.x no longer gives WIC to every WIC-eligible person. `Simulation.run()`, `managed_microsimulation` and `create_datasets` apply the mapping and record the renames applied (output dataset metadata, `release_bundle`, saved output files, run records, `policyengine_bundle` and dataset metadata). A stored table that is not in the simulation's person order is refused. The mapping turns itself off once the data stores `takes_up_wic_if_eligible` or the engine defines `would_claim_wic` again. Saved US outputs from earlier releases are no longer loaded, so `ensure()` recomputes them. Year files written by `ensure_datasets` or `create_datasets` before this fix store neither name; delete and regenerate them. diff --git a/docs/microsim.md b/docs/microsim.md index 91795062..109e0e95 100644 --- a/docs/microsim.md +++ b/docs/microsim.md @@ -324,6 +324,43 @@ mode. GCS dataset URIs are not supported. For managed simulations, `sim.policyengine_bundle` records the actual source package, repository type, revision, verified SHA-256, and local path. +## Renamed inputs in stored data + +A country engine sets a stored column as an input only when it defines a +variable of that name. When policyengine-us renames an input, data written +before the rename keeps the old name, and the engine would skip it. +`policyengine.tax_benefit_models.us.legacy_inputs.LEGACY_INPUT_RENAMES` lists +each such rename; today it holds one entry, `would_claim_wic` → +`takes_up_wic_if_eligible` (the WIC take-up draw, PolicyEngine/microcosm#1026). + +Every US load path applies it: `Simulation.run()`, `managed_microsimulation` +and `create_datasets` (and so `ensure_datasets` when it creates year files). +A rename applies when the data stores the old name, the engine does not define +the old name but does define the new one, and the data does not already store +the new name. The new input is then set from the stored values for every month +of every dataset year. The stored table must list the simulation's entity IDs +in the simulation's order, or loading fails rather than attach values to the +wrong people. The mapping turns itself off once the data stores the new name, +or if the engine defines the old name again. + +The renames applied are recorded as `{old: new}` (`{}` when none applied): + +- `Simulation.run()`: `simulation.output_dataset.metadata["legacy_input_renames"]`, + also shown as `simulation.release_bundle["legacy_input_renames"]`. A US + `save()` writes the record into the output file and `load()` restores it, + and a run record's `results.json` binds it; +- `managed_microsimulation`: `sim.policyengine_bundle["legacy_input_renames"]`; +- `create_datasets`: each returned dataset's `metadata["legacy_input_renames"]`. + +A US output file saved before this mapping existed has no record, and its +results were calculated without the mapped inputs. `load()` refuses it, and +`ensure()` runs the simulation again and saves the new output. + +Year files that `ensure_datasets` or `create_datasets` wrote before this +mapping existed store neither name, so the draw is lost from them, and +`ensure_datasets` reuses existing files as they are. Delete and regenerate +them. + ## Pinned model versions Every `policyengine` release pins specific country-model and country-data versions so results are reproducible. `pe.us.model` and `pe.uk.model` expose the pinned `TaxBenefitModelVersion`. diff --git a/docs/run-records.md b/docs/run-records.md index 590bc406..efb0a510 100644 --- a/docs/run-records.md +++ b/docs/run-records.md @@ -46,7 +46,7 @@ The directory contains: | `bundle.trace.tro.jsonld` | The certified bundle TRO (model + data pins) | | `reform.json` | The reform as parameter values with effective dates | | `input.json` | Input dataset hash, dynamics, scoping, extra variables | -| `results.json` | Output dataset hash and per-entity table summaries | +| `results.json` | Output dataset hash and per-entity table summaries; for US runs, the SPM receipt and the renamed stored inputs mapped onto live ones (see [Renamed inputs in stored data](microsim.md#renamed-inputs-in-stored-data)) | All payload files are written with the same canonical JSON used for hashing, so the record verifies offline exactly as written. diff --git a/pyproject.toml b/pyproject.toml index 2997850f..3cdde640 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -68,6 +68,7 @@ dev = [ "towncrier>=24.8.0", "mypy>=1.11.0", "pytest-cov>=5.0.0", + "hypothesis>=6.100.0", "policyengine-core==3.32.5", "policyengine-us==2.2.1", "policyengine-uk==2.90.2", diff --git a/src/policyengine/core/run_record.py b/src/policyengine/core/run_record.py index c0bf7956..f9f552aa 100644 --- a/src/policyengine/core/run_record.py +++ b/src/policyengine/core/run_record.py @@ -184,6 +184,15 @@ def build_simulation_run_record_payloads( input_payload["spm"] = simulation.spm_config results_payload["spm"] = simulation.spm_provenance() + # Stored inputs the country loader mapped onto renamed live inputs (for + # example the US WIC take-up draw) change the results, so the record + # binds them. See ``policyengine.tax_benefit_models.us.legacy_inputs``. + renames = (getattr(simulation.output_dataset, "metadata", None) or {}).get( + "legacy_input_renames" + ) + if renames is not None: + results_payload["legacy_input_renames"] = dict(renames) + return {"reform": reform, "input": input_payload, "results": results_payload} diff --git a/src/policyengine/core/simulation.py b/src/policyengine/core/simulation.py index f337a7f2..3dd537b2 100644 --- a/src/policyengine/core/simulation.py +++ b/src/policyengine/core/simulation.py @@ -240,7 +240,7 @@ def release_bundle(self) -> dict[str, Any]: if self.tax_benefit_model_version is not None else {} ) - result = { + result: dict[str, Any] = { **bundle, "dataset_filepath": self.dataset.filepath if self.dataset is not None @@ -250,4 +250,12 @@ def release_bundle(self) -> dict[str, Any]: recorded = getattr(self.output_dataset, "metadata", {}).get("spm_config") result["spm_config"] = dict(recorded or self.spm_config) result["spm"] = self.spm_provenance() + # Stored inputs the country loader mapped onto renamed live inputs + # (for example the US WIC take-up draw), as the run recorded them. + # See ``policyengine.tax_benefit_models.us.legacy_inputs``. + renames = (getattr(self.output_dataset, "metadata", None) or {}).get( + "legacy_input_renames" + ) + if renames is not None: + result["legacy_input_renames"] = dict(renames) return result diff --git a/src/policyengine/tax_benefit_models/common/model_version.py b/src/policyengine/tax_benefit_models/common/model_version.py index 1d6982b9..131b2f5a 100644 --- a/src/policyengine/tax_benefit_models/common/model_version.py +++ b/src/policyengine/tax_benefit_models/common/model_version.py @@ -356,8 +356,13 @@ def save(self, simulation: Simulation) -> None: "something to persist." ) serialized_spm = None + serialized_renames = None if self.country_code == "us": from policyengine.core.spm import SPMProvenance + from policyengine.tax_benefit_models.us.legacy_inputs import ( + RENAMES_H5_DATASET, + RENAMES_RECORD_KEY, + ) receipt = SPMProvenance.model_validate(simulation.spm_provenance()) if ( @@ -374,6 +379,16 @@ def save(self, simulation: Simulation) -> None: }, sort_keys=True, ) + renames = (getattr(simulation.output_dataset, "metadata", None) or {}).get( + RENAMES_RECORD_KEY + ) + if renames is None: + raise ValueError( + "This US output does not record which renamed stored " + "inputs were mapped when it was calculated; run again " + "before saving" + ) + serialized_renames = json.dumps(dict(renames), sort_keys=True) simulation.output_dataset.save() if serialized_spm is not None: # Store UTF-8 JSON in a dataset rather than an attribute: the @@ -385,6 +400,12 @@ def save(self, simulation: Simulation) -> None: data=serialized_spm, dtype=h5py.string_dtype("utf-8"), ) + if serialized_renames is not None: + stream.create_dataset( + RENAMES_H5_DATASET, + data=serialized_renames, + dtype=h5py.string_dtype("utf-8"), + ) def load(self, simulation: Simulation) -> None: """Rehydrate the simulation's output dataset from disk. @@ -399,6 +420,10 @@ def load(self, simulation: Simulation) -> None: receipt = None if self.country_code == "us": from policyengine.core.spm import SPMProvenance + from policyengine.tax_benefit_models.us.legacy_inputs import ( + RENAMES_H5_DATASET, + RENAMES_RECORD_KEY, + ) with h5py.File(filepath, "r") as stream: raw = ( @@ -406,6 +431,11 @@ def load(self, simulation: Simulation) -> None: if "policyengine_spm" in stream else None ) + raw_renames = ( + stream[RENAMES_H5_DATASET].asstr()[()] + if RENAMES_H5_DATASET in stream + else None + ) if raw is None: raise ValueError( "Saved US simulation has no SPM configuration or receipt" @@ -414,6 +444,16 @@ def load(self, simulation: Simulation) -> None: if recorded["config"] != simulation.spm_config: raise ValueError("Saved US simulation uses different SPM settings") receipt = SPMProvenance.model_validate(recorded["provenance"]) + # Outputs saved before stored inputs were mapped onto renamed + # live inputs were calculated without them (for example with + # every WIC-eligible person taking WIC up), so they are not + # reused. ``Simulation.ensure()`` runs such a simulation again. + if raw_renames is None: + raise ValueError( + "Saved US simulation predates the mapping of renamed " + "stored inputs (it has no record of them); run it again" + ) + recorded_renames = json.loads(raw_renames) simulation.output_dataset = self._dataset_class( id=simulation.id, @@ -428,7 +468,9 @@ def load(self, simulation: Simulation) -> None: simulation.spm_receipt = receipt simulation.spm = SPMSelection.model_validate(recorded["config"]) - simulation.output_dataset.metadata["spm_config"] = recorded["config"] + simulation.output_dataset.metadata.update( + {"spm_config": recorded["config"], RENAMES_RECORD_KEY: recorded_renames} + ) if os.path.exists(filepath): simulation.created_at = datetime.datetime.fromtimestamp( diff --git a/src/policyengine/tax_benefit_models/us/datasets.py b/src/policyengine/tax_benefit_models/us/datasets.py index 2ff4aa47..6c855969 100644 --- a/src/policyengine/tax_benefit_models/us/datasets.py +++ b/src/policyengine/tax_benefit_models/us/datasets.py @@ -24,6 +24,11 @@ from policyengine.tax_benefit_models.common.model_version import ( build_runtime_dataset_provenance, ) +from policyengine.tax_benefit_models.us.legacy_inputs import ( + RENAMES_RECORD_KEY, + apply_legacy_input_renames_to_microsimulation, + pending_legacy_input_renames, +) from policyengine.utils.hashing import sha256_file @@ -197,7 +202,15 @@ def _core_h5_entity_lengths(h5_file: h5py.File, year: int) -> dict[str, int]: def _core_h5_variable_entities() -> dict[str, str]: from policyengine_us.system import system - return {name: variable.entity.key for name, variable in system.variables.items()} + entities = { + name: variable.entity.key for name, variable in system.variables.items() + } + # A stored input the engine has since renamed belongs to its live input's + # entity. Without this its entity is guessed from its length, and a column + # as long as two entities is dropped before the rename can map it. + for legacy, live in pending_legacy_input_renames(system.variables).items(): + entities[legacy] = entities[live] + return entities def _validate_entity_ids(data: dict[str, pd.DataFrame]) -> None: @@ -353,6 +366,9 @@ def create_datasets( ) dataset_stem = source.name sim = Microsimulation(dataset=source.path, spm=resolve_spm_selection()) + # Map stored inputs the engine has since renamed before extracting, + # so each year's file stores the live input rather than dropping it. + legacy_input_renames = apply_legacy_input_renames_to_microsimulation(sim) for year in years: # Get all input variables from the simulation @@ -496,6 +512,9 @@ def create_datasets( description=f"US Dataset for year {year} based on {dataset_stem}", filepath=f"{data_folder}/{dataset_stem}_year_{year}.h5", year=int(year), + metadata={ + RENAMES_RECORD_KEY: dict(sorted(legacy_input_renames.items())) + }, data=USYearData( person=MicroDataFrame(person_df, weights="person_weight"), household=MicroDataFrame(household_df, weights="household_weight"), diff --git a/src/policyengine/tax_benefit_models/us/legacy_inputs.py b/src/policyengine/tax_benefit_models/us/legacy_inputs.py new file mode 100644 index 00000000..3e71c47c --- /dev/null +++ b/src/policyengine/tax_benefit_models/us/legacy_inputs.py @@ -0,0 +1,345 @@ +"""Map stored US inputs that policyengine-us has since renamed. + +A country engine sets a stored column as an input only when it defines a +variable of that name, and skips every other column. When policyengine-us +renames an input variable, datasets written before the rename still store the +old name, so the engine skips the stored values and the live input falls back +to its default. + +:data:`LEGACY_INPUT_RENAMES` registers each such rename, and +:func:`apply_legacy_input_renames` sets the live input from the stored legacy +column. A rename applies only when all of these hold: + +- the loaded engine does not define the legacy name (an engine that does reads + the stored column itself); +- the loaded engine defines the live name; +- the stored data carries the legacy column but not the live one (data that + already stores the live name loads it natively). + +So the mapping retires itself once the data is re-cut under the live name, or +if the engine defines the legacy name again. +""" + +from __future__ import annotations + +import inspect +from collections.abc import Iterable, Iterator, Mapping +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd + +#: Stored input columns that policyengine-us has since renamed, mapped to the +#: live input that replaced each one. +#: +#: ``would_claim_wic``: policyengine-us 2.2.1 defines no ``would_claim_wic``. +#: Its WIC take-up input is ``takes_up_wic_if_eligible`` (Person, MONTH, +#: ``default_value = True``), and ``wic`` is ``defined_for`` it. The certified +#: US default ``populace-us-2024-spm-20260915`` stores the take-up draw as +#: ``would_claim_wic`` only, so without this mapping every WIC-eligible person +#: takes WIC up (PolicyEngine/microcosm#1026). +LEGACY_INPUT_RENAMES: dict[str, str] = { + "would_claim_wic": "takes_up_wic_if_eligible", +} + +#: Key under which a run records the renames it applied, ``{legacy: live}``: +#: in an output dataset's ``metadata``, ``Simulation.release_bundle``, a +#: managed Microsimulation's ``policyengine_bundle`` and a run record's +#: results. +RENAMES_RECORD_KEY = "legacy_input_renames" + +#: H5 dataset holding that record, as UTF-8 JSON, in a saved US output file. +RENAMES_H5_DATASET = "policyengine_legacy_input_renames" + +#: ``{year: {entity key: stored table}}``, one table per entity per dataset +#: year, in the dataset's own row order. +StoredTables = Mapping[int, Mapping[str, pd.DataFrame]] + + +def pending_legacy_input_renames(variables: Any) -> dict[str, str]: + """Return the register entries an engine needs mapped. + + An entry is pending when the engine's ``variables`` define the live name + and not the legacy name. Anything other than a variable mapping yields + nothing. + """ + if not isinstance(variables, Mapping): + return {} + return { + legacy: live + for legacy, live in LEGACY_INPUT_RENAMES.items() + if legacy not in variables and live in variables + } + + +def apply_legacy_input_renames( + simulation, stored_tables: StoredTables +) -> dict[str, str]: + """Set each pending live input from its stored legacy column. + + For each pending rename (see :func:`pending_legacy_input_renames`), every + year whose stored table for the live input's entity carries the legacy + column but not the live one is mapped: the live input is set to the stored + values for each month of that year (or for the year, if the live input is + yearly). + + Every mapped table is checked before anything is set. Its ``{entity}_id`` + column must list the simulation's entity IDs in the simulation's order, + and its stored values must be complete and, for a boolean live input, + boolean. Otherwise this raises ``ValueError`` rather than set misaligned + or invented values. + + Applying the mapping again sets the same values, so it is idempotent. + + Returns: + The renames applied, ``{legacy: live}``. + """ + variables = simulation.tax_benefit_system.variables + planned: list[tuple[str, str, list[str], np.ndarray]] = [] + for legacy, live in pending_legacy_input_renames(variables).items(): + variable = variables[live] + entity = variable.entity.key + for year, tables in sorted(stored_tables.items()): + table = tables.get(entity) + if table is None or legacy not in table.columns or live in table.columns: + continue + context = f"Cannot map stored {legacy!r} onto {live!r} for {year}" + _check_order(simulation, table, entity, context) + values = _live_values(table[legacy], variable, context) + periods = _periods_of_year(int(year), variable.definition_period, context) + planned.append((legacy, live, periods, values)) + + applied: dict[str, str] = {} + for legacy, live, periods, values in planned: + for period in periods: + simulation.set_input(live, period, values) + applied[legacy] = live + _record_input_variables(simulation, applied.values()) + return applied + + +def apply_legacy_input_renames_to_microsimulation(microsimulation) -> dict[str, str]: + """Apply the register to a country Microsimulation built from a file. + + The stored tables come from the dataset the Microsimulation loaded (see + :func:`stored_entity_tables`). The Microsimulation and every branch it + already has (a reform's ``baseline`` is branched off during construction, + before this runs) are each mapped. + + Returns: + The renames applied, ``{legacy: live}``. + """ + variables = microsimulation.tax_benefit_system.variables + pending = pending_legacy_input_renames(variables) + if not pending: + return {} + renames_by_entity: dict[str, dict[str, str]] = {} + for legacy, live in pending.items(): + renames_by_entity.setdefault(variables[live].entity.key, {})[legacy] = live + stored_tables = stored_entity_tables( + getattr(microsimulation, "dataset", None), renames_by_entity + ) + applied: dict[str, str] = {} + for simulation in _simulation_and_branches(microsimulation): + applied.update(apply_legacy_input_renames(simulation, stored_tables)) + return applied + + +def stored_entity_tables( + dataset: Any, renames_by_entity: Mapping[str, Mapping[str, str]] +) -> dict[int, dict[str, pd.DataFrame]]: + """Return the per-year stored tables a country Microsimulation loaded. + + ``renames_by_entity`` maps each entity key to the ``{legacy: live}`` + renames whose live input belongs to it. + + policyengine-us loads an entity-table H5 (the layout of the certified + Populace default) into a multi-year dataset whose ``datasets`` map each + year to that year's entity DataFrames. Those keep every stored column, including the + ones the engine skipped, so they are returned as they are. + + A policyengine-core ``variable/period`` H5 is read again from + ``dataset.file_path``, but only each entity's ID column and the legacy + columns that need mapping (see :func:`_variable_centric_tables`). + + Any other dataset yields no tables. + """ + datasets = _static_attribute(dataset, "datasets") + if isinstance(datasets, Mapping): + tables: dict[int, dict[str, pd.DataFrame]] = {} + for year, single_year in datasets.items(): + tables[int(year)] = { + entity: table + for entity in renames_by_entity + if isinstance( + table := _static_attribute(single_year, entity), pd.DataFrame + ) + } + return tables + if _static_attribute(dataset, "data_format") in ("arrays", "time_period_arrays"): + file_path = _static_attribute(dataset, "file_path") + if file_path is not None and Path(file_path).suffix == ".h5": + if Path(file_path).is_file(): + return _variable_centric_tables( + Path(file_path), + _static_attribute(dataset, "time_period"), + renames_by_entity, + ) + return {} + + +def _variable_centric_tables( + path: Path, + default_period: Any, + renames_by_entity: Mapping[str, Mapping[str, str]], +) -> dict[int, dict[str, pd.DataFrame]]: + """Read legacy columns from a ``variable/period`` H5 into yearly tables. + + A file that stores a live name under any period loads it natively, so the + legacy column it replaces is not read at all. A legacy column is read only + for yearly periods; one stored for a part of a year is refused, since it + cannot stand for the whole year. + + policyengine-core builds each population from the ID array stored under + the first period key, so a year that stores no IDs of its own is checked + against those. + """ + import h5py + + tables: dict[int, dict[str, pd.DataFrame]] = {} + with h5py.File(path, "r") as file: + for entity, renames in renames_by_entity.items(): + id_column = f"{entity}_id" + legacy_columns = sorted( + legacy + for legacy, live in renames.items() + if legacy in file and live not in file + ) + if not legacy_columns or id_column not in file: + continue + ids_by_period = _stored_periods(file[id_column], default_period) + first_ids = next(iter(ids_by_period.values()), None) + columns_by_year: dict[int, dict[str, np.ndarray]] = {} + for column in legacy_columns: + for period, stored in _stored_periods( + file[column], default_period + ).items(): + if not period.isdigit(): + raise ValueError( + f"Cannot map stored {column!r} from {path}: it is " + f"stored for period {period!r}, and only yearly " + "periods are supported." + ) + columns_by_year.setdefault(int(period), {})[column] = stored + for year, columns in sorted(columns_by_year.items()): + ids = ids_by_period.get(str(year), first_ids) + if ids is None or any( + len(array) != len(ids) for array in columns.values() + ): + raise ValueError( + f"Cannot map stored columns from {path} for {year}: " + f"their length differs from the stored {id_column}." + ) + tables.setdefault(year, {})[entity] = pd.DataFrame( + {id_column: ids, **columns} + ) + return tables + + +def _stored_periods(node: Any, default_period: Any) -> dict[str, np.ndarray]: + """Map each stored period of an H5 node to its values, in file order. + + A flat ``arrays`` file stores one array per variable, which + policyengine-core sets for the dataset's own period. + """ + import h5py + + if isinstance(node, h5py.Dataset): + if default_period is None: + return {} + return {str(default_period): node[()]} + return {str(key): node[key][()] for key in node.keys()} + + +def _static_attribute(obj: Any, name: str) -> Any: + """Return ``obj.``, or ``None`` when ``obj`` has no such attribute. + + Only attributes found without a dynamic ``__getattr__`` count. + policyengine-core's ``Dataset.__getattr__`` loads any other name as an H5 + key and raises ``KeyError`` when the file has none, so probing a dataset + with ``getattr(dataset, name, None)`` is not safe. + """ + try: + inspect.getattr_static(obj, name) + except AttributeError: + return None + return getattr(obj, name) + + +def _check_order(simulation, table: pd.DataFrame, entity: str, context: str) -> None: + id_column = f"{entity}_id" + if id_column not in table.columns: + raise ValueError( + f"{context}: the stored {entity} table has no {id_column} column, " + "so its row order cannot be checked against the simulation." + ) + stored_ids = table[id_column].to_numpy() + simulated_ids = np.asarray(simulation.populations[entity].ids) + if not np.array_equal(stored_ids, simulated_ids): + raise ValueError( + f"{context}: the stored {entity} table is not in the simulation's " + f"{entity} order ({id_column} differs), so its values would attach " + "to the wrong rows." + ) + + +def _live_values(stored: pd.Series, variable: Any, context: str) -> np.ndarray: + if stored.isna().any(): + raise ValueError(f"{context}: the stored column has missing values.") + if variable.value_type is not bool: + return np.asarray(stored.to_numpy()) + if pd.api.types.is_bool_dtype(stored.dtype): + return np.asarray(stored.to_numpy(dtype=bool)) + if pd.api.types.is_numeric_dtype(stored.dtype) and stored.isin((0, 1)).all(): + return np.asarray(stored.to_numpy(), dtype=bool) + raise ValueError(f"{context}: the stored values are not boolean.") + + +def _periods_of_year(year: int, definition_period: Any, context: str) -> list[str]: + unit = str(definition_period) + if unit == "month": + return [f"{year}-{month:02d}" for month in range(1, 13)] + if unit == "year": + return [str(year)] + raise ValueError(f"{context}: definition period {unit!r} is not supported.") + + +def _record_input_variables(simulation, live_names: Iterable[str]) -> None: + """List mapped inputs as inputs, as if the dataset had stored them. + + A country Microsimulation lists ``input_variables`` once, at construction. + Dataset extraction (``create_datasets``) exports that list, and + policyengine-core's ``derivative`` keeps only those inputs in its clone, so + a mapped input must join it. + """ + input_variables = getattr(simulation, "input_variables", None) + if not isinstance(input_variables, list): + return + missing = [name for name in live_names if name not in input_variables] + if missing: + simulation.input_variables = [*input_variables, *missing] + + +def _simulation_and_branches(simulation) -> Iterator[Any]: + seen: set[int] = set() + stack = [simulation] + while stack: + current = stack.pop() + if id(current) in seen: + continue + seen.add(id(current)) + yield current + branches = getattr(current, "branches", None) + if isinstance(branches, Mapping): + stack.extend(branches.values()) diff --git a/src/policyengine/tax_benefit_models/us/model.py b/src/policyengine/tax_benefit_models/us/model.py index 51338b3e..54525575 100644 --- a/src/policyengine/tax_benefit_models/us/model.py +++ b/src/policyengine/tax_benefit_models/us/model.py @@ -17,6 +17,11 @@ ) from .datasets import PolicyEngineUSDataset, USYearData, _validate_entity_ids +from .legacy_inputs import ( + RENAMES_RECORD_KEY, + apply_legacy_input_renames, + apply_legacy_input_renames_to_microsimulation, +) from .spm import ( SPMProvenance, SPMSelection, @@ -219,14 +224,19 @@ class InputMicrosimulation(Microsimulation): # leaves the module-level one untouched. Building populations # against the module-level system would hide reform-registered # variables like ``ctc_minimum_refundable_amount`` at calc time. + legacy_input_renames: dict[str, str] = {} if microsim.baseline is not None: + legacy_input_renames.update( + self._build_simulation_from_dataset( + microsim.baseline, + dataset, + microsim.baseline.tax_benefit_system, + ) + ) + legacy_input_renames.update( self._build_simulation_from_dataset( - microsim.baseline, - dataset, - microsim.baseline.tax_benefit_system, + microsim, dataset, microsim.tax_benefit_system ) - self._build_simulation_from_dataset( - microsim, dataset, microsim.tax_benefit_system ) data = { @@ -316,7 +326,10 @@ class InputMicrosimulation(Microsimulation): filepath=str(_output_dataset_filepath(simulation)), year=simulation.dataset.year, is_output_dataset=True, - metadata={"spm_config": dict(microsim.spm_config)}, + metadata={ + "spm_config": dict(microsim.spm_config), + RENAMES_RECORD_KEY: dict(sorted(legacy_input_renames.items())), + }, data=USYearData( person=data["person"], marital_unit=data["marital_unit"], @@ -337,6 +350,12 @@ def _build_simulation_from_dataset(self, microsim, dataset, system): Mirrors the policyengine-uk pattern of instantiating entities from IDs first and then setting variable inputs. Handles both the legacy ``person_X_id`` and the ``X_id`` column-naming conventions. + + Stored columns the engine has since renamed are mapped onto their + live inputs (see ``legacy_inputs.LEGACY_INPUT_RENAMES``). + + Returns: + The legacy input renames applied, ``{legacy: live}``. """ import numpy as np from policyengine_core.simulations.simulation_builder import ( @@ -453,6 +472,7 @@ def _build_simulation_from_dataset(self, microsim, dataset, system): "person_marital_unit_id", } + stored_tables = {} for entity_name, entity_df in [ ("person", dataset.data.person), ("household", dataset.data.household), @@ -463,10 +483,16 @@ def _build_simulation_from_dataset(self, microsim, dataset, system): ]: df = pd.DataFrame(entity_df).set_index(f"{entity_name}_id", drop=False) df = df.loc[microsim.populations[entity_name].ids] + stored_tables[entity_name] = df for column in df.columns: if column not in id_columns and column in system.variables: microsim.set_input(column, dataset.year, df[column].values) + # The loop above skips columns the engine does not define, including + # stored inputs it has since renamed. The tables are already in the + # simulation's entity order. + return apply_legacy_input_renames(microsim, {dataset.year: stored_tables}) + def managed_microsimulation( *, @@ -480,6 +506,10 @@ def managed_microsimulation( By default this enforces the dataset selection from the bundled ``policyengine.py`` release manifest. Arbitrary dataset URIs require ``allow_unmanaged=True``. + + Stored inputs the engine has since renamed are mapped onto their live + inputs (see ``legacy_inputs.LEGACY_INPUT_RENAMES``), and the renames + applied are recorded as ``policyengine_bundle["legacy_input_renames"]``. """ from policyengine_us import Microsimulation @@ -497,6 +527,7 @@ def managed_microsimulation( allow_unmanaged=allow_unmanaged, ) microsim = Microsimulation(dataset=source.path, spm=selection, **kwargs) + legacy_input_renames = apply_legacy_input_renames_to_microsimulation(microsim) microsim.policyengine_bundle = dict(us_latest.release_bundle) microsim.policyengine_bundle.update( build_runtime_dataset_provenance( @@ -506,6 +537,9 @@ def managed_microsimulation( ) ) microsim.policyengine_bundle["spm"] = dict(microsim.spm_config) + microsim.policyengine_bundle[RENAMES_RECORD_KEY] = dict( + sorted(legacy_input_renames.items()) + ) return microsim diff --git a/tests/test_us_legacy_inputs.py b/tests/test_us_legacy_inputs.py new file mode 100644 index 00000000..f29e915a --- /dev/null +++ b/tests/test_us_legacy_inputs.py @@ -0,0 +1,578 @@ +"""Unit and property tests for the US legacy input rename. + +The certified US default release stores the WIC take-up draw as +``would_claim_wic``, which policyengine-us 2.x renamed to +``takes_up_wic_if_eligible`` (PolicyEngine/microcosm#1026). These tests state +the mapping's invariants over generated inputs, using lightweight fakes of the +slice of a country simulation the mapping touches, and run the draw through +both stored-table readers (the multi-year entity tables and a +policyengine-core ``variable/period`` H5): + +- **Draw preserved.** For any stored boolean draw and any set of dataset years, + the live input equals the stored draw for every month of every year. +- **No-op when it does not apply.** Nothing is set when the engine defines the + legacy name, when the engine lacks the live name, or when the data already + stores the live name. +- **Order check.** A stored table that is not in the simulation's order is + refused, and nothing is set. +- **Idempotence.** Applying the mapping twice gives the same inputs as once. + +``test_us_legacy_inputs_integration.py`` runs the same mapping inside real +policyengine-us simulations. +""" + +from __future__ import annotations + +import tempfile +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import MagicMock + +import h5py +import numpy as np +import pandas as pd +import pytest +from hypothesis import assume, given, settings +from hypothesis import strategies as st + +from policyengine.tax_benefit_models.us import legacy_inputs +from policyengine.tax_benefit_models.us.legacy_inputs import ( + LEGACY_INPUT_RENAMES, + apply_legacy_input_renames, + apply_legacy_input_renames_to_microsimulation, + pending_legacy_input_renames, + stored_entity_tables, +) + +LEGACY = "would_claim_wic" +LIVE = "takes_up_wic_if_eligible" + + +def _variable(period="month", entity="person", value_type=bool): + return SimpleNamespace( + entity=SimpleNamespace(key=entity), + definition_period=period, + value_type=value_type, + ) + + +def _variables(*names, period="month"): + return {name: _variable(period) for name in names} + + +class FakeSimulation: + """The slice of a country simulation the mapping reads and writes.""" + + def __init__(self, variables, ids, *, dataset=None, input_variables=()): + self.tax_benefit_system = SimpleNamespace(variables=variables) + self.populations = {"person": SimpleNamespace(ids=np.asarray(ids))} + self.dataset = dataset + self.input_variables = list(input_variables) + self.branches = {} + self.inputs: dict[tuple[str, str], np.ndarray] = {} + + def set_input(self, variable, period, values): + self.inputs[(variable, period)] = np.array(values, copy=True) + + +def _person(ids, draw, **columns): + return pd.DataFrame({"person_id": list(ids), LEGACY: list(draw), **columns}) + + +def _tables(person_by_year): + return {year: {"person": person} for year, person in person_by_year.items()} + + +def _months(year): + return [f"{year}-{month:02d}" for month in range(1, 13)] + + +def _snapshot(simulation): + return {key: value.tolist() for key, value in simulation.inputs.items()} + + +# --- Strategies ------------------------------------------------------ + +YEARS = st.lists(st.integers(2015, 2040), min_size=1, max_size=4, unique=True) +IDS = st.lists(st.integers(-(10**9), 10**9), min_size=1, max_size=30, unique=True).map( + np.asarray +) +#: The same boolean draw can be stored as bool, integer or float 0/1. +DRAW_DTYPE = st.sampled_from([bool, np.int8, np.int64, np.float64]) + + +@st.composite +def stored_draws(draw): + """Simulation IDs, dataset years, and each year's stored boolean draw.""" + ids = draw(IDS) + years = draw(YEARS) + dtype = draw(DRAW_DTYPE) + draws = { + year: np.asarray( + draw(st.lists(st.booleans(), min_size=len(ids), max_size=len(ids))) + ).astype(dtype) + for year in years + } + return ids, draws + + +# --- Invariant: draw preserved --------------------------------------- + + +@settings(max_examples=200, deadline=None) +@given(case=stored_draws()) +def test_live_input_equals_the_stored_draw_every_month_of_every_year(case): + ids, draws = case + simulation = FakeSimulation(_variables(LIVE, "person_id"), ids) + + applied = apply_legacy_input_renames( + simulation, + _tables({year: _person(ids, draw) for year, draw in draws.items()}), + ) + + assert applied == {LEGACY: LIVE} + assert set(simulation.inputs) == { + (LIVE, month) for year in draws for month in _months(year) + } + for year, draw in draws.items(): + for month in _months(year): + values = simulation.inputs[(LIVE, month)] + assert values.dtype == bool + assert values.tolist() == draw.astype(bool).tolist() + assert simulation.input_variables == [LIVE] + + +# --- Invariant: no-op when it does not apply -------------------------- + + +@settings(max_examples=200, deadline=None) +@given( + case=stored_draws(), + reason=st.sampled_from( + ["engine_defines_legacy", "engine_lacks_live", "data_stores_live"] + ), +) +def test_nothing_is_set_when_the_rename_does_not_apply(case, reason): + ids, draws = case + person_by_year = {year: _person(ids, draw) for year, draw in draws.items()} + if reason == "data_stores_live": + for person in person_by_year.values(): + person[LIVE] = ~person[LEGACY].astype(bool) + names = { + "engine_defines_legacy": (LEGACY, LIVE), + "engine_lacks_live": ("person_id",), + "data_stores_live": (LIVE,), + }[reason] + simulation = FakeSimulation(_variables(*names), ids, input_variables=["age"]) + + assert apply_legacy_input_renames(simulation, _tables(person_by_year)) == {} + assert simulation.inputs == {} + assert simulation.input_variables == ["age"] + + +@settings(max_examples=100, deadline=None) +@given(case=stored_draws(), data=st.data()) +def test_only_years_whose_data_lacks_the_live_name_are_mapped(case, data): + ids, draws = case + live_years = data.draw(st.sets(st.sampled_from(sorted(draws)))) + person_by_year = {year: _person(ids, draw) for year, draw in draws.items()} + for year in live_years: + person_by_year[year][LIVE] = True + simulation = FakeSimulation(_variables(LIVE), ids) + + applied = apply_legacy_input_renames(simulation, _tables(person_by_year)) + + mapped_years = set(draws) - live_years + assert applied == ({LEGACY: LIVE} if mapped_years else {}) + assert set(simulation.inputs) == { + (LIVE, month) for year in mapped_years for month in _months(year) + } + + +# --- Invariant: order check ------------------------------------------- + + +@settings(max_examples=200, deadline=None) +@given(case=stored_draws(), data=st.data()) +def test_a_table_out_of_simulation_order_is_refused_and_nothing_is_set(case, data): + ids, draws = case + years = sorted(draws) + bad_year = data.draw(st.sampled_from(years)) + stored_ids = np.asarray( + data.draw( + st.one_of( + # The same people in another order. + st.permutations(ids.tolist()), + # A person the simulation does not have. + st.just([*ids.tolist()[:-1], int(ids.max()) + 1]), + # A person missing, or one too many. + st.just(ids.tolist()[:-1]), + st.just([*ids.tolist(), int(ids.max()) + 1]), + ) + ) + ) + assume(not np.array_equal(stored_ids, ids)) + person_by_year = {} + for year in years: + row_ids = stored_ids if year == bad_year else ids + person_by_year[year] = _person(row_ids, np.ones(len(row_ids), dtype=bool)) + simulation = FakeSimulation(_variables(LIVE), ids) + + with pytest.raises(ValueError, match="not in the simulation's person order"): + apply_legacy_input_renames(simulation, _tables(person_by_year)) + # Aligned years are checked and set together, so none was set either. + assert simulation.inputs == {} + assert simulation.input_variables == [] + + +def test_a_table_without_ids_is_refused(): + simulation = FakeSimulation(_variables(LIVE), [1, 2]) + person = pd.DataFrame({LEGACY: [True, False]}) + with pytest.raises(ValueError, match="no person_id column"): + apply_legacy_input_renames(simulation, _tables({2024: person})) + assert simulation.inputs == {} + + +# --- Invariant: idempotence ------------------------------------------- + + +@settings(max_examples=200, deadline=None) +@given(case=stored_draws()) +def test_applying_twice_gives_the_same_inputs_as_once(case): + ids, draws = case + tables = _tables({year: _person(ids, draw) for year, draw in draws.items()}) + once = FakeSimulation(_variables(LIVE), ids) + twice = FakeSimulation(_variables(LIVE), ids) + + applied_once = apply_legacy_input_renames(once, tables) + apply_legacy_input_renames(twice, tables) + applied_twice = apply_legacy_input_renames(twice, tables) + + assert applied_once == applied_twice == {LEGACY: LIVE} + assert _snapshot(once) == _snapshot(twice) + assert once.input_variables == twice.input_variables == [LIVE] + + +# --- Examples --------------------------------------------------------- + + +def test_the_register_holds_only_the_wic_take_up_rename(): + assert LEGACY_INPUT_RENAMES == {LEGACY: LIVE} + + +@pytest.mark.parametrize( + ("names", "expected"), + [ + ((LIVE,), {LEGACY: LIVE}), + ((LEGACY, LIVE), {}), + ((LEGACY,), {}), + ((), {}), + ], +) +def test_pending_renames_follow_what_the_engine_defines(names, expected): + assert pending_legacy_input_renames(_variables(*names)) == expected + + +def test_pending_renames_ignore_anything_but_a_variable_mapping(): + assert pending_legacy_input_renames(MagicMock()) == {} + assert pending_legacy_input_renames(None) == {} + + +def test_register_changes_take_effect_without_reloading(monkeypatch): + monkeypatch.setattr(legacy_inputs, "LEGACY_INPUT_RENAMES", {}) + simulation = FakeSimulation(_variables(LIVE), [1]) + assert ( + apply_legacy_input_renames(simulation, _tables({2024: _person([1], [0])})) == {} + ) + assert simulation.inputs == {} + + +def test_a_yearly_live_input_is_set_once_per_year(): + simulation = FakeSimulation(_variables(LIVE, period="year"), [1, 2]) + apply_legacy_input_renames( + simulation, _tables({2024: _person([1, 2], [True, False])}) + ) + assert list(simulation.inputs) == [(LIVE, "2024")] + + +def test_an_unsupported_definition_period_is_refused(): + simulation = FakeSimulation(_variables(LIVE, period="eternity"), [1]) + with pytest.raises(ValueError, match="'eternity' is not supported"): + apply_legacy_input_renames(simulation, _tables({2024: _person([1], [True])})) + assert simulation.inputs == {} + + +@pytest.mark.parametrize( + ("draw", "message"), + [ + ([True, None], "missing values"), + ([1.0, np.nan], "missing values"), + ([0, 2], "not boolean"), + (["False", "True"], "not boolean"), + ], +) +def test_a_draw_that_is_not_a_complete_boolean_is_refused(draw, message): + simulation = FakeSimulation(_variables(LIVE), [1, 2]) + with pytest.raises(ValueError, match=message): + apply_legacy_input_renames(simulation, _tables({2024: _person([1, 2], draw)})) + assert simulation.inputs == {} + + +def test_a_non_boolean_live_input_keeps_the_stored_values(): + variables = {LIVE: _variable(value_type=float)} + simulation = FakeSimulation(variables, [1, 2]) + apply_legacy_input_renames(simulation, _tables({2024: _person([1, 2], [0.5, 2])})) + assert simulation.inputs[(LIVE, "2024-06")].tolist() == [0.5, 2.0] + + +def test_the_live_input_entity_selects_the_stored_table(): + variables = {LIVE: _variable(entity="spm_unit")} + simulation = FakeSimulation(variables, [1]) + simulation.populations["spm_unit"] = SimpleNamespace(ids=np.asarray([7, 8])) + tables = { + 2024: { + "person": _person([1], [False]), + "spm_unit": pd.DataFrame({"spm_unit_id": [7, 8], LEGACY: [True, False]}), + } + } + apply_legacy_input_renames(simulation, tables) + assert simulation.inputs[(LIVE, "2024-01")].tolist() == [True, False] + + +def test_an_existing_input_variable_list_is_not_duplicated(): + simulation = FakeSimulation(_variables(LIVE), [1], input_variables=[LIVE]) + apply_legacy_input_renames(simulation, _tables({2024: _person([1], [False])})) + assert simulation.input_variables == [LIVE] + + +# --- Country Microsimulation entry point ------------------------------ + + +def _multi_year_dataset(person_by_year): + """The shape of policyengine-us's USMultiYearDataset that is read.""" + return SimpleNamespace( + datasets={ + year: SimpleNamespace(person=person, household=pd.DataFrame()) + for year, person in person_by_year.items() + } + ) + + +def test_microsimulation_and_its_existing_branches_are_all_mapped(): + person_by_year = { + 2024: _person([3, 1, 2], [False, True, False]), + 2025: _person([3, 1, 2], [True, True, False]), + } + dataset = _multi_year_dataset(person_by_year) + microsimulation = FakeSimulation(_variables(LIVE), [3, 1, 2], dataset=dataset) + baseline = FakeSimulation(_variables(LIVE), [3, 1, 2], dataset=dataset) + microsimulation.branches = {"baseline": baseline} + baseline.branches = {"loop": microsimulation} + + applied = apply_legacy_input_renames_to_microsimulation(microsimulation) + + assert applied == {LEGACY: LIVE} + for simulation in (microsimulation, baseline): + assert len(simulation.inputs) == 24 + assert simulation.inputs[(LIVE, "2024-03")].tolist() == [False, True, False] + assert simulation.inputs[(LIVE, "2025-11")].tolist() == [True, True, False] + + +def test_microsimulation_with_an_unrecognised_engine_or_dataset_is_left_alone(): + assert apply_legacy_input_renames_to_microsimulation(MagicMock()) == {} + simulation = FakeSimulation(_variables(LIVE), [1], dataset=object()) + assert apply_legacy_input_renames_to_microsimulation(simulation) == {} + assert simulation.inputs == {} + + +@settings(max_examples=100, deadline=None) +@given(case=stored_draws(), branch_count=st.integers(0, 2)) +def test_microsimulation_draw_is_preserved_on_every_branch(case, branch_count): + """Draw preserved, through the country Microsimulation entry point.""" + ids, draws = case + dataset = _multi_year_dataset( + {year: _person(ids, draw) for year, draw in draws.items()} + ) + microsimulation = FakeSimulation(_variables(LIVE), ids, dataset=dataset) + for index in range(branch_count): + microsimulation.branches[f"branch_{index}"] = FakeSimulation( + _variables(LIVE), ids, dataset=dataset + ) + + applied = apply_legacy_input_renames_to_microsimulation(microsimulation) + + assert applied == {LEGACY: LIVE} + for simulation in (microsimulation, *microsimulation.branches.values()): + assert set(simulation.inputs) == { + (LIVE, month) for year in draws for month in _months(year) + } + for year, draw in draws.items(): + for month in _months(year): + assert ( + simulation.inputs[(LIVE, month)].tolist() + == draw.astype(bool).tolist() + ) + + +def test_multi_year_tables_are_returned_as_stored(): + person = _person([1], [True], age=[40]) + tables = stored_entity_tables( + _multi_year_dataset({2024: person}), {"person": {LEGACY: LIVE}} + ) + # Only the requested entities, and the stored frames themselves. + assert list(tables) == [2024] + assert list(tables[2024]) == ["person"] + assert tables[2024]["person"] is person + + +def _core_dataset(path, data_format, time_period="2024"): + return SimpleNamespace( + file_path=path, data_format=data_format, time_period=time_period + ) + + +def _write_h5(path, arrays): + with h5py.File(path, "w") as file: + for key, values in arrays.items(): + file[key] = values + return path + + +# --- Variable-centric (policyengine-core ``variable/period``) files ------ + + +@settings(max_examples=50, deadline=None) +@given(case=stored_draws()) +def test_variable_centric_draw_is_preserved_every_month_of_every_year(case): + """Draw preserved, through a file the engine reads as ``variable/period``.""" + ids, draws = case + with tempfile.TemporaryDirectory() as directory: + path = _write_h5( + Path(directory) / "core.h5", + { + f"person_id/{min(draws)}": ids, + **{f"{LEGACY}/{year}": draw for year, draw in draws.items()}, + }, + ) + dataset = _core_dataset(path, "time_period_arrays", str(min(draws))) + simulation = FakeSimulation(_variables(LIVE), ids, dataset=dataset) + + applied = apply_legacy_input_renames_to_microsimulation(simulation) + + assert applied == {LEGACY: LIVE} + for year, draw in draws.items(): + for month in _months(year): + values = simulation.inputs[(LIVE, month)] + assert values.dtype == bool + assert values.tolist() == draw.astype(bool).tolist() + + +def test_variable_centric_file_is_read_by_year(tmp_path): + path = _write_h5( + tmp_path / "core.h5", + { + "person_id/2024": [5, 6], + f"{LEGACY}/2024": [True, False], + f"{LEGACY}/2025": [False, False], + "age/2024": [30, 1], + }, + ) + + tables = stored_entity_tables( + _core_dataset(path, "time_period_arrays"), {"person": {LEGACY: LIVE}} + ) + + assert sorted(tables) == [2024, 2025] + # Only the IDs and the legacy draw are read. + pd.testing.assert_frame_equal( + tables[2024]["person"], + pd.DataFrame({"person_id": [5, 6], LEGACY: [True, False]}), + ) + # A year without its own IDs is aligned to the first stored IDs, as + # policyengine-core builds the population from them. + assert tables[2025]["person"]["person_id"].tolist() == [5, 6] + + +@pytest.mark.parametrize("live_period", ["2024", "2025", "2024-01"]) +def test_variable_centric_file_storing_the_live_name_is_not_read(tmp_path, live_period): + # Data that stores the live name for any period loads it natively, even + # for a period the legacy draw would not be read for. + path = _write_h5( + tmp_path / "recut.h5", + { + "person_id/2024": [5, 6], + f"{LEGACY}/2024": [True, False], + f"{LEGACY}/2024-02": [True, False], + f"{LIVE}/{live_period}": [False, True], + }, + ) + dataset = _core_dataset(path, "time_period_arrays") + assert stored_entity_tables(dataset, {"person": {LEGACY: LIVE}}) == {} + + simulation = FakeSimulation(_variables(LIVE), [5, 6], dataset=dataset) + assert apply_legacy_input_renames_to_microsimulation(simulation) == {} + assert simulation.inputs == {} + + +def test_flat_variable_file_is_read_for_the_dataset_period(tmp_path): + path = _write_h5(tmp_path / "flat.h5", {"person_id": [5, 6], LEGACY: [False, True]}) + + tables = stored_entity_tables( + _core_dataset(path, "arrays"), {"person": {LEGACY: LIVE}} + ) + + assert tables[2024]["person"][LEGACY].tolist() == [False, True] + + +def test_variable_centric_file_with_a_non_yearly_draw_is_refused(tmp_path): + path = _write_h5( + tmp_path / "monthly.h5", {"person_id/2024": [5], f"{LEGACY}/2024-01": [True]} + ) + with pytest.raises(ValueError, match="only yearly periods"): + stored_entity_tables( + _core_dataset(path, "time_period_arrays"), {"person": {LEGACY: LIVE}} + ) + + +def test_variable_centric_file_with_a_mislengthed_draw_is_refused(tmp_path): + path = _write_h5( + tmp_path / "short.h5", {"person_id/2024": [5, 6], f"{LEGACY}/2024": [True]} + ) + with pytest.raises(ValueError, match="length differs"): + stored_entity_tables( + _core_dataset(path, "time_period_arrays"), {"person": {LEGACY: LIVE}} + ) + + +def test_variable_centric_file_without_the_columns_yields_no_tables(tmp_path): + path = _write_h5(tmp_path / "none.h5", {"person_id/2024": [5]}) + assert ( + stored_entity_tables( + _core_dataset(path, "time_period_arrays"), {"person": {LEGACY: LIVE}} + ) + == {} + ) + + +def test_core_dataset_objects_are_probed_without_loading_keys(tmp_path): + """policyengine-core's ``Dataset`` loads unknown attributes as H5 keys. + + Probing it with ``getattr(dataset, "datasets", None)`` raises the H5 + ``KeyError`` instead of returning ``None``, which crashed + ``managed_microsimulation`` on every variable-centric file. + """ + core_data = pytest.importorskip("policyengine_core.data") + path = _write_h5( + tmp_path / "core.h5", + {"person_id/2024": [5, 6], f"{LEGACY}/2024": [True, False]}, + ) + dataset = core_data.Dataset.from_file(str(path), "2024") + with pytest.raises(KeyError): + getattr(dataset, "datasets", None) + + tables = stored_entity_tables(dataset, {"person": {LEGACY: LIVE}}) + + assert tables[2024]["person"][LEGACY].tolist() == [True, False] + simulation = FakeSimulation(_variables(LIVE), [5, 6], dataset=dataset) + assert apply_legacy_input_renames_to_microsimulation(simulation) == {LEGACY: LIVE} + assert simulation.inputs[(LIVE, "2024-01")].tolist() == [True, False] diff --git a/tests/test_us_legacy_inputs_integration.py b/tests/test_us_legacy_inputs_integration.py new file mode 100644 index 00000000..5d0b7ead --- /dev/null +++ b/tests/test_us_legacy_inputs_integration.py @@ -0,0 +1,528 @@ +"""The legacy WIC take-up draw reaches real policyengine-us simulations. + +Each test builds a tiny US household that stores its WIC take-up draw as +``would_claim_wic`` (the layout of the certified Populace default, see +PolicyEngine/microcosm#1026), loads it through one of the load paths, and +checks that a WIC-eligible person whose stored draw is ``False`` gets no WIC, +while the same person gets WIC when the mapping is switched off. + +The household is one adult and two young children with no income, in Los +Angeles County: an infant (stored draw ``False``) and a two-year-old (stored +draw ``True``). Both children are WIC-eligible, which each test checks. +""" + +from __future__ import annotations + +import h5py +import numpy as np +import pandas as pd +import pytest +from hypothesis import given, settings +from hypothesis import strategies as st +from microdf import MicroDataFrame + +pytest.importorskip("policyengine_us") +pytest.importorskip("spm_calculator.policyengine_adapter") + +import policyengine as pe # noqa: E402 +from policyengine.tax_benefit_models.us import legacy_inputs # noqa: E402 +from policyengine.tax_benefit_models.us.datasets import ( # noqa: E402 + PolicyEngineUSDataset, + USYearData, + create_datasets, +) +from policyengine.tax_benefit_models.us.legacy_inputs import ( # noqa: E402 + apply_legacy_input_renames, + apply_legacy_input_renames_to_microsimulation, +) + +YEAR = 2024 +LEGACY = "would_claim_wic" +LIVE = "takes_up_wic_if_eligible" +RENAME = {LEGACY: LIVE} +PERSON_IDS = [1, 2, 3] +#: The stored draw for the adult, the infant and the two-year-old. +DRAW = [False, False, True] +INFANT, TODDLER = 1, 2 +GROUPS = ("household", "tax_unit", "spm_unit", "family", "marital_unit") +MONTHS = [f"{YEAR}-{month:02d}" for month in range(1, 13)] + + +def _frames() -> dict[str, pd.DataFrame]: + """Native entity tables, with person links named ``person__id``.""" + person = pd.DataFrame( + { + "person_id": PERSON_IDS, + "person_household_id": [1, 1, 1], + "person_tax_unit_id": [1, 1, 1], + "person_spm_unit_id": [1, 1, 1], + "person_family_id": [1, 1, 1], + "person_marital_unit_id": [1, 2, 3], + "person_weight": [1.0, 1.0, 1.0], + "age": [30, 0, 2], + LEGACY: DRAW, + } + ) + frames = {"person": person} + for group in GROUPS: + count = 3 if group == "marital_unit" else 1 + frames[group] = pd.DataFrame( + {f"{group}_id": range(1, count + 1), f"{group}_weight": [1.0] * count} + ) + frames["household"]["state_code"] = ["CA"] + frames["household"]["county_fips"] = ["06037"] + return frames + + +def _in_memory_dataset(directory, frames=None) -> PolicyEngineUSDataset: + frames = frames or _frames() + return PolicyEngineUSDataset( + name="legacy-wic-draw", + description="Uncertified three-person WIC take-up fixture", + filepath=str(directory / "input.h5"), + year=YEAR, + data=USYearData( + **{ + entity: MicroDataFrame(frame, weights=f"{entity}_weight") + for entity, frame in frames.items() + } + ), + ) + + +def _write_entity_tables(path) -> str: + """Write the household in the certified Populace default's layout.""" + with pd.HDFStore(path, mode="w") as store: + for entity, frame in _frames().items(): + store.put(entity, frame, format="table", data_columns=True) + store.put("_time_period", pd.Series([YEAR]), format="table") + return str(path) + + +def _write_variable_centric(path) -> str: + """Write the household as a policyengine-core ``variable/period`` H5.""" + with h5py.File(path, "w") as file: + for entity, frame in _frames().items(): + for name in frame.columns: + values = frame[name].to_numpy() + if values.dtype.kind in {"O", "U"}: + values = np.asarray(values, dtype="S") + file.create_dataset(f"{name}/{YEAR}", data=values) + return str(path) + + +def _run(dataset: PolicyEngineUSDataset): + simulation = pe.Simulation( + dataset=dataset, + tax_benefit_model_version=pe.us.model, + extra_variables={ + "person": ["is_wic_eligible", "wic_if_takes_up", LIVE, "wic"], + }, + ) + simulation.run() + return simulation + + +def _assert_children_are_wic_eligible(eligible, wic_if_takes_up): + for child in (INFANT, TODDLER): + assert eligible[child] + assert wic_if_takes_up[child] > 0 + + +def _person_outputs(simulation) -> pd.DataFrame: + return pd.DataFrame(simulation.output_dataset.data.person).set_index("person_id") + + +@pytest.fixture(scope="module") +def mapped_run(tmp_path_factory): + """Load path 1 over the in-memory household, with the mapping on.""" + return _run(_in_memory_dataset(tmp_path_factory.mktemp("mapped_run"))) + + +@pytest.fixture(scope="module") +def mapped_managed(tmp_path_factory): + """Load path 2 over the household's entity-table file, mapping on.""" + path = _write_entity_tables( + tmp_path_factory.mktemp("mapped_managed") / "populace_layout.h5" + ) + return pe.us.managed_microsimulation(dataset=path, allow_unmanaged=True) + + +# --- Load path 1: Simulation.run() over a policyengine.py dataset ------- + + +def test_run_keeps_a_stored_false_draw(mapped_run): + person = _person_outputs(mapped_run) + + _assert_children_are_wic_eligible( + person["is_wic_eligible"].to_numpy(), person["wic_if_takes_up"].to_numpy() + ) + assert person[LIVE].tolist() == DRAW + assert person["wic"].iloc[INFANT] == 0 + assert person["wic"].iloc[TODDLER] == pytest.approx( + person["wic_if_takes_up"].iloc[TODDLER] + ) + assert mapped_run.output_dataset.metadata["legacy_input_renames"] == RENAME + assert mapped_run.release_bundle["legacy_input_renames"] == RENAME + + +def test_run_over_a_core_h5_keeps_a_stored_false_draw(tmp_path): + """Path 1 over a policyengine-core ``variable/period`` file. + + The dataset loader places a column the engine does not define by its + length. Here the draw is as long as both the person and the marital-unit + tables, so it is placed by its live input's entity instead of dropped. + """ + path = _write_variable_centric(tmp_path / "core_layout.h5") + dataset = PolicyEngineUSDataset( + name="legacy-wic-draw-core", + description="Uncertified three-person WIC take-up fixture", + filepath=path, + year=YEAR, + ) + stored = pd.DataFrame(dataset.data.person) + assert len(stored) == len(pd.DataFrame(dataset.data.marital_unit)) + assert stored[LEGACY].astype(bool).tolist() == DRAW + + person = _person_outputs(_run(dataset)) + + _assert_children_are_wic_eligible( + person["is_wic_eligible"].to_numpy(), person["wic_if_takes_up"].to_numpy() + ) + assert person[LIVE].tolist() == DRAW + assert person["wic"].iloc[INFANT] == 0 + assert person["wic"].iloc[TODDLER] > 0 + + +def test_run_without_the_mapping_gives_every_eligible_person_wic(tmp_path, monkeypatch): + monkeypatch.setattr(legacy_inputs, "LEGACY_INPUT_RENAMES", {}) + simulation = _run(_in_memory_dataset(tmp_path)) + person = _person_outputs(simulation) + + _assert_children_are_wic_eligible( + person["is_wic_eligible"].to_numpy(), person["wic_if_takes_up"].to_numpy() + ) + # The default take-up is True, so the stored False is lost. + assert person[LIVE].tolist() == [True, True, True] + assert person["wic"].iloc[INFANT] > 0 + assert person["wic"].iloc[INFANT] == pytest.approx( + person["wic_if_takes_up"].iloc[INFANT] + ) + assert simulation.output_dataset.metadata["legacy_input_renames"] == {} + assert simulation.release_bundle["legacy_input_renames"] == {} + + +def test_run_over_a_region_keeps_a_stored_false_draw(tmp_path): + """Regional runs simulate a scoped copy of the data; it keeps the draw.""" + from policyengine.core.scoping_strategy import RowFilterStrategy + + simulation = pe.Simulation( + dataset=_in_memory_dataset(tmp_path), + tax_benefit_model_version=pe.us.model, + scoping_strategy=RowFilterStrategy( + variable_name="state_code", variable_value="CA" + ), + extra_variables={"person": ["is_wic_eligible", "wic_if_takes_up", LIVE, "wic"]}, + ) + simulation.run() + person = _person_outputs(simulation) + + _assert_children_are_wic_eligible( + person["is_wic_eligible"].to_numpy(), person["wic_if_takes_up"].to_numpy() + ) + assert person[LIVE].tolist() == DRAW + assert person["wic"].iloc[INFANT] == 0 + assert simulation.release_bundle["legacy_input_renames"] == RENAME + + +def test_run_leaves_data_that_stores_the_live_name_alone(tmp_path): + frames = _frames() + # A re-cut release: the live name, with a draw that differs from the + # legacy column, so any mapping would be visible. + frames["person"][LIVE] = [False, True, False] + simulation = _run(_in_memory_dataset(tmp_path, frames)) + person = _person_outputs(simulation) + + assert person[LIVE].tolist() == [False, True, False] + assert person["wic"].iloc[INFANT] > 0 + assert person["wic"].iloc[TODDLER] == 0 + assert simulation.output_dataset.metadata["legacy_input_renames"] == {} + + +# --- Load path 2: managed_microsimulation() over the country loader ---- + + +def _check_managed(microsim): + for month in MONTHS: + np.testing.assert_array_equal(microsim.calculate(LIVE, month).values, DRAW) + eligible = microsim.calculate("is_wic_eligible", MONTHS[0]).values + wic_if_takes_up = microsim.calculate("wic_if_takes_up", YEAR).values + _assert_children_are_wic_eligible(eligible, wic_if_takes_up) + wic = microsim.calculate("wic", YEAR).values + assert wic[INFANT] == 0 + assert wic[TODDLER] == pytest.approx(wic_if_takes_up[TODDLER]) + + +def test_managed_entity_table_file_keeps_a_stored_false_draw(mapped_managed): + _check_managed(mapped_managed) + # policyengine-us extends the stored year forward; each year is mapped. + np.testing.assert_array_equal( + mapped_managed.calculate(LIVE, "2026-07").values, DRAW + ) + assert mapped_managed.policyengine_bundle["legacy_input_renames"] == RENAME + assert LIVE in mapped_managed.input_variables + + +def test_managed_mapping_is_idempotent(mapped_managed): + before = { + month: mapped_managed.calculate(LIVE, month).values.copy() for month in MONTHS + } + + assert apply_legacy_input_renames_to_microsimulation(mapped_managed) == RENAME + + for month in MONTHS: + np.testing.assert_array_equal( + mapped_managed.calculate(LIVE, month).values, before[month] + ) + assert mapped_managed.input_variables.count(LIVE) == 1 + _check_managed(mapped_managed) + + +def test_managed_person_table_out_of_order_is_refused(mapped_managed): + stored = mapped_managed.dataset.datasets[YEAR].person + shuffled = stored.iloc[::-1].reset_index(drop=True) + shuffled[LEGACY] = True + before = mapped_managed.calculate(LIVE, MONTHS[0]).values.copy() + + with pytest.raises(ValueError, match="not in the simulation's person order"): + apply_legacy_input_renames(mapped_managed, {YEAR: {"person": shuffled}}) + + np.testing.assert_array_equal( + mapped_managed.calculate(LIVE, MONTHS[0]).values, before + ) + + +def test_managed_without_the_mapping_gives_every_eligible_person_wic( + tmp_path, monkeypatch +): + monkeypatch.setattr(legacy_inputs, "LEGACY_INPUT_RENAMES", {}) + path = _write_entity_tables(tmp_path / "populace_layout.h5") + microsim = pe.us.managed_microsimulation(dataset=path, allow_unmanaged=True) + + eligible = microsim.calculate("is_wic_eligible", MONTHS[0]).values + wic_if_takes_up = microsim.calculate("wic_if_takes_up", YEAR).values + _assert_children_are_wic_eligible(eligible, wic_if_takes_up) + wic = microsim.calculate("wic", YEAR).values + assert wic[INFANT] > 0 + assert wic[INFANT] == pytest.approx(wic_if_takes_up[INFANT]) + assert microsim.policyengine_bundle["legacy_input_renames"] == {} + + +def test_managed_reform_maps_its_baseline_too(tmp_path): + path = _write_entity_tables(tmp_path / "populace_layout.h5") + microsim = pe.us.managed_microsimulation( + dataset=path, + allow_unmanaged=True, + reform={ + "gov.irs.credits.ctc.amount.base[0].amount": { + "2024-01-01.2100-12-31": 3_000 + } + }, + ) + + # The baseline branch exists before the mapping runs, so it must be + # mapped as well or reform-minus-baseline would include WIC. + assert microsim.baseline is not None + _check_managed(microsim) + _check_managed(microsim.baseline) + + +def test_managed_variable_centric_file_keeps_a_stored_false_draw(tmp_path): + path = _write_variable_centric(tmp_path / "core_layout.h5") + microsim = pe.us.managed_microsimulation(dataset=path, allow_unmanaged=True) + + _check_managed(microsim) + assert microsim.policyengine_bundle["legacy_input_renames"] == RENAME + + +@pytest.fixture(scope="module") +def engine_microsim(tmp_path_factory): + """A real managed Microsimulation the property below maps draws onto.""" + path = _write_entity_tables( + tmp_path_factory.mktemp("engine_microsim") / "populace_layout.h5" + ) + return pe.us.managed_microsimulation(dataset=path, allow_unmanaged=True) + + +@settings(max_examples=25, deadline=None) +@given(data=st.data()) +def test_real_engine_takes_up_any_stored_draw_every_month(engine_microsim, data): + """Draw preserved and idempotent, on the real policyengine-us engine. + + For any stored draw and any set of the dataset's years, the engine's + take-up input equals the draw in every month of those years, and + mapping it again changes nothing. + """ + dataset_years = sorted(engine_microsim.dataset.datasets) + years = data.draw( + st.sets(st.sampled_from(dataset_years), min_size=1, max_size=3), label="years" + ) + draws = { + year: data.draw( + st.lists(st.booleans(), min_size=3, max_size=3), label=f"draw {year}" + ) + for year in sorted(years) + } + tables = { + year: {"person": pd.DataFrame({"person_id": PERSON_IDS, LEGACY: draw})} + for year, draw in draws.items() + } + + for _ in range(2): + assert apply_legacy_input_renames(engine_microsim, tables) == RENAME + for year, draw in draws.items(): + for month in range(1, 13): + # ``get_array`` reads the input the engine holds without + # calculating anything. Calculating instead raised this + # test's peak memory by about 1 GB for each new year. + stored = engine_microsim.get_array(LIVE, f"{year}-{month:02d}") + assert stored.dtype == bool + np.testing.assert_array_equal(stored, draw) + + +# --- Both load paths agree ---------------------------------------------- + + +def test_both_load_paths_give_the_same_take_up_and_wic(mapped_run, mapped_managed): + """Differential: the two US load paths map the same draw identically.""" + run_person = _person_outputs(mapped_run).loc[PERSON_IDS] + managed_ids = mapped_managed.calculate("person_id", YEAR).values.tolist() + assert managed_ids == PERSON_IDS + + np.testing.assert_array_equal( + run_person[LIVE].to_numpy(), mapped_managed.calculate(LIVE, YEAR).values + ) + np.testing.assert_allclose( + run_person["wic"].to_numpy(), mapped_managed.calculate("wic", YEAR).values + ) + + +# --- create_datasets(): extraction for ensure_datasets() ---------------- + + +def test_create_datasets_extracts_the_draw_under_the_live_name(tmp_path): + path = _write_entity_tables(tmp_path / "populace_layout.h5") + created = create_datasets( + datasets=[path], + years=[YEAR], + data_folder=str(tmp_path / "data"), + allow_unmanaged=True, + ) + (dataset,) = created.values() + person = pd.DataFrame(dataset.data.person) + + assert person[LIVE].tolist() == DRAW + assert LEGACY not in person.columns + assert dataset.metadata["legacy_input_renames"] == RENAME + + # The extracted data now carries the live name, so a run reads it + # natively and the mapping has nothing left to do. + simulation = _run(dataset) + outputs = _person_outputs(simulation) + assert outputs["wic"].iloc[INFANT] == 0 + assert outputs["wic"].iloc[TODDLER] > 0 + assert simulation.output_dataset.metadata["legacy_input_renames"] == {} + + +# --- Saved outputs and run records keep the record ---------------------- + + +def _saved_run(directory, simulation_id): + simulation = pe.Simulation( + id=simulation_id, + dataset=_in_memory_dataset(directory), + tax_benefit_model_version=pe.us.model, + extra_variables={"person": [LIVE, "wic"]}, + ) + simulation.run() + simulation.save() + return simulation + + +def _reloaded(simulation): + restored = pe.Simulation( + id=simulation.id, + dataset=simulation.dataset, + tax_benefit_model_version=pe.us.model, + extra_variables=simulation.extra_variables, + ) + restored.load() + return restored + + +def test_a_saved_output_keeps_the_renames_it_applied(tmp_path): + simulation = _saved_run(tmp_path, "legacy-wic-saved") + + restored = _reloaded(simulation) + + assert restored.output_dataset.metadata["legacy_input_renames"] == RENAME + assert restored.release_bundle["legacy_input_renames"] == RENAME + assert _person_outputs(restored)["wic"].iloc[INFANT] == 0 + + +def test_an_output_saved_before_the_mapping_is_not_reused(tmp_path): + """A saved output without the record predates the fix, so its WIC is wrong. + + ``load()`` refuses it, and ``ensure()`` runs the simulation again. + """ + from policyengine.core.simulation import _cache + from policyengine.tax_benefit_models.us.legacy_inputs import RENAMES_H5_DATASET + + simulation = _saved_run(tmp_path, "legacy-wic-stale") + path = simulation.output_dataset.filepath + # Make the file look like one saved before the mapping existed: no + # record, and the full-take-up WIC every eligible person got then. + with h5py.File(path, "a") as stream: + del stream[RENAMES_H5_DATASET] + stale = pd.DataFrame(pd.read_hdf(path, "person")) + stale["wic"] = stale["wic"].where(stale["person_id"] != 2, 999.0) + stale.to_hdf(path, key="person", format="fixed") + + with pytest.raises(ValueError, match="predates the mapping"): + _reloaded(simulation) + + _cache.clear() + rerun = pe.Simulation( + id=simulation.id, + dataset=simulation.dataset, + tax_benefit_model_version=pe.us.model, + extra_variables=simulation.extra_variables, + ) + rerun.ensure() + assert _person_outputs(rerun)["wic"].iloc[INFANT] == 0 + assert rerun.release_bundle["legacy_input_renames"] == RENAME + # The recomputed output was saved with its record, so it loads again. + assert _reloaded(rerun).release_bundle["legacy_input_renames"] == RENAME + _cache.clear() + + +def test_an_output_without_the_record_is_not_saved(tmp_path): + simulation = _run(_in_memory_dataset(tmp_path)) + del simulation.output_dataset.metadata["legacy_input_renames"] + + with pytest.raises(ValueError, match="run again before saving"): + simulation.save() + + +def test_the_run_record_binds_the_renames_applied(tmp_path): + from policyengine.core.run_record import build_simulation_run_record_payloads + + dataset = _in_memory_dataset(tmp_path) + # A run record binds the bytes of the input and output files. + dataset.save() + simulation = _run(dataset) + simulation.save() + + payloads = build_simulation_run_record_payloads(simulation) + + assert payloads["results"]["legacy_input_renames"] == RENAME diff --git a/tests/test_us_native_alignment.py b/tests/test_us_native_alignment.py index 59b39f01..a5573f27 100644 --- a/tests/test_us_native_alignment.py +++ b/tests/test_us_native_alignment.py @@ -119,8 +119,9 @@ def test_native_inputs_and_calculated_outputs_align_by_id( build = PolicyEngineUSLatest._build_simulation_from_dataset def capture(self, country_simulation, dataset, system): - build(self, country_simulation, dataset, system) + applied = build(self, country_simulation, dataset, system) country_simulations.append(country_simulation) + return applied monkeypatch.setattr(PolicyEngineUSLatest, "_build_simulation_from_dataset", capture) simulation = pe.Simulation(dataset=dataset, tax_benefit_model_version=pe.us.model) diff --git a/uv.lock b/uv.lock index 2df7f127..7a86d360 100644 --- a/uv.lock +++ b/uv.lock @@ -703,6 +703,94 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/39/7b/bb06b061991107cd8783f300adff3e7b7f284e330fd82f507f2a1417b11d/huggingface_hub-0.34.4-py3-none-any.whl", hash = "sha256:9b365d781739c93ff90c359844221beef048403f1bc1f1c123c191257c3c890a", size = 561452, upload-time = "2025-08-08T09:14:50.159Z" }, ] +[[package]] +name = "hypothesis" +version = "6.168.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "sortedcontainers" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/aa/31/b54cd138ee36a32d0047d40fe7bdcb63cefd57ce098a26024161332d15fd/hypothesis-6.168.1.tar.gz", hash = "sha256:fd8acb5c67f260dfbdcc1b2dc95ff48b2b3f90708ae595c76f591bbf3f6f8eeb", size = 510823, upload-time = "2026-09-23T04:36:04.09Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/48/95/3091b9ad24baa9666d49feea433a8c4060ad47245f0d529d736af3672ddd/hypothesis-6.168.1-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:5d9cc2d2a4779c92788389564fc6760df7c2909185f6431378133dea6933194f", size = 791345, upload-time = "2026-09-23T04:35:39.379Z" }, + { url = "https://files.pythonhosted.org/packages/c5/b3/b954ffa69f91d111f801030c28d6721f422222af245a846571fc87874242/hypothesis-6.168.1-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d0e1ec15d32af11da6784d2506de0c901e4c1abf97a3cdcfc144225457783f4e", size = 787104, upload-time = "2026-09-23T04:35:21.799Z" }, + { url = "https://files.pythonhosted.org/packages/e5/dc/732df08f570d845a480d3729bfb46beccc441bfabe2804bb3a458649be73/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4cc38519a57c4b919b2c59714a67501d20b46b0c96a0078cbdc9f86e728efce3", size = 1123848, upload-time = "2026-09-23T04:33:48.857Z" }, + { url = "https://files.pythonhosted.org/packages/88/06/e1e937df42243267ba25ede5e9b29921186f290f6c0515729481cc41726e/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:7e229e937e761830c7fa6437fb12cb801616a1737af0898abf5edba57d2ad691", size = 1147698, upload-time = "2026-09-23T04:35:33.439Z" }, + { url = "https://files.pythonhosted.org/packages/be/d1/3b41c008dc2a49e43bc4893dcd27ea9d9fef64fef50382ca5f7107680b1f/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:569ba21e5543840895d17633cc6716a9578ac83b1193a2978b3f24cc16b93085", size = 1149280, upload-time = "2026-09-23T04:35:45.514Z" }, + { url = "https://files.pythonhosted.org/packages/71/ea/d54781058fe6a487b81ec6c636f3806fff6a417b48cf3fcd2b003d80f3ce/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:0a232665716607c3c50c9a3aeeac5929be1d4e58703f5e7cb51db7099445a8c6", size = 1191693, upload-time = "2026-09-23T04:35:57.896Z" }, + { url = "https://files.pythonhosted.org/packages/4f/f2/43bdd07450978611d25989ffdcbe1579acd46c1500a484f8838359fe7880/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9022fe330d8839259da9b496573c4b83ea248bf32029eba23ff589f3c25754bf", size = 1169725, upload-time = "2026-09-23T04:34:32.59Z" }, + { url = "https://files.pythonhosted.org/packages/d1/72/fd4b657cba66940f208bc0eccaceaa777553009d26e86068ff3ab5e55b0b/hypothesis-6.168.1-cp310-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:69be660b831d55ab35619b30abd311a6cb8cd0068a1192210d194a776b211094", size = 1129181, upload-time = "2026-09-23T04:35:27.495Z" }, + { url = "https://files.pythonhosted.org/packages/d6/3f/516a29d7a2bd31e28972de635350eab086c94faa07afcdd3ad0f06e70db8/hypothesis-6.168.1-cp310-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:42e2cccca80ee4ffe8f66edf5b2adb2a16b8cd3c5110b3a774d70f2fc75a42aa", size = 1160162, upload-time = "2026-09-23T04:35:29.333Z" }, + { url = "https://files.pythonhosted.org/packages/54/09/e2c7e32f281b31d4c19cc2fa935bfdde5b31ad4848a485e5ebb8de2389da/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:11cbd2a2539194f0960de4ed9f980882e3a3f55c476afcc6f4cc342eec5a07d2", size = 1299714, upload-time = "2026-09-23T04:35:12.529Z" }, + { url = "https://files.pythonhosted.org/packages/ba/bc/85850bdb1fc35d0fa9e299ca3b2e354f403959666c7ef4d0bfda9c308eee/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:aa1620111616c660a1db03347b1790b5914ef30f86c6e8ddd6c765785291f0b4", size = 1425335, upload-time = "2026-09-23T04:35:23.557Z" }, + { url = "https://files.pythonhosted.org/packages/40/c0/66bdb24f32d32aa5636655b6037af07962118b89fae16ee74bc9a1bb982e/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:a07c8665c95bb856f99ecd60b6f50d94dcf8157415387c32dcd5bb5b5061d032", size = 1376916, upload-time = "2026-09-23T04:34:59.742Z" }, + { url = "https://files.pythonhosted.org/packages/31/f0/7ad0b17bda81c473cbc539009148f3e0950880c4618028691bcf62eeb8d2/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:8a50b6335d5fa2312dd9f5d50faeb953e61475b301d2c073cd7a4c5efd7e6171", size = 1280975, upload-time = "2026-09-23T04:34:21.966Z" }, + { url = "https://files.pythonhosted.org/packages/e2/01/8d04d1eecf03a6097e26400e8393eff7a81fd52c877a664ee4e97adf7880/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:4ccb0276db8ad197a63dff25c4868e11123346347df6900625d2b1f794ce7c19", size = 1300238, upload-time = "2026-09-23T04:34:52.825Z" }, + { url = "https://files.pythonhosted.org/packages/f5/74/eda0c70b3845789c713fe2ac72d3d4586da83f38a34f1f7aefbda0b5b01e/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:85ca7788574e06d0591d51e19ecbb1d09bc82d50c6eae5de0fe8317590b1e138", size = 1336088, upload-time = "2026-09-23T04:33:42.002Z" }, + { url = "https://files.pythonhosted.org/packages/80/c9/43d8528e6d43eebb7688d36e530d1fc7763809b8f923b62cc8b4d95f99ff/hypothesis-6.168.1-cp310-abi3-win32.whl", hash = "sha256:77378e04aa47a22a615af80f42970dc44782f02b8a82ac105b619a2cc7f5eee2", size = 678004, upload-time = "2026-09-23T04:34:29.564Z" }, + { url = "https://files.pythonhosted.org/packages/8c/8b/f7d4de5b57972f7f68a6c7fb5698f60425589eae6fcb175f285f2525e8c0/hypothesis-6.168.1-cp310-abi3-win_amd64.whl", hash = "sha256:f94d3dbc31112f8fbcb3af19a662ee5e90e380e6ae676357903caa14a1604763", size = 684722, upload-time = "2026-09-23T04:34:41.279Z" }, + { url = "https://files.pythonhosted.org/packages/a5/fd/528eb221a82da0aa477e34c7af6ed2b4a5109b326638d4418b9e49435e96/hypothesis-6.168.1-cp310-abi3-win_arm64.whl", hash = "sha256:9bd7e8c6e2c3f6a5befee6b03ebd2170da2e50658075c6decad184777c72a726", size = 682711, upload-time = "2026-09-23T04:33:46.085Z" }, + { url = "https://files.pythonhosted.org/packages/6d/1b/620484ba32fbb655ccbd02f3c510e7a961e840a466c3672ef0b877ab8716/hypothesis-6.168.1-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:9fd235c4ccfefe18fd97c181bdf247187cb3d4fdc8800a4d623322234ce4cfd3", size = 792049, upload-time = "2026-09-23T04:34:28.21Z" }, + { url = "https://files.pythonhosted.org/packages/03/df/2ad250c2c4ee4d5f43d6d9bc5d852e7b81c771e4c8946212931b8cafe849/hypothesis-6.168.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:0722212cf825d001a0d3b4db9303cc31af467f6814b69acab91d91c3f629c95a", size = 787932, upload-time = "2026-09-23T04:35:20.079Z" }, + { url = "https://files.pythonhosted.org/packages/24/69/4f8e8bf1d5ad81352a05aa0835baef82e691e3e43933bafe6d6f2c171763/hypothesis-6.168.1-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:94a1053575f133ff6978876fd4b4c2a153c1111f1b5f14d6cfb355c306fc4cb5", size = 1124000, upload-time = "2026-09-23T04:34:14.106Z" }, + { url = "https://files.pythonhosted.org/packages/b6/e7/b72a34e0c6b434d3ad2b2e0040fa88d06a15a7880c59bddacf1e27d4902a/hypothesis-6.168.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:6d2c74d496304574cfed537e4f657620be7fee3e287381c0aa4b481211e6ec92", size = 1170172, upload-time = "2026-09-23T04:34:30.93Z" }, + { url = "https://files.pythonhosted.org/packages/b6/6c/afe1b73b58479ba0d75e5408259978d580e96fee812eb4ea2bb83a76f7ff/hypothesis-6.168.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:cf4cc6819ba9689effba38dce98a3025554669f76be07e47cc1002e93e8c4918", size = 1300163, upload-time = "2026-09-23T04:35:53.549Z" }, + { url = "https://files.pythonhosted.org/packages/b0/3d/aad54ffe7942c06da79c719fc98daa1dffdd8ac2d384ca64cc3611b1a833/hypothesis-6.168.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:7888e0876e1d93fb92a6a4869834b0d0de18f07f6600edbf3d1cad16cf089357", size = 1336274, upload-time = "2026-09-23T04:35:10.655Z" }, + { url = "https://files.pythonhosted.org/packages/0f/cb/8d2a3474e7921cfb5e59f933ac6fe1ad83b95cc98a871af8f0a68dcf946e/hypothesis-6.168.1-cp311-cp311-win_amd64.whl", hash = "sha256:4aa312b6447743a72283698292ea0e1c4d684279b38dc7f270729c4efda6c27a", size = 684479, upload-time = "2026-09-23T04:33:59.307Z" }, + { url = "https://files.pythonhosted.org/packages/7e/6f/9d5ae55d81e10f0fc46868c19ab9d468e9959e82ff8cc8c26990b16114f3/hypothesis-6.168.1-cp312-cp312-macosx_10_12_x86_64.whl", hash = "sha256:dfe14a88b1ed03ce47333d800cc37949da73feb0cd9767e7ef076d2202768928", size = 793124, upload-time = "2026-09-23T04:33:52.589Z" }, + { url = "https://files.pythonhosted.org/packages/14/11/54c19d259d7f92176aa2199c90f7951e4e396a26898a596dc875ccb9c389/hypothesis-6.168.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:8acd000c3bf233cdcd28c1ff03f1bf56cb03821b1c742d00d57978b20fc6e6f8", size = 784674, upload-time = "2026-09-23T04:34:38.137Z" }, + { url = "https://files.pythonhosted.org/packages/5f/ae/f80a9810f88bd654ab0d0ed0652c274e4c0384e22d4f9442b6753b9200a6/hypothesis-6.168.1-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:29280d4743546ddcf74644f9acdc5726972cba9182699a86b7e77f3f8f6e5dde", size = 1122860, upload-time = "2026-09-23T04:33:51.375Z" }, + { url = "https://files.pythonhosted.org/packages/f4/5d/aa5b7ffd57ec475a9252c55985c07a366823c3f85267d44648e94c1ec321/hypothesis-6.168.1-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:bd9a488d46b8c7da3103058c7569376a3a08994aee9a7d6f1d160d3d3b49b82d", size = 1168922, upload-time = "2026-09-23T04:34:44.567Z" }, + { url = "https://files.pythonhosted.org/packages/65/10/4862448192d4c9cc61a3b8a620730800f56b1e83b0bfda63dbc728526eb3/hypothesis-6.168.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:451ee6c42955887a9e1850c53813e31182eab1663e4d3dacea4e535dedc45b5d", size = 1298568, upload-time = "2026-09-23T04:35:31.191Z" }, + { url = "https://files.pythonhosted.org/packages/15/ad/55e73fda9afbfcf3336b2eee564833bc9a4fee1f7a32fe5f9d183742da39/hypothesis-6.168.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:bc2c036f4a9534af64977e38b2976295136f83417fdb95bed816a52cc3023497", size = 1335091, upload-time = "2026-09-23T04:34:23.42Z" }, + { url = "https://files.pythonhosted.org/packages/ef/99/b6340fa21a89b0088ed023fe984109e1c47a07dfd741fd6d8e17a9e2795f/hypothesis-6.168.1-cp312-cp312-win_amd64.whl", hash = "sha256:d603d7c96d030a63701b6ad9bff0bc4870e4c226335b3db509f1eed1f7fc061b", size = 682011, upload-time = "2026-09-23T04:34:34.565Z" }, + { url = "https://files.pythonhosted.org/packages/f5/23/093b9dc768e89d2cec74d29115f4004fe1be64218bac65f8d817d727d30b/hypothesis-6.168.1-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:aebf7e9d7ba9f920e154abc65851d4affab497daeaea285474bed93840a983ea", size = 793062, upload-time = "2026-09-23T04:33:58.177Z" }, + { url = "https://files.pythonhosted.org/packages/df/d7/f191686b44b3f892ae1404e238a55260152a645a2deafe8b2949c7e90fe9/hypothesis-6.168.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:b26616604cac458dd83e75900fefd9190f1074e67d5d9dc4d9e321d30f295eca", size = 784518, upload-time = "2026-09-23T04:34:02.012Z" }, + { url = "https://files.pythonhosted.org/packages/10/25/b7a5dca1911c9012a564d0b7aaeb44e393ade9424a695dea179ff32d79fe/hypothesis-6.168.1-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:cb6d89cff2edcd45e5af957c7b6a71ae7887419561f7b9c7055e1c50dea41017", size = 1122844, upload-time = "2026-09-23T04:35:06.897Z" }, + { url = "https://files.pythonhosted.org/packages/ea/60/ddacd0d244e6719ea9a9d322d4d4b6b054fd27ada94ec7aa7dce7ef11297/hypothesis-6.168.1-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:2a4ce6310ec21000446ae75cdc6667b81d4ee0f2d226a0dfaf8baca42dd151d0", size = 1168779, upload-time = "2026-09-23T04:33:56.885Z" }, + { url = "https://files.pythonhosted.org/packages/dc/07/343bab6676a41f38cb7c5912ab3ca059258bf40e41c181d1b77cc7251801/hypothesis-6.168.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:a22c6e730c327d22a4d61b53f66639cdf6f0fe4a5a1b3595c36533e5b8127169", size = 1298434, upload-time = "2026-09-23T04:34:48.576Z" }, + { url = "https://files.pythonhosted.org/packages/7d/bd/4b72fafa5f46a455024a9242b62505b1a53fad95ace162e831c4ffb233b9/hypothesis-6.168.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:dcd66493fe180d38641b02dd0456cc1e429655f8940dd1396ab874808f16dd34", size = 1334975, upload-time = "2026-09-23T04:35:01.531Z" }, + { url = "https://files.pythonhosted.org/packages/3f/ae/48643915cda400ca2dccb9bcdc363690b13bcd5c179172bc6450c3d76ce8/hypothesis-6.168.1-cp313-cp313-win_amd64.whl", hash = "sha256:a8a96f2e9fa2a57f95865aaf092943775e7ea96f72535371b2be20fe5464298c", size = 681978, upload-time = "2026-09-23T04:36:00.056Z" }, + { url = "https://files.pythonhosted.org/packages/0c/0e/e6999cddbaef975ecc1b0a7e9028a3d1a1014b542fa003889d0cb3870fa9/hypothesis-6.168.1-cp314-cp314-macosx_10_12_x86_64.whl", hash = "sha256:35b6582ecc22ed682ea39bbf972f8b8b1798043a3bc624ae1dce02d42fcf3ad4", size = 793097, upload-time = "2026-09-23T04:35:03.374Z" }, + { url = "https://files.pythonhosted.org/packages/d5/f3/2ff6bf4bf43ab12641b163bd8fea4b6cddc9a0f07499c4779519099034c9/hypothesis-6.168.1-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:48e467c73b7fb46bc072323c5d67e8e683cf32218fbba47c4e6ee0fe0d72687f", size = 784662, upload-time = "2026-09-23T04:35:08.976Z" }, + { url = "https://files.pythonhosted.org/packages/10/3a/b3653de877981cdab6861476c3b68d02b7c682c2875557cff7b92dbde068/hypothesis-6.168.1-cp314-cp314-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:9c82fcc3dece367bd43f3eca5c2746cf60b73f61bba071ef9421d4cfd6188a85", size = 1123106, upload-time = "2026-09-23T04:34:26.476Z" }, + { url = "https://files.pythonhosted.org/packages/49/25/24c725cb94c8953a98d8bdec83971f76c8f4b103070df1fdacd8b7f37253/hypothesis-6.168.1-cp314-cp314-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:36031c8f8fd4bec7185cf70ed080ba54d45c106cf4f575bc8c2482c01f5f9b21", size = 1168942, upload-time = "2026-09-23T04:35:41.237Z" }, + { url = "https://files.pythonhosted.org/packages/1d/e4/cbe45b9e78e4dd9c18ed2328ff95f659f59235b04ae3c73cf7e9c498adc9/hypothesis-6.168.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:ddfe21791579fdc90423973c733578efa23b010cbfb685dcd5201cf7b0a928b6", size = 1298937, upload-time = "2026-09-23T04:33:50.181Z" }, + { url = "https://files.pythonhosted.org/packages/93/e9/cacaa3f58ee6eb98ce29a2182483d57686011bba35c8fb8a9bc8bf981c7f/hypothesis-6.168.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:ecf86b1c9c8700f6d45324becbbe25dd581c69dc03e1357a64f4bf9a1afe823f", size = 1335198, upload-time = "2026-09-23T04:36:02.021Z" }, + { url = "https://files.pythonhosted.org/packages/31/d8/ca51626959f59a557656cf8024d9afda4d213257c170412fa3193d11e11f/hypothesis-6.168.1-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:d5ab336a7dd233858bc3066a69479e295da788cef345d7e4efd7b77dad724db0", size = 624105, upload-time = "2026-09-23T04:34:08.466Z" }, + { url = "https://files.pythonhosted.org/packages/83/61/6ce96e7d58fc7a592ad7c84d8c7cf9ea36db51468b30ac8306bbf4c3e6ba/hypothesis-6.168.1-cp314-cp314-win_amd64.whl", hash = "sha256:63244ba76464202d3d24126d2e4c2c1abf36db60fffe16ebaa93292572c2a42a", size = 681859, upload-time = "2026-09-23T04:34:50.801Z" }, + { url = "https://files.pythonhosted.org/packages/ff/b1/6c7345e2b37ea8a1df2520a0ecfbdc0243a897895ff950309ca7f687429a/hypothesis-6.168.1-cp314-cp314t-macosx_10_12_x86_64.whl", hash = "sha256:6cec76c8cf76738ad4c0c0c5d6e4632a32d74748639ea6ce26fa672490d9bf95", size = 791685, upload-time = "2026-09-23T04:33:54.243Z" }, + { url = "https://files.pythonhosted.org/packages/72/8c/4f6cfafde4b31add263d6a562f615939cd49649e87a3695ac8693506c811/hypothesis-6.168.1-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:9da10ba2dbbcaba93c6a11eb394fd1a0cfee6e2ba21345c3d6f2fc7401d80670", size = 783237, upload-time = "2026-09-23T04:34:25.049Z" }, + { url = "https://files.pythonhosted.org/packages/a0/25/646c301a9d1fe3825a2ff8362594c4aee0ceb2ce192f49469f7f2824a742/hypothesis-6.168.1-cp314-cp314t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:3218c6d18a041c0eba0cccd88b20d74be24687e42a73338abd0a31ec5294c265", size = 1121414, upload-time = "2026-09-23T04:34:09.876Z" }, + { url = "https://files.pythonhosted.org/packages/27/4f/f84ec8434ef765e6a5c4f9942bbb7652be444b23b9ff868da43e96e61e86/hypothesis-6.168.1-cp314-cp314t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:760cad0526a84122bbe2a0157217e16d76d649d333422142853c7f1ba43ca44d", size = 1167536, upload-time = "2026-09-23T04:33:55.443Z" }, + { url = "https://files.pythonhosted.org/packages/38/74/c53334d793b0c54054ffc781cdf79ccae6ee51daf347bd9cfa7e0015cece/hypothesis-6.168.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:50a60b57b485dff68b56f99e4506b7f1b52994f801ee1aaa4cb9125a90bf7834", size = 1297122, upload-time = "2026-09-23T04:34:17.235Z" }, + { url = "https://files.pythonhosted.org/packages/43/e7/b115a1faea6d40bddce6342da703b7536d803d2384fa50d277dc55170c64/hypothesis-6.168.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:68017d433ca5c558c34e3308a6653fc63077749af8bf86c10850cb6bb3cb559f", size = 1334055, upload-time = "2026-09-23T04:35:37.298Z" }, + { url = "https://files.pythonhosted.org/packages/1d/dd/f272620d304dd67ff7815509745e2fb8d2bf58e4e1da4b7842a83f1de2da/hypothesis-6.168.1-cp314-cp314t-win_amd64.whl", hash = "sha256:ff92099a0185ba14bf4bc2111299b456074596199148ae84590eee4d480fac2e", size = 681804, upload-time = "2026-09-23T04:35:35.374Z" }, + { url = "https://files.pythonhosted.org/packages/54/e9/45595951f8bc9bd03f032325e7b176ca990a0c86bf2e09d481e8ee5a5cab/hypothesis-6.168.1-cp315-abi3.abi3t-macosx_10_12_x86_64.whl", hash = "sha256:e9510e33467d586a7700d56b9a6ee65b44a06832c1cab3d270369d7c53381276", size = 791046, upload-time = "2026-09-23T04:35:55.724Z" }, + { url = "https://files.pythonhosted.org/packages/8d/3e/718431d0ab72c4c07454f71040b4de298a63e765f20088557dc1bef0a48b/hypothesis-6.168.1-cp315-abi3.abi3t-macosx_11_0_arm64.whl", hash = "sha256:99970e3a75f11dde0414d3d6724bd8b7a2bd1deeb9a509ec9236c91d41a5cff0", size = 782982, upload-time = "2026-09-23T04:34:06.017Z" }, + { url = "https://files.pythonhosted.org/packages/d7/05/c60db6df1456a70a3c589e42acaa446c22f56ed342c1b8dd04b8e49df8fd/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a5ea78773de5bcc25266e37eb086472e90a85fc81b8d4a9ca8afc15858ee15fb", size = 1120961, upload-time = "2026-09-23T04:33:43.494Z" }, + { url = "https://files.pythonhosted.org/packages/8b/fe/d943250395750ecc3b4ec95b1ab3c8119f6cbad08c1ba9169ffee7d89049/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:13f502ce2423e46dfcd2af38950a86ecf97084d94e721f7f13dd5a70afb36e0a", size = 1143877, upload-time = "2026-09-23T04:34:56.065Z" }, + { url = "https://files.pythonhosted.org/packages/d3/9d/8c1f8e11f20fd81ce40caed7ad6d6a9776e24775070e87f3aa9ba3683272/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:69b4bcaa8b0e753fc8921011d5d10c8a7c3caf814c6a968bc140a08917973fdc", size = 1146503, upload-time = "2026-09-23T04:34:04.791Z" }, + { url = "https://files.pythonhosted.org/packages/71/d8/ac2e9fa653d2f05a13724837007e857d376d81fdf728543ab85f6c7f7350/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:7ed811908d38ca8e05c20ad25ecb61edc1e20a519667d4b668be49dd09e6bf3e", size = 1189216, upload-time = "2026-09-23T04:34:36.271Z" }, + { url = "https://files.pythonhosted.org/packages/2b/7b/640952448781c7df5098880e2e2d6e7702b3a1c0badd4c0ea5fca6d6ab53/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9239c868418088d58ae18d041f9f424bd7a75dd143f209c35fde38770b65f74f", size = 1166919, upload-time = "2026-09-23T04:35:16.498Z" }, + { url = "https://files.pythonhosted.org/packages/b6/de/0d46b330a28f7614850fb1a0564cf88af7c0b29cbd791e9d04ade7f66f13/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_31_riscv64.whl", hash = "sha256:5cac279798ecc92d3301ec4421c0458e22627f8a79fdde8a6c98ab1d4b8811f6", size = 1126652, upload-time = "2026-09-23T04:35:49.432Z" }, + { url = "https://files.pythonhosted.org/packages/92/36/6adeca1724ed4966095406ffe7308fcff338cf9f5b7443d5ea34a01ab6e1/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:6bc660dc14d633641e22e407417d35bf2925331273ed71bc6ab5536dfca5c72d", size = 1155687, upload-time = "2026-09-23T04:35:18.234Z" }, + { url = "https://files.pythonhosted.org/packages/bd/00/e9b7c7e4a0a75d1621be95a126999352e01a16d68b0a59b396695c7b4991/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_aarch64.whl", hash = "sha256:157a29e4d6bcb77ae732cb83d6546beef47265ecac88408fdd5d5b2bd325685d", size = 1296471, upload-time = "2026-09-23T04:34:12.763Z" }, + { url = "https://files.pythonhosted.org/packages/8f/d6/52cb320e9e07cc67f718506617b5900344370394e39f4affada317aab313/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_armv7l.whl", hash = "sha256:b125aac89594ee2a76687b724f057b2423ea83dfee6b8ff59f54289d6eef19a1", size = 1421838, upload-time = "2026-09-23T04:34:20.451Z" }, + { url = "https://files.pythonhosted.org/packages/9b/93/969c33a87cd2865beffebe491294b8115e7e7421c6840f79a67486535bfc/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_i686.whl", hash = "sha256:4b960c9d31b85dbc6a25888897c02bbea5f498844932841612925184fc78d9eb", size = 1373994, upload-time = "2026-09-23T04:34:00.608Z" }, + { url = "https://files.pythonhosted.org/packages/34/c9/5fe4ade4bd23baa2e4cd27668bb8ac10e840ce19fbcfc74116ea529cbaa7/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_ppc64le.whl", hash = "sha256:d4d76f35874471fe9b8e1bb696fb532363ef3d9588161a06d30520f638f15a39", size = 1278216, upload-time = "2026-09-23T04:34:07.253Z" }, + { url = "https://files.pythonhosted.org/packages/a1/17/30b9a1ad975e90700dea74ce8b08ab07b9e58c4d83918422bc8841942c1e/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_riscv64.whl", hash = "sha256:d5147c5f7497d109eed0267c498adaf5cad6e4beeaca6fad6057f2007bae70a7", size = 1297647, upload-time = "2026-09-23T04:35:51.491Z" }, + { url = "https://files.pythonhosted.org/packages/e0/2f/233a9d5adafad6e187ab8f4d9275592cf120b56e63ea1cfb02213c01021e/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_x86_64.whl", hash = "sha256:a909f89758e184c4f86418598de297a2ced5b922979c0f98c5d73ff6bafe38e5", size = 1333790, upload-time = "2026-09-23T04:34:42.828Z" }, + { url = "https://files.pythonhosted.org/packages/10/d7/4c9d4d597f70316d781cd99b600adae090a6f5d675487196e10510a23afe/hypothesis-6.168.1-cp315-abi3.abi3t-win32.whl", hash = "sha256:66b9fb6f02840f8ffa2a4e6359aff6ecc1b2c6630c8a2066771044fa22956c7a", size = 675206, upload-time = "2026-09-23T04:35:14.592Z" }, + { url = "https://files.pythonhosted.org/packages/b4/a8/f9fc0c0dbb9e80238fc7145cb4ce52e0897ea417a23217022911855bccda/hypothesis-6.168.1-cp315-abi3.abi3t-win_amd64.whl", hash = "sha256:9246eaa80044c33a182be661cbed7e6a7314c61359ef501ea9e9efd90d50ca3b", size = 681503, upload-time = "2026-09-23T04:34:11.223Z" }, + { url = "https://files.pythonhosted.org/packages/11/b4/962ae350c10860be37c9abf98c0ef1ef23e4baca3ee7ad77208941e370e5/hypothesis-6.168.1-cp315-abi3.abi3t-win_arm64.whl", hash = "sha256:52e0c804cbf697d4760400c84444a91cbd5dfe7ae0ef1bb3228f74fb5547eb9e", size = 679199, upload-time = "2026-09-23T04:34:03.472Z" }, + { url = "https://files.pythonhosted.org/packages/aa/c5/2bfdcc601b08eb9611bc524d18ca0a8cdbbdd7bd07bc68c38e0196a00668/hypothesis-6.168.1-pp311-pypy311_pp73-macosx_10_12_x86_64.whl", hash = "sha256:91e3700d9c35e184dd253cff2f151f890557242bede75a2535901e9062d42975", size = 792943, upload-time = "2026-09-23T04:35:25.308Z" }, + { url = "https://files.pythonhosted.org/packages/4b/52/36ae5128d9102d73dc578cd7b241c33f5483bd56e05799a2d06043a1ce7d/hypothesis-6.168.1-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:c19bc5da324a5899527ea749522e9ceb47f1eb54772869673abe59ae0b703aef", size = 788802, upload-time = "2026-09-23T04:34:39.655Z" }, + { url = "https://files.pythonhosted.org/packages/23/34/698fc3481757503dc60f07547c29098dce97ac035a1ee134d66d88814281/hypothesis-6.168.1-pp311-pypy311_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4a4c9e46b2f349658d3013a357e1fbf114bc08abced7ee8e91cd6e0bc718b195", size = 1124766, upload-time = "2026-09-23T04:34:57.869Z" }, + { url = "https://files.pythonhosted.org/packages/02/c9/1f871666106d90048027fc6b309ec505f0fb44e4519add73ddaad98bcd9a/hypothesis-6.168.1-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:95a0beee16fcbc0516951f2ef3ef1a479ad11d9726da49840139303b36ebc1d0", size = 1171766, upload-time = "2026-09-23T04:34:54.476Z" }, + { url = "https://files.pythonhosted.org/packages/17/44/eff662526259ac4447dde4abe4fc285d0e8e5ff5345d35af836c6f0f12cc/hypothesis-6.168.1-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:4a8beb513c066fcb187b885fdfa93da4006f223d50d605b687bfb7e60aac33e9", size = 685449, upload-time = "2026-09-23T04:33:44.837Z" }, +] + [[package]] name = "idna" version = "3.10" @@ -1737,6 +1825,7 @@ dev = [ { name = "autodoc-pydantic" }, { name = "build" }, { name = "furo" }, + { name = "hypothesis" }, { name = "itables" }, { name = "jupyter-book" }, { name = "mypy" }, @@ -1780,6 +1869,7 @@ requires-dist = [ { name = "build", marker = "extra == 'dev'" }, { name = "furo", marker = "extra == 'dev'" }, { name = "h5py", specifier = ">=3.0.0" }, + { name = "hypothesis", marker = "extra == 'dev'", specifier = ">=6.100.0" }, { name = "itables", marker = "extra == 'dev'" }, { name = "jsonschema", specifier = ">=4.0.0" }, { name = "jupyter-book", marker = "extra == 'dev'" }, From 7e0ae3073e78c5bc584d084bf23255d32c39a9b4 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 25 Sep 2026 20:32:33 -0400 Subject: [PATCH 2/3] Keep the renames record in US year files and refuse files without it Review of #531 found that year files ensure_datasets or create_datasets wrote under 6.0.0 to 6.1.1 had lost the WIC take-up draw, and that ensure_datasets kept reusing them. - PolicyEngineUSDataset.save() writes the renames record into the file and load() restores it, so create_datasets year files keep it. - ensure_datasets creates year files without the record again, and load_datasets refuses them. - Simulation.run() carries the input dataset's record into its output, so a run over a year file shows the rename applied when it was cut. - Simulation.run() refuses a core H5 that stores the legacy draw for part of a year, as managed_microsimulation already did. - Tests cover the reform baseline on run(), a later core H5 year that stores its own person IDs, the record round trip, stale year files and part-year draws. - The docs say what an empty record does and does not show, and the changelog gains a changed fragment for the files no longer reused. Co-Authored-By: Claude Opus 5.5 --- changelog.d/530.changed.md | 1 + changelog.d/530.fixed.md | 2 +- docs/microsim.md | 37 ++-- .../common/model_version.py | 22 +-- .../tax_benefit_models/us/datasets.py | 72 +++++++- .../tax_benefit_models/us/legacy_inputs.py | 79 ++++++-- .../tax_benefit_models/us/model.py | 8 +- tests/test_us_legacy_inputs.py | 78 ++++++++ tests/test_us_legacy_inputs_integration.py | 172 ++++++++++++++++-- 9 files changed, 404 insertions(+), 67 deletions(-) create mode 100644 changelog.d/530.changed.md diff --git a/changelog.d/530.changed.md b/changelog.d/530.changed.md new file mode 100644 index 00000000..2827e344 --- /dev/null +++ b/changelog.d/530.changed.md @@ -0,0 +1 @@ +US files written before the WIC take-up mapping are no longer reused, because they may have lost the draw or were calculated without it. `Simulation.load()` raises for a saved US output that has no record of renamed stored inputs (`Simulation.ensure()` calculates it again), `load_datasets` raises for such a year file, and `ensure_datasets` creates such year files again. diff --git a/changelog.d/530.fixed.md b/changelog.d/530.fixed.md index d285face..d937ac27 100644 --- a/changelog.d/530.fixed.md +++ b/changelog.d/530.fixed.md @@ -1 +1 @@ -Map the stored US WIC take-up draw `would_claim_wic` onto `takes_up_wic_if_eligible` when loading US data, so policyengine-us 2.x no longer gives WIC to every WIC-eligible person. `Simulation.run()`, `managed_microsimulation` and `create_datasets` apply the mapping and record the renames applied (output dataset metadata, `release_bundle`, saved output files, run records, `policyengine_bundle` and dataset metadata). A stored table that is not in the simulation's person order is refused. The mapping turns itself off once the data stores `takes_up_wic_if_eligible` or the engine defines `would_claim_wic` again. Saved US outputs from earlier releases are no longer loaded, so `ensure()` recomputes them. Year files written by `ensure_datasets` or `create_datasets` before this fix store neither name; delete and regenerate them. +Map the stored US WIC take-up draw `would_claim_wic` onto `takes_up_wic_if_eligible` when loading US data, so policyengine-us 2.x no longer gives WIC to every WIC-eligible person in data that stores the draw under its old name, such as the certified default. `Simulation.run()`, `managed_microsimulation` and `create_datasets` apply the mapping and record the renames applied (output dataset metadata, `release_bundle`, saved output and year files, run records and `policyengine_bundle`). A stored table that is not in the simulation's person order, or that stores the draw for only part of a year, is refused. The mapping turns itself off once the data stores `takes_up_wic_if_eligible` or the engine defines `would_claim_wic` again. diff --git a/docs/microsim.md b/docs/microsim.md index 109e0e95..dcfaffb3 100644 --- a/docs/microsim.md +++ b/docs/microsim.md @@ -340,26 +340,35 @@ the old name but does define the new one, and the data does not already store the new name. The new input is then set from the stored values for every month of every dataset year. The stored table must list the simulation's entity IDs in the simulation's order, or loading fails rather than attach values to the -wrong people. The mapping turns itself off once the data stores the new name, +wrong people. Only values stored for a whole year are mapped, so a file that +stores the old name for part of a year is refused. The mapping turns itself off once the data stores the new name, or if the engine defines the old name again. The renames applied are recorded as `{old: new}` (`{}` when none applied): - `Simulation.run()`: `simulation.output_dataset.metadata["legacy_input_renames"]`, - also shown as `simulation.release_bundle["legacy_input_renames"]`. A US - `save()` writes the record into the output file and `load()` restores it, - and a run record's `results.json` binds it; + also shown as `simulation.release_bundle["legacy_input_renames"]`. It + includes renames applied when the input year file was cut, so a run over a + `create_datasets` year file records the rename even though the file already + stores the new name. A US `save()` writes the record into the output file, + `load()` restores it, and a run record's `results.json` binds it; - `managed_microsimulation`: `sim.policyengine_bundle["legacy_input_renames"]`; -- `create_datasets`: each returned dataset's `metadata["legacy_input_renames"]`. - -A US output file saved before this mapping existed has no record, and its -results were calculated without the mapped inputs. `load()` refuses it, and -`ensure()` runs the simulation again and saves the new output. - -Year files that `ensure_datasets` or `create_datasets` wrote before this -mapping existed store neither name, so the draw is lost from them, and -`ensure_datasets` reuses existing files as they are. Delete and regenerate -them. +- `create_datasets`: each year file, and each returned dataset's + `metadata["legacy_input_renames"]`. + +`{}` means that no stored column was mapped. It does not show that the data +carried the draw: data that stores neither name runs with the new input's +default, which for WIC is full take-up. + +Files written before this mapping existed have no record, and are not reused: + +- A saved US output may have been calculated without the mapped inputs. + `load()` refuses it, and `ensure()` runs the simulation again and saves the new + output. +- A year file that `ensure_datasets` or `create_datasets` wrote stores + neither name, so the draw is lost from it. `ensure_datasets` creates such + year files again, and `load_datasets` refuses them. A year file opened + directly, as `PolicyEngineUSDataset(filepath=...)`, is not checked. ## Pinned model versions diff --git a/src/policyengine/tax_benefit_models/common/model_version.py b/src/policyengine/tax_benefit_models/common/model_version.py index 131b2f5a..576c0c0f 100644 --- a/src/policyengine/tax_benefit_models/common/model_version.py +++ b/src/policyengine/tax_benefit_models/common/model_version.py @@ -356,11 +356,9 @@ def save(self, simulation: Simulation) -> None: "something to persist." ) serialized_spm = None - serialized_renames = None if self.country_code == "us": from policyengine.core.spm import SPMProvenance from policyengine.tax_benefit_models.us.legacy_inputs import ( - RENAMES_H5_DATASET, RENAMES_RECORD_KEY, ) @@ -379,6 +377,8 @@ def save(self, simulation: Simulation) -> None: }, sort_keys=True, ) + # ``PolicyEngineUSDataset.save()`` writes this record into the + # file, and ``load()`` refuses an output without it. renames = (getattr(simulation.output_dataset, "metadata", None) or {}).get( RENAMES_RECORD_KEY ) @@ -388,7 +388,6 @@ def save(self, simulation: Simulation) -> None: "inputs were mapped when it was calculated; run again " "before saving" ) - serialized_renames = json.dumps(dict(renames), sort_keys=True) simulation.output_dataset.save() if serialized_spm is not None: # Store UTF-8 JSON in a dataset rather than an attribute: the @@ -400,12 +399,6 @@ def save(self, simulation: Simulation) -> None: data=serialized_spm, dtype=h5py.string_dtype("utf-8"), ) - if serialized_renames is not None: - stream.create_dataset( - RENAMES_H5_DATASET, - data=serialized_renames, - dtype=h5py.string_dtype("utf-8"), - ) def load(self, simulation: Simulation) -> None: """Rehydrate the simulation's output dataset from disk. @@ -421,8 +414,8 @@ def load(self, simulation: Simulation) -> None: if self.country_code == "us": from policyengine.core.spm import SPMProvenance from policyengine.tax_benefit_models.us.legacy_inputs import ( - RENAMES_H5_DATASET, RENAMES_RECORD_KEY, + read_renames_record, ) with h5py.File(filepath, "r") as stream: @@ -431,11 +424,7 @@ def load(self, simulation: Simulation) -> None: if "policyengine_spm" in stream else None ) - raw_renames = ( - stream[RENAMES_H5_DATASET].asstr()[()] - if RENAMES_H5_DATASET in stream - else None - ) + recorded_renames = read_renames_record(filepath) if raw is None: raise ValueError( "Saved US simulation has no SPM configuration or receipt" @@ -448,12 +437,11 @@ def load(self, simulation: Simulation) -> None: # live inputs were calculated without them (for example with # every WIC-eligible person taking WIC up), so they are not # reused. ``Simulation.ensure()`` runs such a simulation again. - if raw_renames is None: + if recorded_renames is None: raise ValueError( "Saved US simulation predates the mapping of renamed " "stored inputs (it has no record of them); run it again" ) - recorded_renames = json.loads(raw_renames) simulation.output_dataset = self._dataset_class( id=simulation.id, diff --git a/src/policyengine/tax_benefit_models/us/datasets.py b/src/policyengine/tax_benefit_models/us/datasets.py index 6c855969..1612d55d 100644 --- a/src/policyengine/tax_benefit_models/us/datasets.py +++ b/src/policyengine/tax_benefit_models/us/datasets.py @@ -27,7 +27,10 @@ from policyengine.tax_benefit_models.us.legacy_inputs import ( RENAMES_RECORD_KEY, apply_legacy_input_renames_to_microsimulation, + check_yearly_periods, pending_legacy_input_renames, + read_renames_record, + write_renames_record, ) from policyengine.utils.hashing import sha256_file @@ -102,12 +105,22 @@ def save(self) -> None: store["spm_unit"] = pd.DataFrame(self.data.spm_unit) store["tax_unit"] = pd.DataFrame(self.data.tax_unit) store["household"] = pd.DataFrame(self.data.household) + # The renamed stored inputs mapped when this data was cut or + # calculated. Its presence marks a year file or output as written + # with the mapping (see ``legacy_inputs.RENAMES_H5_DATASET``). + renames = self.metadata.get(RENAMES_RECORD_KEY) + if renames is not None: + write_renames_record(filepath, renames) def load(self) -> None: """Load dataset from HDF5 file into this instance.""" filepath = self.filepath - if _is_policyengine_core_h5(Path(filepath)): - self.data = _load_policyengine_core_h5(Path(filepath), self.year) + path = Path(filepath) + renames = read_renames_record(path) + if renames is not None: + self.metadata[RENAMES_RECORD_KEY] = renames + if _is_policyengine_core_h5(path): + self.data = _load_policyengine_core_h5(path, self.year) return with pd.HDFStore(filepath, mode="r") as store: @@ -199,7 +212,8 @@ def _core_h5_entity_lengths(h5_file: h5py.File, year: int) -> dict[str, int]: return lengths -def _core_h5_variable_entities() -> dict[str, str]: +def _core_h5_variable_entities() -> tuple[dict[str, str], set[str]]: + """Return each variable's entity, and the stored legacy names to map.""" from policyengine_us.system import system entities = { @@ -208,9 +222,10 @@ def _core_h5_variable_entities() -> dict[str, str]: # A stored input the engine has since renamed belongs to its live input's # entity. Without this its entity is guessed from its length, and a column # as long as two entities is dropped before the rename can map it. - for legacy, live in pending_legacy_input_renames(system.variables).items(): + pending = pending_legacy_input_renames(system.variables) + for legacy, live in pending.items(): entities[legacy] = entities[live] - return entities + return entities, set(pending) def _validate_entity_ids(data: dict[str, pd.DataFrame]) -> None: @@ -301,11 +316,21 @@ def _load_policyengine_core_h5(path: Path, year: int) -> USYearData: """Load a PolicyEngine core variable/period H5 into .py entity DataFrames.""" data = {entity: pd.DataFrame() for entity in US_ENTITY_KEYS} - variable_entities = _core_h5_variable_entities() + variable_entities, legacy_names = _core_h5_variable_entities() with h5py.File(path, "r") as h5_file: entity_lengths = _core_h5_entity_lengths(h5_file, year) for variable_name in h5_file.keys(): + if variable_name in legacy_names: + # A stored legacy column is mapped onto every month of the + # year, so a part-year value is refused, as it is when + # ``managed_microsimulation`` reads this file. + node = h5_file[variable_name] + check_yearly_periods( + variable_name, + node.keys() if isinstance(node, h5py.Group) else [], + path, + ) values = _read_core_h5_period_values(h5_file, variable_name, year) entity = variable_entities.get(variable_name) if entity is None: @@ -534,6 +559,16 @@ def create_datasets( return result +def _year_file_records_renames(path: Path) -> bool: + """Return whether a year file records the renamed stored inputs mapped. + + ``create_datasets`` writes the record into every year file (``{}`` when + nothing needed mapping). Files written before it did may have lost a + renamed input such as the WIC take-up draw. + """ + return read_renames_record(path) is not None + + def load_datasets( datasets: Optional[list[str]] = None, years: list[int] = [2024, 2025, 2026, 2027, 2028], @@ -541,6 +576,11 @@ def load_datasets( ) -> dict[str, PolicyEngineUSDataset]: """Load PolicyEngineUSDataset instances from saved HDF5 files. + A year file without the record of renamed stored inputs that + ``create_datasets`` writes was cut before those inputs were mapped, and + may have lost one (see ``legacy_inputs.RENAMES_H5_DATASET``), so it is + refused. + Args: datasets: List of HuggingFace dataset paths (used to derive file names) years: List of years to load data for @@ -556,6 +596,17 @@ def load_datasets( dataset_stem = dataset_logical_name(resolved_dataset) for year in years: filepath = f"{data_folder}/{dataset_stem}_year_{year}.h5" + if Path(filepath).exists() and not _year_file_records_renames( + Path(filepath) + ): + raise ValueError( + f"US year file {filepath} has no record of the renamed " + "stored inputs mapped when it was cut, so it was written " + "before policyengine.py mapped them and may have lost " + "the WIC take-up draw (every WIC-eligible person would " + "then take WIC up). Regenerate it with ensure_datasets() " + "or create_datasets()." + ) us_dataset = PolicyEngineUSDataset( name=f"{dataset_stem}-year-{year}", description=f"US Dataset for year {year} based on {dataset_stem}", @@ -1201,6 +1252,11 @@ def ensure_datasets( ) -> dict[str, PolicyEngineUSDataset]: """Ensure datasets exist, loading if available or creating if not. + Year files without the record of renamed stored inputs that + ``create_datasets`` writes were cut before those inputs were mapped, and + may have lost one such as the WIC take-up draw, so they are created + again rather than loaded. + Args: datasets: List of HuggingFace dataset paths years: List of years to load/create data for @@ -1218,7 +1274,9 @@ def ensure_datasets( dataset_stem = dataset_logical_name(resolved_dataset) for year in years: filepath = Path(f"{data_folder}/{dataset_stem}_year_{year}.h5") - if not filepath.exists(): + # A year file written before renamed stored inputs were mapped + # may have lost them, so it is regenerated rather than reused. + if not filepath.exists() or not _year_file_records_renames(filepath): all_exist = False break if not all_exist: diff --git a/src/policyengine/tax_benefit_models/us/legacy_inputs.py b/src/policyengine/tax_benefit_models/us/legacy_inputs.py index 3e71c47c..9558cd6d 100644 --- a/src/policyengine/tax_benefit_models/us/legacy_inputs.py +++ b/src/policyengine/tax_benefit_models/us/legacy_inputs.py @@ -23,6 +23,7 @@ from __future__ import annotations import inspect +import json from collections.abc import Iterable, Iterator, Mapping from pathlib import Path from typing import Any @@ -44,12 +45,22 @@ } #: Key under which a run records the renames it applied, ``{legacy: live}``: -#: in an output dataset's ``metadata``, ``Simulation.release_bundle``, a -#: managed Microsimulation's ``policyengine_bundle`` and a run record's -#: results. +#: in a dataset's ``metadata`` (a ``create_datasets`` year file or a run's +#: output), ``Simulation.release_bundle``, a managed Microsimulation's +#: ``policyengine_bundle`` and a run record's results. +#: +#: ``{}`` means no stored column was mapped. It does not show that the data +#: carried a take-up draw at all: data that stores neither name runs with +#: the live input's default. RENAMES_RECORD_KEY = "legacy_input_renames" -#: H5 dataset holding that record, as UTF-8 JSON, in a saved US output file. +#: H5 dataset holding that record, as UTF-8 JSON, in a US file written by +#: ``PolicyEngineUSDataset.save()`` (a saved output or a ``create_datasets`` +#: year file). Files written before this mapping existed have none, so its +#: absence marks them as calculated or cut without it: ``Simulation.load()`` +#: refuses such an output, ``load_datasets`` refuses such a year file and +#: ``ensure_datasets`` regenerates it. A new register entry would need those +#: checks to tell files written before it apart as well. RENAMES_H5_DATASET = "policyengine_legacy_input_renames" #: ``{year: {entity key: stored table}}``, one table per entity per dataset @@ -73,6 +84,54 @@ def pending_legacy_input_renames(variables: Any) -> dict[str, str]: } +def read_renames_record(path: str | Path) -> dict[str, str] | None: + """Return the renames record stored in an H5 file, or ``None``. + + ``None`` means the file stores no record (see :data:`RENAMES_H5_DATASET`) + or is not an H5 file. + """ + import h5py + + try: + with h5py.File(path, "r") as file: + if RENAMES_H5_DATASET not in file: + return None + raw = file[RENAMES_H5_DATASET].asstr()[()] + except OSError: + return None + return dict(json.loads(raw)) + + +def write_renames_record(path: str | Path, record: Mapping[str, str]) -> None: + """Store a renames record in an H5 file, replacing any it has.""" + import h5py + + with h5py.File(path, "a") as file: + if RENAMES_H5_DATASET in file: + del file[RENAMES_H5_DATASET] + # UTF-8 JSON in a dataset, as for the SPM receipt of a saved output. + file.create_dataset( + RENAMES_H5_DATASET, + data=json.dumps(dict(record), sort_keys=True), + dtype=h5py.string_dtype("utf-8"), + ) + + +def check_yearly_periods(column: str, periods: Iterable[Any], source: Any) -> None: + """Refuse a legacy column stored for any period that is not a year. + + A value stored for part of a year (or for ``ETERNITY``) cannot stand for + every month of a year, so it is not mapped rather than spread over them. + """ + for period in periods: + if not str(period).isdigit(): + raise ValueError( + f"Cannot map stored {column!r} from {source}: it is stored " + f"for period {str(period)!r}, and only yearly periods are " + "supported." + ) + + def apply_legacy_input_renames( simulation, stored_tables: StoredTables ) -> dict[str, str]: @@ -222,15 +281,9 @@ def _variable_centric_tables( first_ids = next(iter(ids_by_period.values()), None) columns_by_year: dict[int, dict[str, np.ndarray]] = {} for column in legacy_columns: - for period, stored in _stored_periods( - file[column], default_period - ).items(): - if not period.isdigit(): - raise ValueError( - f"Cannot map stored {column!r} from {path}: it is " - f"stored for period {period!r}, and only yearly " - "periods are supported." - ) + stored_by_period = _stored_periods(file[column], default_period) + check_yearly_periods(column, stored_by_period, path) + for period, stored in stored_by_period.items(): columns_by_year.setdefault(int(period), {})[column] = stored for year, columns in sorted(columns_by_year.items()): ids = ids_by_period.get(str(year), first_ids) diff --git a/src/policyengine/tax_benefit_models/us/model.py b/src/policyengine/tax_benefit_models/us/model.py index 54525575..f07f1611 100644 --- a/src/policyengine/tax_benefit_models/us/model.py +++ b/src/policyengine/tax_benefit_models/us/model.py @@ -224,7 +224,13 @@ class InputMicrosimulation(Microsimulation): # leaves the module-level one untouched. Building populations # against the module-level system would hide reform-registered # variables like ``ctc_minimum_refundable_amount`` at calc time. - legacy_input_renames: dict[str, str] = {} + # Renames already applied when the input data was cut (a + # ``create_datasets`` year file stores the mapped input under its + # live name) count as applied to this run's inputs too, so the + # output's record keeps the whole chain. + legacy_input_renames: dict[str, str] = dict( + simulation.dataset.metadata.get(RENAMES_RECORD_KEY) or {} + ) if microsim.baseline is not None: legacy_input_renames.update( self._build_simulation_from_dataset( diff --git a/tests/test_us_legacy_inputs.py b/tests/test_us_legacy_inputs.py index f29e915a..b106d928 100644 --- a/tests/test_us_legacy_inputs.py +++ b/tests/test_us_legacy_inputs.py @@ -38,10 +38,14 @@ from policyengine.tax_benefit_models.us import legacy_inputs from policyengine.tax_benefit_models.us.legacy_inputs import ( LEGACY_INPUT_RENAMES, + RENAMES_H5_DATASET, apply_legacy_input_renames, apply_legacy_input_renames_to_microsimulation, + check_yearly_periods, pending_legacy_input_renames, + read_renames_record, stored_entity_tables, + write_renames_record, ) LEGACY = "would_claim_wic" @@ -493,6 +497,32 @@ def test_variable_centric_file_is_read_by_year(tmp_path): assert tables[2025]["person"]["person_id"].tolist() == [5, 6] +def test_a_year_storing_its_own_ids_is_checked_against_them(tmp_path): + """Order check, for a year that stores its own person IDs. + + policyengine-core builds the population from the first period's IDs. + A later year that stores the same people in another order would attach + each value to the wrong person, so it is refused and nothing is set. + """ + path = _write_h5( + tmp_path / "core.h5", + { + "person_id/2024": [5, 6], + "person_id/2025": [6, 5], + f"{LEGACY}/2025": [True, False], + }, + ) + dataset = _core_dataset(path, "time_period_arrays") + + tables = stored_entity_tables(dataset, {"person": {LEGACY: LIVE}}) + assert tables[2025]["person"]["person_id"].tolist() == [6, 5] + + simulation = FakeSimulation(_variables(LIVE), [5, 6], dataset=dataset) + with pytest.raises(ValueError, match="not in the simulation's person order"): + apply_legacy_input_renames_to_microsimulation(simulation) + assert simulation.inputs == {} + + @pytest.mark.parametrize("live_period", ["2024", "2025", "2024-01"]) def test_variable_centric_file_storing_the_live_name_is_not_read(tmp_path, live_period): # Data that stores the live name for any period loads it natively, even @@ -576,3 +606,51 @@ def test_core_dataset_objects_are_probed_without_loading_keys(tmp_path): simulation = FakeSimulation(_variables(LIVE), [5, 6], dataset=dataset) assert apply_legacy_input_renames_to_microsimulation(simulation) == {LEGACY: LIVE} assert simulation.inputs[(LIVE, "2024-01")].tolist() == [True, False] + + +# --- The record a US file keeps ----------------------------------------- + +RECORDS = st.dictionaries(st.text(min_size=1), st.text(min_size=1), max_size=4) + + +@settings(max_examples=50, deadline=None) +@given(record=RECORDS, replaced=RECORDS) +def test_a_stored_record_reads_back_as_written(record, replaced): + """Round trip: a file's record reads back exactly, and a new one replaces it.""" + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "output.h5" + _write_h5(path, {"person_id": [5, 6]}) + assert read_renames_record(path) is None + + write_renames_record(path, record) + assert read_renames_record(path) == record + + write_renames_record(path, replaced) + assert read_renames_record(path) == replaced + with h5py.File(path, "r") as file: + assert file["person_id"][()].tolist() == [5, 6] + + +def test_a_missing_or_foreign_file_has_no_record(tmp_path): + assert read_renames_record(tmp_path / "missing.h5") is None + text = tmp_path / "notes.txt" + text.write_text("not an H5 file") + assert read_renames_record(text) is None + + +def test_the_record_is_stored_as_utf8_json(tmp_path): + path = _write_h5(tmp_path / "output.h5", {"person_id": [5]}) + write_renames_record(path, {LEGACY: LIVE}) + with h5py.File(path, "r") as file: + assert file[RENAMES_H5_DATASET].asstr()[()] == f'{{"{LEGACY}": "{LIVE}"}}' + + +@pytest.mark.parametrize("period", ["2024-01", "2024-01-01", "ETERNITY", "month"]) +def test_a_part_year_period_is_refused(period): + with pytest.raises(ValueError, match="only yearly periods"): + check_yearly_periods(LEGACY, ["2024", period], "source.h5") + + +def test_yearly_periods_are_accepted(): + check_yearly_periods(LEGACY, ["2024", 2025, np.int64(2026)], "source.h5") + check_yearly_periods(LEGACY, [], "source.h5") diff --git a/tests/test_us_legacy_inputs_integration.py b/tests/test_us_legacy_inputs_integration.py index 5d0b7ead..406a9185 100644 --- a/tests/test_us_legacy_inputs_integration.py +++ b/tests/test_us_legacy_inputs_integration.py @@ -30,10 +30,17 @@ PolicyEngineUSDataset, USYearData, create_datasets, + ensure_datasets, + load_datasets, ) from policyengine.tax_benefit_models.us.legacy_inputs import ( # noqa: E402 + RENAMES_H5_DATASET, apply_legacy_input_renames, apply_legacy_input_renames_to_microsimulation, + read_renames_record, +) +from policyengine.tax_benefit_models.us.model import ( # noqa: E402 + PolicyEngineUSLatest, ) YEAR = 2024 @@ -99,15 +106,19 @@ def _write_entity_tables(path) -> str: return str(path) -def _write_variable_centric(path) -> str: - """Write the household as a policyengine-core ``variable/period`` H5.""" +def _write_variable_centric(path, draw_period=YEAR) -> str: + """Write the household as a policyengine-core ``variable/period`` H5. + + The draw is stored for ``draw_period``; every other column for ``YEAR``. + """ with h5py.File(path, "w") as file: for entity, frame in _frames().items(): for name in frame.columns: values = frame[name].to_numpy() if values.dtype.kind in {"O", "U"}: values = np.asarray(values, dtype="S") - file.create_dataset(f"{name}/{YEAR}", data=values) + period = draw_period if name == LEGACY else YEAR + file.create_dataset(f"{name}/{period}", data=values) return str(path) @@ -235,6 +246,42 @@ def test_run_over_a_region_keeps_a_stored_false_draw(tmp_path): assert simulation.release_bundle["legacy_input_renames"] == RENAME +def test_run_maps_the_baseline_of_a_reform_too(tmp_path, monkeypatch): + """A reformed run builds its baseline from the data as well. + + policyengine-us reads baseline incomes from the baseline branch (for + example for labor-supply responses), so an unmapped baseline would give + every eligible person WIC there while the reform keeps the draw. + """ + built = [] + build = PolicyEngineUSLatest._build_simulation_from_dataset + + def capture(self, country_simulation, dataset, system): + applied = build(self, country_simulation, dataset, system) + built.append(country_simulation) + return applied + + monkeypatch.setattr(PolicyEngineUSLatest, "_build_simulation_from_dataset", capture) + simulation = pe.Simulation( + dataset=_in_memory_dataset(tmp_path), + tax_benefit_model_version=pe.us.model, + policy={"gov.irs.credits.ctc.amount.base[0].amount": 3_000}, + extra_variables={"person": [LIVE, "wic"]}, + ) + simulation.run() + + assert len(built) == 2 + baseline, reform = built + assert baseline is reform.baseline + for country_simulation in built: + for month in MONTHS: + np.testing.assert_array_equal( + country_simulation.get_array(LIVE, month), DRAW + ) + assert country_simulation.calculate("wic", YEAR).values[INFANT] == 0 + assert simulation.output_dataset.metadata["legacy_input_renames"] == RENAME + + def test_run_leaves_data_that_stores_the_live_name_alone(tmp_path): frames = _frames() # A re-cut release: the live name, with a draw that differs from the @@ -249,6 +296,24 @@ def test_run_leaves_data_that_stores_the_live_name_alone(tmp_path): assert simulation.output_dataset.metadata["legacy_input_renames"] == {} +def test_a_core_h5_with_a_part_year_draw_is_refused_by_run(tmp_path): + """A draw stored for one month cannot stand for all twelve. + + ``managed_microsimulation`` refuses such a file (see the managed test + below); ``Simulation.run()`` loads it through the dataset loader, which + would otherwise read the first stored month for the whole year. + """ + path = _write_variable_centric(tmp_path / "monthly.h5", draw_period=f"{YEAR}-01") + + with pytest.raises(ValueError, match="only yearly periods"): + PolicyEngineUSDataset( + name="legacy-wic-draw-monthly", + description="Uncertified three-person WIC take-up fixture", + filepath=path, + year=YEAR, + ) + + # --- Load path 2: managed_microsimulation() over the country loader ---- @@ -345,6 +410,13 @@ def test_managed_variable_centric_file_keeps_a_stored_false_draw(tmp_path): assert microsim.policyengine_bundle["legacy_input_renames"] == RENAME +def test_managed_core_h5_with_a_part_year_draw_is_refused(tmp_path): + path = _write_variable_centric(tmp_path / "monthly.h5", draw_period=f"{YEAR}-01") + + with pytest.raises(ValueError, match="only yearly periods"): + pe.us.managed_microsimulation(dataset=path, allow_unmanaged=True) + + @pytest.fixture(scope="module") def engine_microsim(tmp_path_factory): """A real managed Microsimulation the property below maps draws onto.""" @@ -410,28 +482,100 @@ def test_both_load_paths_give_the_same_take_up_and_wic(mapped_run, mapped_manage # --- create_datasets(): extraction for ensure_datasets() ---------------- -def test_create_datasets_extracts_the_draw_under_the_live_name(tmp_path): - path = _write_entity_tables(tmp_path / "populace_layout.h5") +def _create_year_file(directory): + """Cut the household's entity-table file into a year file.""" + source = _write_entity_tables(directory / "populace_layout.h5") created = create_datasets( - datasets=[path], + datasets=[source], years=[YEAR], - data_folder=str(tmp_path / "data"), + data_folder=str(directory / "data"), allow_unmanaged=True, ) (dataset,) = created.values() + return source, dataset + + +def test_create_datasets_extracts_the_draw_under_the_live_name(tmp_path): + _, dataset = _create_year_file(tmp_path) person = pd.DataFrame(dataset.data.person) assert person[LIVE].tolist() == DRAW assert LEGACY not in person.columns assert dataset.metadata["legacy_input_renames"] == RENAME + # The year file keeps the record, so reloading it keeps the chain. + assert read_renames_record(dataset.filepath) == RENAME + reloaded = PolicyEngineUSDataset( + name=dataset.name, + description=dataset.description, + filepath=dataset.filepath, + year=YEAR, + ) + assert reloaded.metadata["legacy_input_renames"] == RENAME + + # The extracted data carries the live name, so a run reads it natively + # and maps nothing itself. Its record still shows the rename applied + # when the year file was cut. + for input_dataset in (dataset, reloaded): + simulation = _run(input_dataset) + outputs = _person_outputs(simulation) + assert outputs[LIVE].tolist() == DRAW + assert outputs["wic"].iloc[INFANT] == 0 + assert outputs["wic"].iloc[TODDLER] > 0 + assert simulation.output_dataset.metadata["legacy_input_renames"] == RENAME + assert simulation.release_bundle["legacy_input_renames"] == RENAME + + +def _cut_before_the_mapping(path): + """Make a year file look like one ``create_datasets`` wrote before the fix. + + policyengine.py 6.0.0 to 6.1.1 stored neither name, since the engine + skipped the legacy column, and wrote no record. + """ + frames = {} + with pd.HDFStore(path, mode="r") as store: + for key in store.keys(): + frames[key.strip("/")] = store[key] + frames["person"] = frames["person"].drop(columns=[LIVE]) + with pd.HDFStore(path, mode="w") as store: + for key, frame in frames.items(): + store[key] = frame + with h5py.File(path, "r") as file: + assert RENAMES_H5_DATASET not in file - # The extracted data now carries the live name, so a run reads it - # natively and the mapping has nothing left to do. - simulation = _run(dataset) - outputs = _person_outputs(simulation) - assert outputs["wic"].iloc[INFANT] == 0 - assert outputs["wic"].iloc[TODDLER] > 0 - assert simulation.output_dataset.metadata["legacy_input_renames"] == {} + +def test_ensure_datasets_regenerates_a_year_file_cut_before_the_mapping(tmp_path): + source, dataset = _create_year_file(tmp_path) + _cut_before_the_mapping(dataset.filepath) + data_folder = str(tmp_path / "data") + + # The stale file has lost the draw: a run over it gives the infant WIC. + stale = PolicyEngineUSDataset( + name="stale", description="stale", filepath=dataset.filepath, year=YEAR + ) + assert LIVE not in pd.DataFrame(stale.data.person).columns + assert _person_outputs(_run(stale))["wic"].iloc[INFANT] > 0 + + with pytest.raises(ValueError, match="no record of the renamed stored inputs"): + load_datasets(datasets=[source], years=[YEAR], data_folder=data_folder) + + (regenerated,) = ensure_datasets( + datasets=[source], + years=[YEAR], + data_folder=data_folder, + allow_unmanaged=True, + ).values() + + assert regenerated.filepath == dataset.filepath + assert read_renames_record(dataset.filepath) == RENAME + assert pd.DataFrame(regenerated.data.person)[LIVE].tolist() == DRAW + # The regenerated file is current, so it is reused from now on. + (loaded,) = load_datasets( + datasets=[source], years=[YEAR], data_folder=data_folder + ).values() + assert pd.DataFrame(loaded.data.person)[LIVE].tolist() == DRAW + simulation = _run(loaded) + assert _person_outputs(simulation)["wic"].iloc[INFANT] == 0 + assert simulation.release_bundle["legacy_input_renames"] == RENAME # --- Saved outputs and run records keep the record ---------------------- From 51065c6ab29261bca0dd36216ee95570e370ede2 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 25 Sep 2026 20:51:11 -0400 Subject: [PATCH 3/3] Leave the legacy draw alone in a core H5 that stores the live name Simulation.run() refused a core variable/period H5 whose legacy would_claim_wic column was stored for part of a year even when the file also stored takes_up_wic_if_eligible. managed_microsimulation ignores the legacy column in that case, and the mapping is meant to do nothing once the data stores the live name. The part-year check now applies only when the file lacks the live name, on both paths, and a test covers a file that stores both. Co-Authored-By: Claude Opus 5.5 --- docs/microsim.md | 5 +-- .../tax_benefit_models/us/datasets.py | 15 +++++---- tests/test_us_legacy_inputs_integration.py | 32 +++++++++++++++++++ 3 files changed, 44 insertions(+), 8 deletions(-) diff --git a/docs/microsim.md b/docs/microsim.md index dcfaffb3..55820429 100644 --- a/docs/microsim.md +++ b/docs/microsim.md @@ -341,8 +341,9 @@ the new name. The new input is then set from the stored values for every month of every dataset year. The stored table must list the simulation's entity IDs in the simulation's order, or loading fails rather than attach values to the wrong people. Only values stored for a whole year are mapped, so a file that -stores the old name for part of a year is refused. The mapping turns itself off once the data stores the new name, -or if the engine defines the old name again. +stores the old name for part of a year, and not the new name, is refused. The +mapping turns itself off once the data stores the new name, or if the engine +defines the old name again. The renames applied are recorded as `{old: new}` (`{}` when none applied): diff --git a/src/policyengine/tax_benefit_models/us/datasets.py b/src/policyengine/tax_benefit_models/us/datasets.py index 1612d55d..111f4880 100644 --- a/src/policyengine/tax_benefit_models/us/datasets.py +++ b/src/policyengine/tax_benefit_models/us/datasets.py @@ -212,8 +212,8 @@ def _core_h5_entity_lengths(h5_file: h5py.File, year: int) -> dict[str, int]: return lengths -def _core_h5_variable_entities() -> tuple[dict[str, str], set[str]]: - """Return each variable's entity, and the stored legacy names to map.""" +def _core_h5_variable_entities() -> tuple[dict[str, str], dict[str, str]]: + """Return each variable's entity, and the pending renames ``{legacy: live}``.""" from policyengine_us.system import system entities = { @@ -225,7 +225,7 @@ def _core_h5_variable_entities() -> tuple[dict[str, str], set[str]]: pending = pending_legacy_input_renames(system.variables) for legacy, live in pending.items(): entities[legacy] = entities[live] - return entities, set(pending) + return entities, pending def _validate_entity_ids(data: dict[str, pd.DataFrame]) -> None: @@ -316,15 +316,18 @@ def _load_policyengine_core_h5(path: Path, year: int) -> USYearData: """Load a PolicyEngine core variable/period H5 into .py entity DataFrames.""" data = {entity: pd.DataFrame() for entity in US_ENTITY_KEYS} - variable_entities, legacy_names = _core_h5_variable_entities() + variable_entities, pending_renames = _core_h5_variable_entities() with h5py.File(path, "r") as h5_file: entity_lengths = _core_h5_entity_lengths(h5_file, year) for variable_name in h5_file.keys(): - if variable_name in legacy_names: + live_name = pending_renames.get(variable_name) + if live_name is not None and live_name not in h5_file: # A stored legacy column is mapped onto every month of the # year, so a part-year value is refused, as it is when - # ``managed_microsimulation`` reads this file. + # ``managed_microsimulation`` reads this file. A file that + # also stores the live name loads that natively, and its + # legacy column is left alone on both paths. node = h5_file[variable_name] check_yearly_periods( variable_name, diff --git a/tests/test_us_legacy_inputs_integration.py b/tests/test_us_legacy_inputs_integration.py index 406a9185..a03e755c 100644 --- a/tests/test_us_legacy_inputs_integration.py +++ b/tests/test_us_legacy_inputs_integration.py @@ -314,6 +314,38 @@ def test_a_core_h5_with_a_part_year_draw_is_refused_by_run(tmp_path): ) +def test_a_core_h5_storing_the_live_name_ignores_a_part_year_legacy_draw( + tmp_path, +): + """No-op: data that stores the live name is loaded natively on both paths. + + Its legacy column is not mapped, so the period it is stored for does not + matter, and neither load path refuses the file. + """ + path = _write_variable_centric(tmp_path / "both.h5", draw_period=f"{YEAR}-01") + live = [False, True, False] + with h5py.File(path, "a") as file: + file.create_dataset(f"{LIVE}/{YEAR}", data=np.array(live)) + + dataset = PolicyEngineUSDataset( + name="legacy-wic-draw-both", + description="Uncertified three-person WIC take-up fixture", + filepath=path, + year=YEAR, + ) + simulation = _run(dataset) + person = _person_outputs(simulation) + assert person[LIVE].tolist() == live + assert person["wic"].iloc[INFANT] > 0 + assert person["wic"].iloc[TODDLER] == 0 + assert simulation.output_dataset.metadata["legacy_input_renames"] == {} + + microsim = pe.us.managed_microsimulation(dataset=path, allow_unmanaged=True) + for month in MONTHS: + np.testing.assert_array_equal(microsim.calculate(LIVE, month).values, live) + assert microsim.policyengine_bundle["legacy_input_renames"] == {} + + # --- Load path 2: managed_microsimulation() over the country loader ----