From 10d3a340318d8a445fc44fc2950a4063feea8574 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 19 Sep 2026 17:25:08 -0400 Subject: [PATCH 1/5] Add pinned annual US static-aging candidate exports --- CLAUDE.md | 6 + .../annual-static-aging-export.added.md | 1 + docs/us-annual-static-aging.md | 63 +++ .../microcosm/build/us_annual_static_aging.py | 466 ++++++++++++++++++ .../tests/test_us_annual_static_aging.py | 287 +++++++++++ 5 files changed, 823 insertions(+) create mode 100644 changelog.d/annual-static-aging-export.added.md create mode 100644 docs/us-annual-static-aging.md create mode 100644 packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py create mode 100644 packages/microcosm-build/tests/test_us_annual_static_aging.py diff --git a/CLAUDE.md b/CLAUDE.md index 9e34670f6..7e614f82e 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -135,6 +135,12 @@ A sealed deny-list in `microcosm.build.us_runtime.h5_io` overrides this opt-in for known-excluded publications while preserving their scoring-only diagnostic path. +The independent US annual static-aging candidate builder lives in +`microcosm.build.us_annual_static_aging`; it consumes a pinned published parent +and writes local annual H5 files without running the base graph or publishing. +See [the annual candidate guide](docs/us-annual-static-aging.md). Its completion +manifest is build evidence, not release certification. + ## Root journals are history, not state The root `PROGRESS*.md`, `FINAL_REPORT.md`, `*_COVERAGE_PROGRESS.md`, and diff --git a/changelog.d/annual-static-aging-export.added.md b/changelog.d/annual-static-aging-export.added.md new file mode 100644 index 000000000..1871c7a71 --- /dev/null +++ b/changelog.d/annual-static-aging-export.added.md @@ -0,0 +1 @@ +Add a pinned local US annual static-aging candidate builder with preserved base artifacts, native single-year H5 files, source provenance, and checked round trips. diff --git a/docs/us-annual-static-aging.md b/docs/us-annual-static-aging.md new file mode 100644 index 000000000..4e659c015 --- /dev/null +++ b/docs/us-annual-static-aging.md @@ -0,0 +1,63 @@ +# US annual static-aging candidates + +`microcosm.build.us_annual_static_aging.build_annual_static_aging` creates a local +candidate containing one native single-year H5 for each year from the source year +through the requested end year (2035 by default). It reuses `static_aging` and +`multi_year_dataset`; it does not run the base calibration graph or publish data. + +The inputs are an exact base H5 SHA256 and parent release identifier, an exact SSA +population CSV SHA256, and the version, full commit, and clean source checkout of +the imported PolicyEngine-US model. The caller supplies the parent release claim; +release certification must authenticate that release and its base SHA separately. +The numerical environment must contain the reviewed model and Microcosm packages. + +After setting the input paths and reviewed pins, use the CLI in that environment: + +```sh +uv run --no-sync python -m microcosm.build.us_annual_static_aging \ + --base-h5 "$BASE_H5" --base-sha256 "$BASE_SHA256" \ + --parent-release "$PARENT_RELEASE" \ + --ssa-csv "$SSA_CSV" --ssa-sha256 "$SSA_SHA256" \ + --model-source "$MODEL_SOURCE" --model-commit "$MODEL_COMMIT" \ + --model-version "$MODEL_VERSION" --output-dir "$NEW_OUTPUT_DIR" \ + --base-year 2024 --end-year 2035 +``` + +The builder uses the frame-anchored demographic targets, SSA ages 80–84 pooled at +80 and 85+ pooled at 85, 300 calibration epochs, seed 0, and a maximum weight ratio +of 5. It records all settings in the manifest. Every projected year starts from +the original frame; years are calculated and serialized individually to avoid +retaining the entire budget window in memory. + +The source-year H5 is copied byte for byte. Projected files preserve all six +entities, columns, rows, IDs, and unscaled inputs. Household weights and mapped +monetary inputs change according to the existing static-aging operator. Signed +income inputs retain their separate positive and negative scale factors. Each +annual H5 uses root entity tables and an explicit `_time_period` equal to its +year. Logical and native-loader round trips compare all table values. Consumers +must select the exact annual file and year; its inputs are already projected. +The current publication contract requires table-format storage with direct HDF +fields. Fixed-format inputs, including missing nullable booleans that require +that storage format, are rejected before building and need a separate certified +layout contract. + +The output directory must be new. A failed attempt retains its partial files and +`build_status.json`; rerunning requires another directory. `annual_manifest.json` +is written last, after all years finish and source/input hashes are rechecked. +It contains: + +- `base`: source dataset, year, parent release claim, path, and SHA256. +- `inputs`, `model`, and `runtime`: SSA pin, model commit and source hashes, actual + implementation hashes, and dependency versions. +- `metadata.dataset_years`: a map from the source dataset to year-keyed annual + dataset names, including the preserved source year. +- `artifacts`: annual names mapped to relative H5 paths, hashes, years, entity row + counts, ordered columns, and round-trip receipts. +- Per-year projection receipt paths and hashes, containing the actual factors, + parameter series, demographic targets and achievements, and calibration fit. + +A complete candidate is not a certified release. Release preparation must add +normal immutable repository/revision pins, authenticate the parent, and run +independent demographic, monetary, identity, and runtime acceptance checks. +That process supplies the separate annual projection acceptance report and +publication decision; this module supplies no certification override. diff --git a/packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py b/packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py new file mode 100644 index 000000000..5166eaf20 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py @@ -0,0 +1,466 @@ +"""Build local, pinned annual US H5 candidates without publishing a release. + +Each projection starts from the same base frame. Only one projected year is +held in memory at a time. The original H5 is copied byte for byte, and all later +files retain its entity/column/row layout with an explicit ``_time_period``. +Run ``python -m microcosm.build.us_annual_static_aging --help``. +""" + +from __future__ import annotations + +import argparse +import hashlib +import importlib.metadata +import json +import shutil +import subprocess +from collections.abc import Mapping +from numbers import Integral +from pathlib import Path +from typing import Any + +import pandas as pd + +from microcosm.calibrate import ( + SeriesProjection, + ssa_population_projection, + static_aging, +) +from microcosm.frame import US_SCHEMA, Frame, SignedScale, WeightKind, Weights +from microcosm.frame.adapters.policyengine_us import multi_year_dataset, uprating_series +from microcosm.frame.materialize import ( + engine_tables, + materialize_nullable_booleans_for_pytables, + put_frame_table, + read_frame_table, +) + + +def _sha256(path: Path) -> str: + with path.open("rb") as stream: + return hashlib.file_digest(stream, "sha256").hexdigest() + + +def _json(path: Path, value: Any) -> None: + temporary = path.with_suffix(".json.tmp") + temporary.write_text( + json.dumps(value, indent=2, sort_keys=True, allow_nan=False) + "\n" + ) + temporary.replace(path) + + +def _check_pin(path: Path, expected: str) -> None: + if len(expected) != 64 or _sha256(path) != expected: + raise ValueError(f"SHA256 mismatch for {path}") + + +def _model_provenance(*, source: Path, commit: str, version: str) -> dict: + import policyengine_us + + source = source.resolve() + if ( + not Path(policyengine_us.__file__) + .resolve() + .is_relative_to(source / "policyengine_us") + ): + raise ValueError("Imported policyengine_us does not come from model_source") + actual_version = importlib.metadata.version("policyengine-us") + if actual_version != version: + raise ValueError( + f"Model version mismatch: expected {version}, got {actual_version}" + ) + + def git(*args: str) -> str: + return subprocess.check_output( + ["git", "-C", str(source), *args], text=True + ).strip() + + head = git("rev-parse", "HEAD") + if len(commit) != 40 or head != commit: + raise ValueError(f"Model commit mismatch: expected {commit}, got {head}") + if git("status", "--porcelain", "--untracked-files=all"): + raise ValueError("Model source must be a clean, committed checkout") + files = { + name: _sha256(source / name) + for name in git( + "ls-files", "--", "policyengine_us", "pyproject.toml", "uv.lock" + ).splitlines() + if (source / name).is_file() + } + digest = hashlib.sha256( + json.dumps(files, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest() + return { + "version": actual_version, + "commit": head, + "source": str(source), + "source_tree_sha256": digest, + "source_files": files, + } + + +def _runtime_provenance() -> dict: + # Hash the installed implementation, including solver/Frame helpers, rather + # than inferring a source identity from an editable package version. + import microcosm.calibrate + import microcosm.frame + + sources = {} + for package in (microcosm.calibrate, microcosm.frame): + root = Path(package.__file__).resolve().parent + sources.update( + { + f"{package.__name__}/{p.relative_to(root)}": _sha256(p) + for p in sorted(root.rglob("*.py")) + } + ) + sources["annual_static_aging.py"] = _sha256(Path(__file__)) + return { + "source_sha256": sources, + "versions": { + name: importlib.metadata.version(name) + for name in ( + "policyengine-us", + "policyengine-core", + "spm-calculator", + "microdf-python", + "numpy", + "pandas", + "scipy", + "torch", + "tables", + "microcosm-frame", + "microcosm-calibrate", + "microcosm-build", + ) + }, + } + + +def _period(store: pd.HDFStore, year: int) -> None: + if "/_time_period" not in store.keys(): + raise ValueError("Annual H5 must have explicit _time_period metadata") + if not store.get_storer("_time_period").is_table: + raise ValueError("Annual H5 requires table-format _time_period metadata") + values = store["_time_period"] + if ( + len(values) != 1 + or not isinstance(values.iloc[0], Integral) + or values.iloc[0] != year + ): + raise ValueError(f"Annual H5 _time_period must be exactly {year}") + + +def _verify(path: Path, tables: Mapping[str, pd.DataFrame], year: int) -> dict: + from policyengine_us.data import USSingleYearDataset + + native = USSingleYearDataset(file_path=str(path)) + if str(native.time_period) != str(year): + raise ValueError(f"Native loader did not retain year {year}") + with pd.HDFStore(path, "r") as store: + _period(store, year) + for entity, expected in tables.items(): + if not store.get_storer(entity).is_table: + raise ValueError( + f"Annual native layout requires table-format storage for {entity}; " + "fixed-format nullable inputs need a separately certified contract" + ) + if set(expected.columns) - set(store.get_storer(entity).table.colnames): + raise ValueError( + f"Annual native layout requires direct HDF fields for {entity}; " + "packed value blocks need a separately certified contract" + ) + actual = read_frame_table(store, entity) + pd.testing.assert_frame_equal( + actual, expected, check_dtype=False, check_exact=True + ) + external = materialize_nullable_booleans_for_pytables(expected).table + pd.testing.assert_frame_equal( + getattr(native, entity), external, check_dtype=False, check_exact=True + ) + return {"logical_tables": True, "native_loader": True, "time_period": year} + + +def _write_year(path: Path, tables: Mapping[str, pd.DataFrame], year: int) -> dict: + temporary = path.with_suffix(".tmp.h5") + with pd.HDFStore(temporary, "w") as store: + for entity, table in tables.items(): + put_frame_table( + store, entity, table, preferred_format="table", data_columns=True + ) + store.put("_time_period", pd.Series([year]), format="table") + receipt = _verify(temporary, tables, year) + temporary.replace(path) + return receipt + + +def build_annual_static_aging( + *, + base_h5: str | Path, + base_sha256: str, + parent_release: str, + ssa_csv: str | Path, + ssa_sha256: str, + model_source: str | Path, + model_commit: str, + model_version: str, + output_dir: str | Path, + base_year: int = 2024, + end_year: int = 2035, + epochs: int = 300, + seed: int = 0, + max_weight_ratio: float = 5.0, +) -> dict: + """Build a fresh local candidate, requiring explicit source and input pins. + + The caller attests ``parent_release``; certification must independently + verify that release's base SHA. A complete candidate is not a certified + release. Existing output directories are never reused, including failed + attempts. A failure preserves its partial artifacts and status receipt. + """ + output = Path(output_dir).resolve() + if output.exists(): + raise FileExistsError(f"Refusing to reuse output directory: {output}") + if ( + any( + not isinstance(y, Integral) or isinstance(y, bool) + for y in (base_year, end_year) + ) + or end_year <= base_year + ): + raise ValueError("end_year must be an integer after integer base_year") + if not parent_release.strip(): + raise ValueError("parent_release must identify the pinned base release") + base, ssa = Path(base_h5).resolve(), Path(ssa_csv).resolve() + _check_pin(base, base_sha256) + _check_pin(ssa, ssa_sha256) + model_kwargs = { + "source": Path(model_source), + "commit": model_commit, + "version": model_version, + } + model = _model_provenance(**model_kwargs) + runtime = _runtime_provenance() + with pd.HDFStore(base, "r") as store: + _period(store, base_year) + tables = { + entity: read_frame_table(store, entity) for entity in US_SCHEMA.entities + } + columns = {entity: list(table.columns) for entity, table in tables.items()} + weights = tables["household"].pop("household_weight").to_numpy(dtype=float) + if any( + f"{entity}_weight" in table + for entity in US_SCHEMA.entities + for table in tables.values() + ): + raise ValueError("Base must store only household weights") + frame = Frame( + tables, US_SCHEMA, {"household": Weights(weights, WeightKind.CALIBRATED)} + ) + base_tables = { + entity: table.loc[:, columns[entity]] + for entity, table in engine_tables( + frame, weighted_entities=("household",) + ).items() + } + _verify(base, base_tables, base_year) + del tables, base_tables + demographics = ssa_population_projection(ssa, age_top=85, age_bands={80: 84}) + for year in range(base_year, end_year + 1): + demographics.for_year(year) + cells = list(demographics.cells) + known_cells = pd.MultiIndex.from_frame(demographics.for_year(base_year)[cells]) + observed_cells = pd.MultiIndex.from_frame(frame.person[cells].drop_duplicates()) + if not observed_cells.isin(known_cells).all(): + raise ValueError("Base demographic cells do not match SSA age/sex coding") + + from policyengine_us import CountryTaxBenefitSystem + + system = CountryTaxBenefitSystem() + totals, indices, mapping = uprating_series( + [column for entity in frame.entities for column in frame.table(entity)], + tuple(range(base_year, end_year + 1)), + system=system, + ) + family = f"populace_us_{base_year}" + manifest = { + "schema_version": 1, + "kind": "us_annual_static_aging_candidate", + "status": "complete", + "base": { + "dataset": family, + "year": base_year, + "parent_release": parent_release, + "path": str(base), + "sha256": base_sha256, + }, + "inputs": {"ssa": {"path": str(ssa), "sha256": ssa_sha256}}, + "model": model, + "runtime": runtime, + "settings": { + "base_year": base_year, + "end_year": end_year, + "age_top": 85, + "age_bands": {"80": 84}, + "epochs": epochs, + "seed": seed, + "max_weight_ratio": max_weight_ratio, + "anchor": "frame", + "learning_rate": 0.02, + "l2_lambda": 0.0, + }, + "metadata": { + "dataset_years": { + family: { + str(year): f"populace_us_{year}" + for year in range(base_year, end_year + 1) + } + } + }, + "artifacts": {}, + } + output.mkdir(parents=True, exist_ok=False) + _json(output / "build_status.json", {"status": "running", "completed_years": []}) + try: + for year in range(base_year, end_year + 1): + name = f"populace_us_{year}" + path = output / f"{name}.h5" + if year == base_year: + shutil.copyfile(base, path) + _check_pin(path, base_sha256) + round_trip = { + "logical_tables": True, + "native_loader": True, + "time_period": year, + } + projection_receipt = None + else: + result = static_aging( + frame, + base_year=base_year, + years=(year,), + demographics=demographics, + series=SeriesProjection(totals=totals, indices=indices), + column_series=mapping, + epochs=epochs, + seed=seed, + max_weight_ratio=max_weight_ratio, + anchor="frame", + learning_rate=0.02, + l2_lambda=0.0, + ) + projection = result.year(year) + dataset = multi_year_dataset( + frame, + base_year, + {year: (projection.weights.values, projection.factors)}, + ).datasets[year] + projected_tables = { + entity: getattr(dataset, entity).loc[:, columns[entity]] + for entity in frame.entities + } + round_trip = _write_year(path, projected_tables, year) + receipt_path = output / f"projection_{year}.json" + _json( + receipt_path, + { + "year": year, + "base_year": base_year, + "column_series": mapping, + "totals": totals, + "indices": indices, + "factors": { + column: { + "positive": factor.positive, + "negative": factor.negative, + } + if isinstance(factor, SignedScale) + else float(factor) + for column, factor in projection.factors.items() + }, + "demographic_fit": projection.demographic_fit.to_dict( + orient="records" + ), + "fraction_within_10pct": float( + projection.fraction_within_10pct + ), + }, + ) + projection_receipt = { + "path": receipt_path.name, + "sha256": _sha256(receipt_path), + } + del projected_tables, dataset, projection, result + artifact = { + "path": path.name, + "sha256": _sha256(path), + "year": year, + "rows": {entity: frame.n(entity) for entity in frame.entities}, + "columns": columns, + "round_trip": round_trip, + } + if projection_receipt: + artifact["projection_receipt"] = projection_receipt + manifest["artifacts"][name] = artifact + _json( + output / "build_status.json", + { + "status": "running", + "completed_years": list(range(base_year, year + 1)), + }, + ) + _check_pin(base, base_sha256) + _check_pin(ssa, ssa_sha256) + if ( + _model_provenance(**model_kwargs) != model + or _runtime_provenance() != runtime + ): + raise ValueError("Implementation or model source changed during the build") + _json( + output / "build_status.json", + { + "status": "complete", + "completed_years": list(range(base_year, end_year + 1)), + }, + ) + # The manifest is the completion marker and is written last. + _json(output / "annual_manifest.json", manifest) + except BaseException as exc: + _json( + output / "build_status.json", + { + "status": "failed", + "error": f"{type(exc).__name__}: {exc}", + "completed_years": [ + item["year"] for item in manifest["artifacts"].values() + ], + }, + ) + raise + return manifest + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + for argument in ( + "base-h5", + "base-sha256", + "parent-release", + "ssa-csv", + "ssa-sha256", + "model-source", + "model-commit", + "model-version", + "output-dir", + ): + parser.add_argument(f"--{argument}", required=True) + parser.add_argument("--base-year", type=int, default=2024) + parser.add_argument("--end-year", type=int, default=2035) + parser.add_argument("--epochs", type=int, default=300) + parser.add_argument("--seed", type=int, default=0) + parser.add_argument("--max-weight-ratio", type=float, default=5.0) + build_annual_static_aging(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/packages/microcosm-build/tests/test_us_annual_static_aging.py b/packages/microcosm-build/tests/test_us_annual_static_aging.py new file mode 100644 index 000000000..a4284fde6 --- /dev/null +++ b/packages/microcosm-build/tests/test_us_annual_static_aging.py @@ -0,0 +1,287 @@ +"""Small-fixture checks for annual candidate serialization and evidence safety.""" + +from __future__ import annotations + +import json +import subprocess +import sys +from types import SimpleNamespace + +import numpy as np +import pandas as pd +import pytest + +from microcosm.build import us_annual_static_aging as annual +from microcosm.frame import US_SCHEMA, SignedScale, WeightKind, Weights +from microcosm.frame.materialize import put_frame_table, read_frame_table + +pytestmark = pytest.mark.requires_us + + +@pytest.fixture +def inputs(tmp_path, monkeypatch): + tables = { + entity: pd.DataFrame({US_SCHEMA.entity_id_column(entity): [1, 2, 3]}) + for entity in US_SCHEMA.entities + } + for group in US_SCHEMA.group_entities: + tables["person"][US_SCHEMA.membership_column(group)] = [1, 2, 3] + tables["person"]["age"] = [40, 40, 40] + tables["person"]["is_female"] = [False, True, False] + tables["person"]["employment_income_before_lsr"] = [100.0, 200.0, 300.0] + tables["person"]["miscellaneous_income"] = [10.0, -5.0, 0.0] + tables["person"]["input_flag"] = pd.Series([True, False, True], dtype="boolean") + tables["household"]["household_weight"] = [1.0, 2.0, 3.0] + tables["household"]["state_fips"] = [6, 6, 6] + base = tmp_path / "base.h5" + with pd.HDFStore(base, "w") as store: + for entity, table in tables.items(): + put_frame_table( + store, entity, table, preferred_format="table", data_columns=True + ) + store.put("_time_period", pd.Series([2024]), format="table") + ssa = tmp_path / "ssa.csv" + pd.DataFrame( + { + "Year": [2024, 2025, 2026], + "Age": [40] * 3, + "M Tot": [4, 5, 6], + "F Tot": [2, 3, 4], + } + ).to_csv(ssa, index=False) + monkeypatch.setattr( + annual, + "_model_provenance", + lambda **kwargs: {"version": "test", "commit": "a" * 40}, + ) + return dict( + base_h5=base, + base_sha256=annual._sha256(base), + parent_release="pinned-parent", + ssa_csv=ssa, + ssa_sha256=annual._sha256(ssa), + model_source=tmp_path, + model_commit="a" * 40, + model_version="test", + output_dir=tmp_path / "candidate", + base_year=2024, + end_year=2026, + epochs=2, + ) + + +def test_annual_files_preserve_base_shape_values_and_year(inputs, monkeypatch): + calls = [] + + def project(frame, **kwargs): + (year,) = kwargs["years"] + calls.append(year) + projection = SimpleNamespace( + weights=Weights(np.array([2.0, 3.0, 4.0]), WeightKind.CALIBRATED), + factors={ + "employment_income_before_lsr": 2.0, + "miscellaneous_income": SignedScale(3.0, 4.0), + }, + demographic_fit=pd.DataFrame({"target": [1.0], "achieved": [1.0]}), + fraction_within_10pct=1.0, + ) + return SimpleNamespace(year=lambda requested: projection) + + monkeypatch.setattr(annual, "static_aging", project) + manifest = annual.build_annual_static_aging(**inputs) + assert calls == [2025, 2026] # Every projection starts from the original frame. + output = inputs["output_dir"] + assert (output / "populace_us_2024.h5").read_bytes() == inputs[ + "base_h5" + ].read_bytes() + assert manifest["metadata"]["dataset_years"] == { + "populace_us_2024": {str(y): f"populace_us_{y}" for y in (2024, 2025, 2026)} + } + for year in (2025, 2026): + path = output / f"populace_us_{year}.h5" + import h5py + + with ( + h5py.File(inputs["base_h5"], "r") as base, + h5py.File(path, "r") as projected, + ): + for entity in US_SCHEMA.entities: + assert ( + projected[f"{entity}/table"].dtype.names + == base[f"{entity}/table"].dtype.names + ) + with pd.HDFStore(path, "r") as store: + assert store["_time_period"].iloc[0] == year + person = read_frame_table(store, "person") + assert person["employment_income_before_lsr"].tolist() == [200, 400, 600] + assert person["miscellaneous_income"].tolist() == [30, -20, 0] + assert person["person_id"].tolist() == [1, 2, 3] + assert person["age"].tolist() == [40, 40, 40] + artifact = manifest["artifacts"][f"populace_us_{year}"] + assert artifact["sha256"] == annual._sha256(path) + assert artifact["rows"] == {entity: 3 for entity in US_SCHEMA.entities} + assert artifact["round_trip"]["time_period"] == year + from policyengine_us.data import USSingleYearDataset + from policyengine_us.data.economic_assumptions import extend_single_year_dataset + + reloaded = USSingleYearDataset(file_path=str(path)) + extended = extend_single_year_dataset(reloaded) + # Native model loading may extend other years, but the selected annual + # inputs are already projected and must not receive a second factor. + assert extended.datasets[year].person[ + "employment_income_before_lsr" + ].tolist() == [200, 400, 600] + assert extended.datasets[year].household["household_weight"].tolist() == [ + 2, + 3, + 4, + ] + + +def test_small_actual_calibration_exports_independent_years(inputs): + manifest = annual.build_annual_static_aging(**inputs) + assert manifest["status"] == "complete" + receipt = json.loads((inputs["output_dir"] / "projection_2025.json").read_text()) + assert receipt["demographic_fit"] + assert receipt["column_series"]["employment_income_before_lsr"] in receipt["totals"] + + +def test_existing_output_is_rejected_before_work(inputs, monkeypatch): + output = inputs["output_dir"] + output.mkdir() + evidence = output / "prior.json" + evidence.write_text("untouched") + monkeypatch.setattr( + annual, "_model_provenance", lambda **kwargs: pytest.fail("must reject first") + ) + with pytest.raises(FileExistsError): + annual.build_annual_static_aging(**inputs) + assert evidence.read_text() == "untouched" + + +@pytest.mark.parametrize("corruption", ["hash", "year", "missing_year", "shape"]) +def test_invalid_base_is_rejected_without_completed_manifest(inputs, corruption): + if corruption == "hash": + inputs["base_sha256"] = "0" * 64 + else: + with pd.HDFStore(inputs["base_h5"], "a") as store: + if corruption == "year": + store.put("_time_period", pd.Series([2023])) + elif corruption == "missing_year": + del store["_time_period"] + else: + del store["family"] + inputs["base_sha256"] = annual._sha256(inputs["base_h5"]) + with pytest.raises((ValueError, KeyError)): + annual.build_annual_static_aging(**inputs) + assert not (inputs["output_dir"] / "annual_manifest.json").exists() + + +def test_failed_projection_retains_evidence_and_cannot_be_retried_in_place( + inputs, monkeypatch +): + def fail(*args, **kwargs): + raise RuntimeError("fixture projection failure") + + monkeypatch.setattr(annual, "static_aging", fail) + with pytest.raises(RuntimeError, match="fixture projection failure"): + annual.build_annual_static_aging(**inputs) + output = inputs["output_dir"] + before = {p.name: p.read_bytes() for p in output.iterdir()} + assert json.loads(before["build_status.json"])["status"] == "failed" + assert "annual_manifest.json" not in before + with pytest.raises(FileExistsError): + annual.build_annual_static_aging(**inputs) + assert before == {p.name: p.read_bytes() for p in output.iterdir()} + + +def test_source_drift_prevents_completion(inputs, monkeypatch): + snapshots = iter([{"sha": "before"}, {"sha": "after"}]) + monkeypatch.setattr(annual, "_runtime_provenance", lambda: next(snapshots)) + with pytest.raises(ValueError, match="source changed"): + annual.build_annual_static_aging(**inputs) + output = inputs["output_dir"] + assert not (output / "annual_manifest.json").exists() + assert json.loads((output / "build_status.json").read_text())["status"] == "failed" + + +def test_fixed_storage_base_requires_a_separate_layout_contract(inputs): + with pd.HDFStore(inputs["base_h5"], "a") as store: + person = read_frame_table(store, "person") + del store["person"] + person["input_flag"] = pd.Series([True, None, False], dtype="boolean") + put_frame_table( + store, "person", person, preferred_format="table", data_columns=True + ) + inputs["base_sha256"] = annual._sha256(inputs["base_h5"]) + with pytest.raises(ValueError, match="table-format storage"): + annual.build_annual_static_aging(**inputs) + assert not inputs["output_dir"].exists() + + +@pytest.mark.parametrize("packed", ["person", "_time_period"]) +def test_non_native_table_storage_rejected_before_build(inputs, packed): + with pd.HDFStore(inputs["base_h5"], "a") as store: + value = store[packed] + del store[packed] + store.put( + packed, value, format="fixed" if packed == "_time_period" else "table" + ) + inputs["base_sha256"] = annual._sha256(inputs["base_h5"]) + with pytest.raises(ValueError, match="direct HDF fields|table-format _time_period"): + annual.build_annual_static_aging(**inputs) + assert not inputs["output_dir"].exists() + + +def test_round_trip_rejects_changed_values_and_wrong_year(inputs): + with pd.HDFStore(inputs["base_h5"], "r") as store: + tables = { + entity: read_frame_table(store, entity) for entity in US_SCHEMA.entities + } + tables["person"].loc[0, "employment_income_before_lsr"] += 1 + with pytest.raises(AssertionError): + annual._verify(inputs["base_h5"], tables, 2024) + with pytest.raises(ValueError, match="year|time_period"): + annual._verify(inputs["base_h5"], tables, 2025) + + +def test_model_provenance_requires_imported_clean_pinned_source(tmp_path, monkeypatch): + package = tmp_path / "policyengine_us" + package.mkdir() + source_file = package / "__init__.py" + source_file.write_text("# fixture model\n") + + def git(*args): + return subprocess.check_output( + ["git", "-C", str(tmp_path), *args], text=True + ).strip() + + git("init", "-q") + git("add", "policyengine_us") + git( + "-c", + "user.name=Fixture", + "-c", + "user.email=fixture@example.com", + "commit", + "-qm", + "fixture", + ) + commit = git("rev-parse", "HEAD") + monkeypatch.setitem( + sys.modules, "policyengine_us", SimpleNamespace(__file__=str(source_file)) + ) + monkeypatch.setattr(annual.importlib.metadata, "version", lambda name: "test") + kwargs = dict(source=tmp_path, commit=commit, version="test") + receipt = annual._model_provenance(**kwargs) + assert receipt["commit"] == commit + assert receipt["source_files"] == { + "policyengine_us/__init__.py": annual._sha256(source_file) + } + with pytest.raises(ValueError, match="commit mismatch"): + annual._model_provenance(**{**kwargs, "commit": "0" * 40}) + with pytest.raises(ValueError, match="version mismatch"): + annual._model_provenance(**{**kwargs, "version": "other"}) + source_file.write_text("# edited\n") + with pytest.raises(ValueError, match="clean, committed"): + annual._model_provenance(**kwargs) From f08ebe044e406d3460063351f1e6c9d00cd15294 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 19 Sep 2026 17:30:11 -0400 Subject: [PATCH 2/5] Preserve logical H5 paths for cached base symlinks --- .../src/microcosm/build/us_annual_static_aging.py | 6 +++++- .../tests/test_us_annual_static_aging.py | 12 ++++++++++++ 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py b/packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py index 5166eaf20..1022364ec 100644 --- a/packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py +++ b/packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py @@ -231,7 +231,11 @@ def build_annual_static_aging( raise ValueError("end_year must be an integer after integer base_year") if not parent_release.strip(): raise ValueError("parent_release must identify the pinned base release") - base, ssa = Path(base_h5).resolve(), Path(ssa_csv).resolve() + # Hugging Face snapshot paths are .h5 symlinks to extensionless blobs. + # Keep the logical suffix required by the native loader; hashing and reads + # still follow the link to authenticate the actual bytes. + base = Path(base_h5).expanduser().absolute() + ssa = Path(ssa_csv).expanduser().absolute() _check_pin(base, base_sha256) _check_pin(ssa, ssa_sha256) model_kwargs = { diff --git a/packages/microcosm-build/tests/test_us_annual_static_aging.py b/packages/microcosm-build/tests/test_us_annual_static_aging.py index a4284fde6..5c8ad848b 100644 --- a/packages/microcosm-build/tests/test_us_annual_static_aging.py +++ b/packages/microcosm-build/tests/test_us_annual_static_aging.py @@ -146,6 +146,18 @@ def test_small_actual_calibration_exports_independent_years(inputs): assert receipt["column_series"]["employment_income_before_lsr"] in receipt["totals"] +def test_base_h5_symlink_to_extensionless_cache_blob(inputs): + base = inputs["base_h5"] + blob = base.with_name("cached-blob-without-extension") + base.rename(blob) + base.symlink_to(blob) + manifest = annual.build_annual_static_aging(**inputs) + assert manifest["base"]["path"] == str(base) + assert ( + inputs["output_dir"] / "populace_us_2024.h5" + ).read_bytes() == blob.read_bytes() + + def test_existing_output_is_rejected_before_work(inputs, monkeypatch): output = inputs["output_dir"] output.mkdir() From a40e47e55a76966b49e5b0cf904a21d9cfe7ac4c Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 19 Sep 2026 17:55:26 -0400 Subject: [PATCH 3/5] Validate immutable annual projection release cuts --- CLAUDE.md | 5 + DESIGN.md | 21 + .../333-annual-projection-contract.changed.md | 1 + docs/static-aging.md | 57 +- docs/us-annual-static-aging.md | 41 ++ .../src/microcosm/data/annual_projections.py | 411 ++++++++++++++ .../src/microcosm/data/contract.py | 40 +- .../src/microcosm/data/release.py | 2 +- .../src/microcosm/data/source_enrichment.py | 44 +- .../tests/test_annual_projections.py | 510 ++++++++++++++++++ packages/microcosm-data/tests/test_release.py | 108 +++- .../tests/test_source_enrichment.py | 74 +++ 12 files changed, 1286 insertions(+), 28 deletions(-) create mode 100644 changelog.d/333-annual-projection-contract.changed.md create mode 100644 packages/microcosm-data/src/microcosm/data/annual_projections.py create mode 100644 packages/microcosm-data/tests/test_annual_projections.py diff --git a/CLAUDE.md b/CLAUDE.md index 7e614f82e..4197ba491 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -140,6 +140,11 @@ The independent US annual static-aging candidate builder lives in and writes local annual H5 files without running the base graph or publishing. See [the annual candidate guide](docs/us-annual-static-aging.md). Its completion manifest is build evidence, not release certification. +Optional annual release metadata invokes additional artifact, identity, and +acceptance checks within the normal release gates. Annual cuts use one pinned +`-annual--` tag and cannot update latest +pointers. Qualify source-enrichment bases before adding annual metadata; use +the candidate guide's qualification order and tag-only publication route. ## Root journals are history, not state diff --git a/DESIGN.md b/DESIGN.md index fb6a9cf2f..27be01cf0 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -146,6 +146,27 @@ Logbook attempt row beside the output before the tool returns. The retiring two-spine lineage remains available only through its explicit compatibility flag for byte-reproducible historical builds. +## Annual cross-sectional projections + +Static aging supplies independent annual cross-sections for budget-window +estimates. It runs downstream of an accepted base population: demographic +projections change household weights, and monetary factors preserve projected +input aggregates under those weights. It preserves the base records, entity +IDs, and memberships. These repeated IDs identify source records; they do not +describe individual trajectories. + +The US release path exports one single-year H5 per supported year with the +base dataset's entity-table layout. Each annual artifact records its source +year, projection year, parent release and dataset hash, model and projection +inputs, and annual acceptance results. Base-year calibration evidence applies +to the base population; each projected year requires its own demographic, +aggregate, and runtime checks. Consumers select a declared annual artifact and +reject requests outside its published coverage. + +The base graph and calibration remain the source of the population. Annual +projection artifacts retain that source identity and do not certify a new +base population. See [static aging](docs/static-aging.md) and issue #333. + ## Longitudinal (the social-security-model direction) This section names kernel changes the current `Frame` does NOT yet support; diff --git a/changelog.d/333-annual-projection-contract.changed.md b/changelog.d/333-annual-projection-contract.changed.md new file mode 100644 index 000000000..d0863f819 --- /dev/null +++ b/changelog.d/333-annual-projection-contract.changed.md @@ -0,0 +1 @@ +Adopt static aging for annual budget-window cross-sections and define the downstream annual-file release contract, separate from base graph certification and individual trajectories. diff --git a/docs/static-aging.md b/docs/static-aging.md index 43d43d941..753c269b9 100644 --- a/docs/static-aging.md +++ b/docs/static-aging.md @@ -1,9 +1,10 @@ # Static aging `microcosm.calibrate.static_aging` projects a base-year frame to later years. -It reweights a fixed cross-section independently for each year. The proposed -operator remains subject to the cross-sectional-versus-longitudinal design -decision in [#333](https://github.com/PolicyEngine/microcosm/issues/333). +It reweights a fixed cross-section independently for each year and supplies +annual budget-window estimates. Dynamics remains the separate path for +individual trajectories. See the design decision in +[#333](https://github.com/PolicyEngine/microcosm/issues/333). ## The split @@ -85,12 +86,12 @@ Both `frame_for` and the US dataset exporter apply factors in float64. ## What it is not -The base cross-section is reweighted once per year with no person identity -across years. No transition happens: employment status stays at its -base-year distribution by age, so a projected downturn appears only as slower -per-capita income growth spread over everyone. That is the Dynamics -operator's job (sequencing step 6), and this step is replaced, not extended, -when it lands. +Static aging reweights the base cross-section once per year. Each output keeps +the source record IDs and memberships, but those IDs do not track individual +lives across years. The operator leaves employment status and other +demographic columns unchanged; weights change their representation. Monetary +factors change income amounts without simulating employment transitions. +Dynamics will model those transitions and individual trajectories. ## Country adapters @@ -101,3 +102,41 @@ series, including the source totals behind derived `_per_capita` parameters, come back as totals; other series come back as indices. `multi_year_dataset` exports a base-year bundle plus its projected years as a `USMultiYearDataset`, which the engine treats as already extended. + +## Annual release integration + +The publication layout uses one H5 per year with the existing single-year +entity-table format. The in-memory multi-year container can supply each +year's `USSingleYearDataset`; consumers do not need a combined on-disk file. +Each annual file keeps the base tables, columns, row counts, IDs, and +memberships. The year, household weights, and mapped monetary values change. +Floating-point precision and compression can change the file's byte size. + +Annual projections run downstream of an explicitly pinned accepted base. +A new graph build must pass its own release gates before it can supply that +base. Projection evidence records the parent release and H5 hash, source year, +projection year, model and projection-input identities, and annual checks. +The base's calibration receipt does not certify its projected years. + +Producer release metadata maps a dataset family to its annual artifact keys: + +```json +{ + "dataset_years": { + "populace_us_2024": { + "2024": "populace_us_2024", + "2030": "populace_us_2030", + "2035": "populace_us_2035" + } + } +} +``` + +This example abbreviates the mapping; a release through 2035 lists every +supported year. Each value references a normal revision- and hash-pinned +artifact. The wrapper validates the mapping during certification, checks the +file's stored year, and refuses unavailable years. Cache identities include +the actual artifact and relevant runtime versions. + +Local annual candidates are build evidence. Publication and wrapper +certification follow their separate acceptance checks. diff --git a/docs/us-annual-static-aging.md b/docs/us-annual-static-aging.md index 4e659c015..538b4aaf6 100644 --- a/docs/us-annual-static-aging.md +++ b/docs/us-annual-static-aging.md @@ -61,3 +61,44 @@ normal immutable repository/revision pins, authenticate the parent, and run independent demographic, monetary, identity, and runtime acceptance checks. That process supplies the separate annual projection acceptance report and publication decision; this module supplies no certification override. + +## Qualification and publication order + +1. Qualify the base release under its existing contract with the intended + country, Core, wrapper, and calculator wheels. Source-enrichment bases also + require the exact original parent H5 and their producer-source evidence. + Merely downloading an older bundle and rerunning compatibility does not + update its producer-source pins. If those pins no longer qualify, reproduce + the base bundle with reviewed producer code and prove preservation before + adding annual artifacts. +2. Run independent checks on the saved annual files. Record schema, source + identity, year, demographics, input aggregates, and runtime acceptance for + every declared year, including the base year. Runtime checks must exercise + the intended wrapper and model with the earlier annual inputs that policy + formulas need. Bind the acceptance report to the candidate manifest hash, + model commit and source hash, and country/Core versions. +3. Add `metadata.dataset_years`, `metadata.annual_projection_manifest`, and + `metadata.annual_projection_acceptance` to the qualified release manifest. + The latter two values name artifact keys for the candidate manifest and + the separate schema-1 `us_annual_projection_acceptance` report. Keep H5 + paths at the repository root; give both reports and every per-year + projection receipt their exact `releases//filename.json` + paths. The gate verifies each per-year receipt's hash and years against + the candidate manifest. Pin all artifacts to one + `-annual--` tag and their exact hashes. + Complete source-enrichment certification before this step: that operation + writes base-tag compatibility metadata. +4. Run normal publisher preparation with the artifact root, annual tag, and + `update_latest=False`. For source enrichment, also supply the original + parent H5 and compatibility wheels. The annual extension adds checks; the + publisher still enforces every original base-release gate. Publish with + `tag_only=True` to preserve the existing main-branch files as well as both + latest pointers. +5. Certify the wrapper against the explicit annual revision and release + manifest, then release the wrapper and update consumers. Certification + must retain the complete year map and exact annual artifact pins. + +The annual tag cannot become `latest.json` through this route. It augments the +same accepted base population; it does not promote another agent's new graph +candidate. A later graph release can supply a new base only after passing its +own qualification, followed by a fresh annual build and acceptance. diff --git a/packages/microcosm-data/src/microcosm/data/annual_projections.py b/packages/microcosm-data/src/microcosm/data/annual_projections.py new file mode 100644 index 000000000..25e3786c6 --- /dev/null +++ b/packages/microcosm-data/src/microcosm/data/annual_projections.py @@ -0,0 +1,411 @@ +"""Additional publication checks for explicit annual US dataset families. + +The parent release still has to pass its ordinary contract. These checks bind +annual acceptance evidence to the exact files, years, and source population; +they do not turn a candidate build receipt into a calibration certificate. +""" + +from __future__ import annotations + +import hashlib +import json +import re +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path, PurePosixPath + +import h5py +import numpy as np +from packaging.specifiers import SpecifierSet + +_SHA256 = re.compile(r"[0-9a-f]{64}\Z") +_YEAR = re.compile(r"[1-9][0-9]{3}\Z") +_ENTITIES = ("person", "household", "tax_unit", "spm_unit", "family", "marital_unit") +_CHECKS = frozenset( + {"schema", "source_identity", "year", "demographics", "input_aggregates", "runtime"} +) + + +@dataclass(frozen=True) +class AnnualProjectionExtension: + """Validated publication additions; inherited artifacts keep their contract.""" + + revision: str + additional_artifacts: Mapping[str, Path] + projected_artifacts: frozenset[str] + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for block in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(block) + return digest.hexdigest() + + +def _object(value: object, label: str) -> Mapping: + if not isinstance(value, Mapping): + raise ValueError(f"{label} must be an object") + return value + + +def _read_json(path: Path) -> Mapping: + return _object(json.loads(path.read_text()), path.name) + + +def _artifact( + key: str, artifacts: Mapping, release_dir: Path, artifact_root: Path | None +) -> tuple[Path, str]: + record = _object(artifacts.get(key), f"artifact {key!r}") + relative = record.get("path") + sha = record.get("sha256") + if ( + not isinstance(relative, str) + or not relative + or "\\" in relative + or PurePosixPath(relative).is_absolute() + or PurePosixPath(relative).as_posix() != relative + or any(part in {".", ".."} for part in relative.split("/")) + ): + raise ValueError(f"annual artifact {key!r} must have a clean relative path") + if not isinstance(sha, str) or not _SHA256.fullmatch(sha): + raise ValueError(f"annual artifact {key!r} requires sha256") + if not isinstance(record.get("revision"), str) or not record["revision"]: + raise ValueError(f"annual artifact {key!r} requires a revision pin") + prefix = f"releases/{release_dir.name}/" + if relative.startswith(prefix): + filename = relative.removeprefix(prefix) + if not filename or "/" in filename: + raise ValueError("annual release-local artifacts must be bare filenames") + local = release_dir / filename + elif relative.startswith("releases/"): + raise ValueError("annual artifacts cannot use another release's prefix") + elif (release_dir / relative).is_file() or (release_dir / relative).is_symlink(): + raise ValueError( + "annual release-local artifacts require their exact release prefix" + ) + elif artifact_root is not None: + local = artifact_root / relative + else: + raise ValueError(f"annual artifact {key!r} requires artifact_root") + if not local.is_file() or _sha256(local) != sha: + raise ValueError(f"annual artifact {key!r} bytes do not match its sha256") + return local, sha + + +def _native_year(h5: h5py.File) -> int: + if "_time_period/table" not in h5: + raise ValueError("annual H5 requires explicit _time_period metadata") + stored = h5["_time_period/table"] + if ( + not isinstance(stored, h5py.Dataset) + or stored.shape != (1,) + or "values" not in (stored.dtype.names or ()) + or stored.dtype["values"].kind not in {"i", "u"} + ): + raise ValueError("annual H5 _time_period must contain one integer year") + return int(stored["values"][0]) + + +def _check_native_identity( + base: Path, projected: Path, year: int, record: Mapping +) -> None: + with h5py.File(base) as original, h5py.File(projected) as annual: + if _native_year(annual) != year: + raise ValueError(f"annual H5 stored year does not match {year}") + rows = _object(record.get("rows"), f"{year} row counts") + for entity in _ENTITIES: + path = f"{entity}/table" + if path not in original or path not in annual: + raise ValueError(f"annual H5 lacks native {entity} table") + source, target = original[path], annual[path] + if not isinstance(source, h5py.Dataset) or not isinstance( + target, h5py.Dataset + ): + raise ValueError(f"{entity} table must be a dataset") + if source.shape != target.shape or rows.get(entity) != len(target): + raise ValueError(f"{year} {entity} row count differs from its source") + columns = source.dtype.names + if not columns or columns != target.dtype.names: + raise ValueError( + f"{year} {entity} native columns differ from its source" + ) + identity_columns = [ + column + for column in columns + if column == f"{entity}_id" + or ( + entity == "person" + and column.startswith("person_") + and column.endswith("_id") + ) + ] + if f"{entity}_id" not in identity_columns: + raise ValueError(f"{year} {entity} lacks its native identity column") + for column in identity_columns: + if not np.array_equal(source[column], target[column]): + raise ValueError(f"{year} {column} differs from its source") + weights = annual["household/table"]["household_weight"] + if ( + not np.isfinite(weights).all() + or (weights < 0).any() + or not (weights > 0).any() + ): + raise ValueError( + f"{year} household weights must be finite nonnegative with positive mass" + ) + + +def validate_annual_projection_extension( + release_dir: Path | str, + manifest: Mapping, + *, + artifact_root: Path | str | None = None, +) -> AnnualProjectionExtension | None: + """Check optional annual metadata and every referenced artifact locally. + + Releases without ``metadata.dataset_years`` retain their existing contract. + A present annual map requires separate passing acceptance evidence. The + publisher calls this in addition to the base release's validation. + """ + metadata = manifest.get("metadata", {}) + if not isinstance(metadata, Mapping) or "dataset_years" not in metadata: + return + release_dir = Path(release_dir) + root = Path(artifact_root) if artifact_root is not None else None + families = _object(metadata["dataset_years"], "metadata.dataset_years") + if not families: + raise ValueError("metadata.dataset_years must not be empty") + artifacts = _object(manifest.get("artifacts"), "release artifacts") + evidence_key = metadata.get("annual_projection_manifest") + acceptance_key = metadata.get("annual_projection_acceptance") + if not isinstance(evidence_key, str) or not isinstance(acceptance_key, str): + raise ValueError( + "annual releases require projection manifest and acceptance artifact keys" + ) + evidence_path, evidence_sha = _artifact(evidence_key, artifacts, release_dir, root) + acceptance_path, _ = _artifact(acceptance_key, artifacts, release_dir, root) + prefix = f"releases/{release_dir.name}/" + for key in (evidence_key, acceptance_key): + if not artifacts[key]["path"].startswith(prefix): + raise ValueError("annual receipts require their exact release prefix") + evidence, acceptance = _read_json(evidence_path), _read_json(acceptance_path) + if ( + type(evidence.get("schema_version")) is not int + or evidence.get("schema_version") != 1 + or evidence.get("kind") != "us_annual_static_aging_candidate" + or evidence.get("status") != "complete" + ): + raise ValueError( + "annual projection manifest must describe a complete schema-1 candidate" + ) + if ( + type(acceptance.get("schema_version")) is not int + or acceptance.get("schema_version") != 1 + or acceptance.get("kind") != "us_annual_projection_acceptance" + or acceptance.get("status") != "passed" + or acceptance.get("projection_manifest_sha256") != evidence_sha + ): + raise ValueError( + "annual acceptance must pass and bind the exact projection manifest" + ) + if ( + _object(evidence.get("metadata"), "projection metadata").get("dataset_years") + != families + ): + raise ValueError("annual release coverage differs from its projection evidence") + base = _object(evidence.get("base"), "projection base") + base_key = base.get("dataset") + base_year = base.get("year") + if ( + not isinstance(base_key, str) + or type(base_year) is not int + or not isinstance(base.get("parent_release"), str) + or not base["parent_release"] + ): + raise ValueError( + "projection base requires dataset, integer year and parent release" + ) + base_path, base_sha = _artifact(base_key, artifacts, release_dir, root) + if base_sha != base.get("sha256"): + raise ValueError( + "projection base hash differs from the release's base artifact" + ) + defaults = _object(manifest.get("default_datasets"), "release defaults") + build = _object(manifest.get("build"), "release build") + if ( + defaults.get("national") != base_key + or base["parent_release"] != build.get("build_id") + or base["parent_release"] != release_dir.name + ): + raise ValueError( + "annual projections must augment this release's certified national base" + ) + model = _object(evidence.get("model"), "projection model") + model_identity = { + name: model.get(name) for name in ("version", "commit", "source_tree_sha256") + } + if ( + not isinstance(model_identity["version"], str) + or not isinstance(model_identity["commit"], str) + or not re.fullmatch(r"[0-9a-f]{40}", model_identity["commit"]) + or not isinstance(model_identity["source_tree_sha256"], str) + or not _SHA256.fullmatch(model_identity["source_tree_sha256"]) + or acceptance.get("model") != model_identity + ): + raise ValueError( + "annual acceptance must bind the projection model's version and source identity" + ) + runtime = _object( + _object(evidence.get("runtime"), "projection runtime").get("versions"), + "projection versions", + ) + accepted_runtime = _object(acceptance.get("runtime"), "accepted runtime") + for package, compatible_field, built_field in ( + ("policyengine-us", "compatible_model_packages", "built_with_model_package"), + ("policyengine-core", "compatible_core_packages", "built_with_core_package"), + ): + version = runtime.get(package) + built = _object(build.get(built_field), f"release {built_field}") + allowed = built.get("name") == package and built.get("version") == version + for claim in manifest.get(compatible_field, []): + claim = _object(claim, f"release {compatible_field} entry") + if claim.get("name") == package and isinstance(version, str): + raw_specifier = claim.get("specifier") + if not isinstance(raw_specifier, str) or not raw_specifier.strip(): + raise ValueError( + "annual compatibility claims require nonempty specifiers" + ) + specifier = SpecifierSet(raw_specifier) + if not tuple(specifier): + raise ValueError( + "annual compatibility claims require nonempty specifiers" + ) + allowed = allowed or version in specifier + if ( + not isinstance(version, str) + or accepted_runtime.get(package) != version + or not allowed + ): + raise ValueError( + f"annual runtime {package} must match acceptance and the release compatibility contract" + ) + if runtime["policyengine-us"] != model["version"]: + raise ValueError("projection model and runtime versions differ") + evidence_artifacts = _object(evidence.get("artifacts"), "projection artifacts") + accepted_years = _object(acceptance.get("years"), "annual acceptance years") + all_years: set[str] = set() + additional_artifacts = { + evidence_key: evidence_path, + acceptance_key: acceptance_path, + } + projected_artifacts: set[str] = set() + for family, mapping in families.items(): + if family != base_key: + raise ValueError( + "each projection receipt must describe its declared base family" + ) + years = _object(mapping, f"annual family {family!r}") + if not years or years.get(str(base_year)) != base_key: + raise ValueError("annual family must include its unchanged base year") + seen: set[str] = set() + for year_text, key in years.items(): + if not isinstance(year_text, str) or not _YEAR.fullmatch(year_text): + raise ValueError("annual year keys must be four-digit decimal strings") + if not isinstance(key, str) or key in seen or int(year_text) < base_year: + raise ValueError( + "annual artifacts must be unique and no earlier than their source" + ) + seen.add(key) + all_years.add(year_text) + path, sha = _artifact(key, artifacts, release_dir, root) + if key != base_key: + if artifacts[key]["path"].startswith("releases/"): + raise ValueError( + "projected annual H5 artifacts must use root paths" + ) + if artifacts[key].get("kind") != "microdata": + raise ValueError("projected annual H5 artifacts must be microdata") + additional_artifacts[key] = path + projected_artifacts.add(key) + record = _object( + evidence_artifacts.get(key), f"projection artifact {key!r}" + ) + if record.get("sha256") != sha or record.get("year") != int(year_text): + raise ValueError( + f"{year_text} artifact does not match projection evidence" + ) + if key != base_key: + receipt = _object( + record.get("projection_receipt"), + f"{year_text} projection receipt", + ) + filename = receipt.get("path") + if ( + not isinstance(filename, str) + or not filename + or filename in {".", ".."} + or "/" in filename + or "\\" in filename + ): + raise ValueError("projection receipt path must be a bare filename") + matches = [ + receipt_key + for receipt_key, entry in artifacts.items() + if isinstance(entry, Mapping) + and entry.get("path") == prefix + filename + ] + if len(matches) != 1: + raise ValueError( + f"{year_text} projection receipt requires one declared " + "artifact with its exact release prefix" + ) + receipt_key = matches[0] + receipt_path, receipt_sha = _artifact( + receipt_key, artifacts, release_dir, root + ) + if receipt_sha != receipt.get("sha256"): + raise ValueError( + f"{year_text} projection receipt hash differs from candidate" + ) + receipt_data = _read_json(receipt_path) + if ( + receipt_data.get("year") != int(year_text) + or receipt_data.get("base_year") != base_year + ): + raise ValueError( + f"{year_text} projection receipt has different year/base_year" + ) + additional_artifacts[receipt_key] = receipt_path + accepted = _object(accepted_years.get(year_text), f"{year_text} acceptance") + checks = _object(accepted.get("checks"), f"{year_text} acceptance checks") + if ( + accepted.get("dataset") != key + or accepted.get("sha256") != sha + or not _CHECKS.issubset(checks) + or any(checks[name] != "passed" for name in _CHECKS) + ): + raise ValueError( + f"{year_text} lacks passing acceptance for its exact artifact" + ) + _check_native_identity(base_path, path, int(year_text), record) + if set(accepted_years) != all_years: + raise ValueError("annual acceptance coverage differs from the release coverage") + revisions = [ + _object(record, f"artifact {key!r}").get("revision") + for key, record in artifacts.items() + ] + if ( + any(not isinstance(revision, str) for revision in revisions) + or len(set(revisions)) != 1 + ): + raise ValueError("annual artifacts must share one immutable annual cut tag") + revision = revisions[0] + if not isinstance(revision, str) or not re.fullmatch( + re.escape(release_dir.name) + r"-annual-[0-9]{8}T[0-9]{6}Z-[0-9a-f]{8}", + revision, + ): + raise ValueError("annual artifacts require an immutable annual cut tag") + return AnnualProjectionExtension( + revision, additional_artifacts, frozenset(projected_artifacts) + ) diff --git a/packages/microcosm-data/src/microcosm/data/contract.py b/packages/microcosm-data/src/microcosm/data/contract.py index e32fb95d6..a56eafbbd 100644 --- a/packages/microcosm-data/src/microcosm/data/contract.py +++ b/packages/microcosm-data/src/microcosm/data/contract.py @@ -1181,6 +1181,7 @@ def _check_release_manifest( failures: list[str], *, expected_schema_version: object = RELEASE_MANIFEST_SCHEMA_VERSION, + annual_revision: str | None = None, ) -> None: schema_version = manifest.get("schema_version") if schema_version is None: @@ -1279,14 +1280,18 @@ def _check_release_manifest( f"release_manifest.json artifact {key!r} is missing {field!r}." ) revision = entry.get("revision") - revision_matches_release = revision == release_id or ( - release_id == _UK_NATIONAL_RELEASE_ID - and isinstance(revision, str) - and revision.startswith(release_id + "-") - and _UK_NATIONAL_REVISION_SUFFIX_RE.fullmatch( - revision[len(release_id) + 1 :] + revision_matches_release = ( + revision == release_id + or (annual_revision is not None and revision == annual_revision) + or ( + release_id == _UK_NATIONAL_RELEASE_ID + and isinstance(revision, str) + and revision.startswith(release_id + "-") + and _UK_NATIONAL_REVISION_SUFFIX_RE.fullmatch( + revision[len(release_id) + 1 :] + ) + is not None ) - is not None ) # A present-but-non-string revision must fail here rather than # slide past the isinstance guard: publish collects only string @@ -5127,6 +5132,7 @@ def validate_release_dir( # than a silent fallback. role: str = NATIONAL_DEFAULT_DATASET_ROLE manifest_probe_path = release_dir / "release_manifest.json" + annual_extension = None if manifest_probe_path.is_file(): try: manifest_probe = json.loads(manifest_probe_path.read_text()) @@ -5155,6 +5161,19 @@ def validate_release_dir( f"{manifest_probe['release_type']!r}." ], ) + if isinstance(manifest_probe, Mapping): + from microcosm.data.annual_projections import ( + validate_annual_projection_extension, + ) + + try: + annual_extension = validate_annual_projection_extension( + release_dir, manifest_probe, artifact_root=artifact_root + ) + except (OSError, ValueError, KeyError, TypeError) as exc: + raise ReleaseContractError( + release_dir, [f"annual projection extension: {exc}"] + ) from exc if isinstance(manifest_probe, Mapping) and "dataset_role" in manifest_probe: declared_role = manifest_probe["dataset_role"] if declared_role not in ( @@ -5200,7 +5219,12 @@ def validate_release_dir( manifest = _load_json(release_manifest_path, failures) if manifest is not None: release_manifest = manifest - _check_release_manifest(manifest, release_id, failures) + _check_release_manifest( + manifest, + release_id, + failures, + annual_revision=annual_extension.revision if annual_extension else None, + ) calibration_diagnostics_path = release_dir / "calibration_diagnostics.json" if calibration_diagnostics_path.is_file(): diff --git a/packages/microcosm-data/src/microcosm/data/release.py b/packages/microcosm-data/src/microcosm/data/release.py index cdf0afbe4..710b0ce0f 100644 --- a/packages/microcosm-data/src/microcosm/data/release.py +++ b/packages/microcosm-data/src/microcosm/data/release.py @@ -218,7 +218,7 @@ def prepare_release( compatibility_wheels=compatibility_wheels, ) else: - validate_release_dir(release_dir) + validate_release_dir(release_dir, artifact_root=artifact_root) release_id = release_dir.name role = release_dataset_role(release_dir) if role != NATIONAL_DEFAULT_DATASET_ROLE and update_latest: diff --git a/packages/microcosm-data/src/microcosm/data/source_enrichment.py b/packages/microcosm-data/src/microcosm/data/source_enrichment.py index 59048d0fd..509ced1fa 100644 --- a/packages/microcosm-data/src/microcosm/data/source_enrichment.py +++ b/packages/microcosm-data/src/microcosm/data/source_enrichment.py @@ -249,11 +249,20 @@ def validate_source_enrichment_candidate( as measurements of the enriched model. Both actual H5 files are mandatory: a hand-written preservation receipt is not sufficient evidence. """ + from microcosm.data.annual_projections import validate_annual_projection_extension from microcosm.data.h5_enrichment import compare_h5_enrichment release_dir = Path(release_dir) failures: list[str] = [] manifest = _json(release_dir / "release_manifest.json", failures) + try: + annual_extension = validate_annual_projection_extension( + release_dir, manifest, artifact_root=artifact_root + ) + except (OSError, ValueError, KeyError, TypeError) as exc: + raise ReleaseContractError( + release_dir, [f"annual projection extension: {exc}"] + ) from exc build = _json(release_dir / "build_manifest.json", failures) report = _json(release_dir / SOURCE_ENRICHMENT_FILE, failures) if manifest.get("release_type") != SOURCE_ENRICHMENT_RELEASE_TYPE: @@ -421,11 +430,15 @@ def validate_source_enrichment_candidate( SOURCE_PROVENANCE_FILE, } artifacts = _mapping(manifest.get("artifacts")) + annual_artifacts = annual_extension.additional_artifacts if annual_extension else {} + revision = annual_extension.revision if annual_extension else release_dir.name by_path = {} for key, raw in artifacts.items(): entry = _mapping(raw) path = entry.get("path") - if not isinstance(path, str) or Path(path).name != path: + if not isinstance(path, str) or ( + key not in annual_artifacts and Path(path).name != path + ): failures.append( f"source enrichment artifact {key} must use a bare filename" ) @@ -435,12 +448,18 @@ def validate_source_enrichment_candidate( by_path[path] = entry if ( entry.get("repo_id") != "policyengine/populace-us" - or entry.get("revision") != release_dir.name + or entry.get("revision") != revision ): failures.append( f"source enrichment artifact {key} must pin the new repo/tag" ) - local = candidate if path == filename else release_dir / path + local = ( + annual_artifacts[key] + if key in annual_artifacts + else candidate + if path == filename + else release_dir / path + ) if ( local is None or not local.is_file() @@ -461,7 +480,17 @@ def validate_source_enrichment_candidate( ): failures.append("default_datasets.national must select the enriched native H5") if ( - len([entry for entry in by_path.values() if entry.get("kind") == "microdata"]) + len( + [ + entry + for key, entry in artifacts.items() + if _mapping(entry).get("kind") == "microdata" + and ( + annual_extension is None + or key not in annual_extension.projected_artifacts + ) + ] + ) != 1 ): failures.append( @@ -505,6 +534,7 @@ def validate_source_enrichment_candidate( require_compatibility, compatibility_wheels, failures, + annual_revision=annual_extension.revision if annual_extension else None, ) else: failures.append( @@ -908,6 +938,8 @@ def _check_compatibility( require_wheel_proof, wheels, failures, + *, + annual_revision: str | None = None, ): receipt_path = release_dir / COMPATIBILITY_FILE receipt = _json(receipt_path, failures) @@ -931,7 +963,9 @@ def _check_compatibility( ) from microcosm.data.contract import _check_release_manifest - _check_release_manifest(manifest, release_dir.name, failures) + _check_release_manifest( + manifest, release_dir.name, failures, annual_revision=annual_revision + ) raw_claims = compatibility.get("publisher_claims") claims = _mapping(raw_claims) malformed_claims = raw_claims is not None and ( diff --git a/packages/microcosm-data/tests/test_annual_projections.py b/packages/microcosm-data/tests/test_annual_projections.py new file mode 100644 index 000000000..50388052b --- /dev/null +++ b/packages/microcosm-data/tests/test_annual_projections.py @@ -0,0 +1,510 @@ +"""Annual publication binds accepted years to exact native H5 artifacts.""" + +from __future__ import annotations + +import hashlib +import json +import shutil + +import h5py +import numpy as np +import pytest + +from microcosm.data.annual_projections import validate_annual_projection_extension +from microcosm.data.contract import ReleaseContractError, validate_release_dir + + +def _sha(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _h5(path, year, *, person_id=1, weight=10.0): + with h5py.File(path, "w") as store: + store.create_dataset( + "_time_period/table", + data=np.array([(0, year)], dtype=[("index", "i8"), ("values", "i8")]), + ) + for entity in ( + "person", + "household", + "tax_unit", + "spm_unit", + "family", + "marital_unit", + ): + fields = [("index", "i8"), (f"{entity}_id", "i8")] + row = [0, person_id if entity == "person" else 1] + if entity == "person": + fields.extend( + [("person_household_id", "i8"), ("employment_income", "f8")] + ) + row.extend([1, 100.0 * (year - 2023)]) + if entity == "household": + fields.append(("household_weight", "f8")) + row.append(weight) + store.create_dataset( + f"{entity}/table", data=np.array([tuple(row)], dtype=fields) + ) + + +@pytest.fixture +def candidate(tmp_path): + release = tmp_path / "release-id" + release.mkdir() + root = tmp_path / "artifacts" + root.mkdir() + mapping = { + "populace_us_2024": {"2024": "populace_us_2024", "2030": "populace_us_2030"} + } + manifest = { + "metadata": {"dataset_years": mapping}, + "artifacts": {}, + "default_datasets": {"national": "populace_us_2024"}, + "build": { + "build_id": release.name, + "built_with_model_package": { + "name": "policyengine-us", + "version": "2.6.15", + }, + "built_with_core_package": { + "name": "policyengine-core", + "version": "3.32.5", + }, + }, + } + evidence = { + "schema_version": 1, + "kind": "us_annual_static_aging_candidate", + "status": "complete", + "base": { + "dataset": "populace_us_2024", + "year": 2024, + "parent_release": release.name, + }, + "metadata": {"dataset_years": mapping}, + "artifacts": {}, + "model": { + "version": "2.6.15", + "commit": "a" * 40, + "source_tree_sha256": "b" * 64, + }, + "runtime": { + "versions": {"policyengine-us": "2.6.15", "policyengine-core": "3.32.5"} + }, + } + acceptance = { + "schema_version": 1, + "kind": "us_annual_projection_acceptance", + "status": "passed", + "years": {}, + "model": dict(evidence["model"]), + "runtime": dict(evidence["runtime"]["versions"]), + } + for year in (2024, 2030): + key = f"populace_us_{year}" + path = root / f"{key}.h5" + _h5(path, year) + sha = _sha(path) + manifest["artifacts"][key] = { + "path": path.name, + "sha256": sha, + "revision": f"{release.name}-annual-20260919T220000Z-a1b2c3d4", + "kind": "microdata", + } + evidence["artifacts"][key] = { + "year": year, + "sha256": sha, + "rows": { + name: 1 + for name in ( + "person", + "household", + "tax_unit", + "spm_unit", + "family", + "marital_unit", + ) + }, + } + acceptance["years"][str(year)] = { + "dataset": key, + "sha256": sha, + "checks": { + name: "passed" + for name in ( + "schema", + "source_identity", + "year", + "demographics", + "input_aggregates", + "runtime", + ) + }, + } + evidence["base"]["sha256"] = manifest["artifacts"]["populace_us_2024"]["sha256"] + projection_path = release / "projection_2030.json" + projection_path.write_text( + json.dumps({"base_year": 2024, "year": 2030, "factors": {}}) + ) + evidence["artifacts"]["populace_us_2030"]["projection_receipt"] = { + "path": projection_path.name, + "sha256": _sha(projection_path), + } + manifest["artifacts"]["projection_2030"] = { + "path": f"releases/{release.name}/{projection_path.name}", + "sha256": _sha(projection_path), + "revision": f"{release.name}-annual-20260919T220000Z-a1b2c3d4", + "kind": "diagnostics", + } + + def save(): + evidence_path = release / "annual_manifest.json" + evidence_path.write_text(json.dumps(evidence)) + acceptance["projection_manifest_sha256"] = _sha(evidence_path) + acceptance_path = release / "annual_acceptance.json" + acceptance_path.write_text(json.dumps(acceptance)) + for key, field, path in ( + ("annual_manifest", "annual_projection_manifest", evidence_path), + ("annual_acceptance", "annual_projection_acceptance", acceptance_path), + ): + manifest["metadata"][field] = key + manifest["artifacts"][key] = { + "path": f"releases/{release.name}/{path.name}", + "sha256": _sha(path), + "revision": f"{release.name}-annual-20260919T220000Z-a1b2c3d4", + } + (release / "release_manifest.json").write_text(json.dumps(manifest)) + + save() + return release, root, manifest, evidence, acceptance, save + + +def add_annual_extension(release, root): + """Augment an otherwise valid release without replacing its base evidence.""" + manifest = json.loads((release / "release_manifest.json").read_text()) + base_key = manifest["default_datasets"]["national"] + base = manifest["artifacts"][base_key] + projected = root / "annual_2030.h5" + shutil.copyfile(root / base["path"], projected) + with h5py.File(projected, "r+") as store: + row = store["_time_period/table"][:] + row["values"] = 2030 + store["_time_period/table"][:] = row + rows = { + entity: len(store[f"{entity}/table"]) + for entity in ( + "person", + "household", + "tax_unit", + "spm_unit", + "family", + "marital_unit", + ) + } + runtime = { + "policyengine-us": manifest["build"]["built_with_model_package"]["version"], + "policyengine-core": manifest["build"]["built_with_core_package"]["version"], + } + model = { + "version": runtime["policyengine-us"], + "commit": "a" * 40, + "source_tree_sha256": "b" * 64, + } + mapping = {base_key: {"2024": base_key, "2030": "annual_2030"}} + evidence = { + "schema_version": 1, + "kind": "us_annual_static_aging_candidate", + "status": "complete", + "base": { + "dataset": base_key, + "year": 2024, + "sha256": base["sha256"], + "parent_release": release.name, + }, + "metadata": {"dataset_years": mapping}, + "model": model, + "runtime": {"versions": runtime}, + "artifacts": {}, + } + acceptance = { + "schema_version": 1, + "kind": "us_annual_projection_acceptance", + "status": "passed", + "model": model, + "runtime": runtime, + "years": {}, + } + manifest["artifacts"]["annual_2030"] = { + "kind": "microdata", + "path": projected.name, + "sha256": _sha(projected), + "repo_id": base["repo_id"], + } + for year, key in mapping[base_key].items(): + sha = manifest["artifacts"][key]["sha256"] + evidence["artifacts"][key] = {"year": int(year), "sha256": sha, "rows": rows} + acceptance["years"][year] = { + "dataset": key, + "sha256": sha, + "checks": { + name: "passed" + for name in ( + "schema", + "source_identity", + "year", + "demographics", + "input_aggregates", + "runtime", + ) + }, + } + projection_path = release / "projection_2030.json" + projection_path.write_text( + json.dumps({"base_year": 2024, "year": 2030, "factors": {}}) + ) + evidence["artifacts"]["annual_2030"]["projection_receipt"] = { + "path": projection_path.name, + "sha256": _sha(projection_path), + } + manifest["artifacts"]["projection_2030"] = { + "kind": "diagnostics", + "path": f"releases/{release.name}/{projection_path.name}", + "sha256": _sha(projection_path), + "repo_id": base["repo_id"], + } + evidence_path = release / "annual_manifest.json" + evidence_path.write_text(json.dumps(evidence)) + acceptance["projection_manifest_sha256"] = _sha(evidence_path) + acceptance_path = release / "annual_acceptance.json" + acceptance_path.write_text(json.dumps(acceptance)) + manifest["metadata"] = { + **manifest.get("metadata", {}), + "dataset_years": mapping, + "annual_projection_manifest": "annual_manifest", + "annual_projection_acceptance": "annual_acceptance", + } + for key, path in ( + ("annual_manifest", evidence_path), + ("annual_acceptance", acceptance_path), + ): + manifest["artifacts"][key] = { + "kind": "diagnostics", + "path": f"releases/{release.name}/{path.name}", + "sha256": _sha(path), + "repo_id": base["repo_id"], + } + tag = f"{release.name}-annual-20260919T220000Z-a1b2c3d4" + for artifact in manifest["artifacts"].values(): + artifact["revision"] = tag + (release / "release_manifest.json").write_text(json.dumps(manifest)) + return tag + + +def test_unrelated_release_retains_existing_contract(tmp_path): + validate_annual_projection_extension(tmp_path, {}) + + +def test_accepted_files_match_year_identity_and_hash(candidate): + release, root, manifest, *_ = candidate + result = validate_annual_projection_extension(release, manifest, artifact_root=root) + assert result.revision == f"{release.name}-annual-20260919T220000Z-a1b2c3d4" + assert result.projected_artifacts == {"populace_us_2030"} + assert set(result.additional_artifacts) == { + "annual_manifest", + "annual_acceptance", + "populace_us_2030", + "projection_2030", + } + + +@pytest.mark.parametrize( + "revision", + [ + None, + [], + "release-id", + "release-id-annual-20260919T220000Z-invalid", + "other-annual-20260919T220000Z-a1b2c3d4", + ], +) +def test_annual_extension_requires_immutable_uniform_cut(candidate, revision): + release, root, manifest, *_ = candidate + manifest["artifacts"]["populace_us_2030"]["revision"] = revision + with pytest.raises(ValueError, match="revision|annual cut"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +def test_nested_root_artifact_cannot_be_shadowed_by_release_file(candidate): + release, root, manifest, *_ = candidate + relative = "annual/populace_us_2030.h5" + (root / "annual").mkdir() + (release / "annual").mkdir() + shutil.copyfile(root / "populace_us_2030.h5", root / relative) + shutil.copyfile(root / relative, release / relative) + manifest["artifacts"]["populace_us_2030"]["path"] = relative + with pytest.raises(ValueError, match="exact release prefix"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +@pytest.mark.parametrize( + "problem", + [ + "missing_reference", + "undeclared", + "missing_file", + "bare_path", + "candidate_hash", + "changed_bytes", + "wrong_year", + "duplicate_path", + ], +) +def test_per_year_projection_receipt_must_be_delivered(candidate, problem): + release, root, manifest, evidence, _, save = candidate + record = evidence["artifacts"]["populace_us_2030"] + artifact = manifest["artifacts"]["projection_2030"] + path = release / "projection_2030.json" + if problem == "missing_reference": + del record["projection_receipt"] + elif problem == "undeclared": + del manifest["artifacts"]["projection_2030"] + elif problem == "missing_file": + path.unlink() + elif problem == "bare_path": + artifact["path"] = path.name + elif problem == "candidate_hash": + record["projection_receipt"]["sha256"] = "0" * 64 + elif problem == "changed_bytes": + path.write_text("{}") + elif problem == "wrong_year": + path.write_text(json.dumps({"base_year": 2024, "year": 2031})) + artifact["sha256"] = record["projection_receipt"]["sha256"] = _sha(path) + else: + manifest["artifacts"]["duplicate_receipt"] = dict(artifact) + save() + with pytest.raises(ValueError, match="projection receipt|sha256"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +@pytest.mark.parametrize( + "problem", ["missing_year", "failed_runtime", "changed_sha", "extra_year"] +) +def test_incomplete_or_stale_acceptance_refuses(candidate, problem): + release, root, manifest, _, acceptance, save = candidate + if problem == "missing_year": + del acceptance["years"]["2030"] + elif problem == "failed_runtime": + acceptance["years"]["2030"]["checks"]["runtime"] = "pending" + elif problem == "changed_sha": + acceptance["years"]["2030"]["sha256"] = "0" * 64 + else: + acceptance["years"]["2035"] = acceptance["years"]["2030"] + save() + with pytest.raises(ValueError, match="acceptance"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +@pytest.mark.parametrize("problem", ["year", "person", "negative_weight", "nan_weight"]) +def test_native_content_checks_do_not_trust_receipt_claims(candidate, problem): + release, root, manifest, evidence, acceptance, save = candidate + key = "populace_us_2030" + path = root / f"{key}.h5" + _h5( + path, + 2035 if problem == "year" else 2030, + person_id=2 if problem == "person" else 1, + weight=-1 + if problem == "negative_weight" + else float("nan") + if problem == "nan_weight" + else 10, + ) + sha = _sha(path) + manifest["artifacts"][key]["sha256"] = sha + evidence["artifacts"][key]["sha256"] = sha + acceptance["years"]["2030"]["sha256"] = sha + save() + with pytest.raises(ValueError, match="year|person_id|weights"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +def test_changed_h5_bytes_refuse(candidate): + release, root, manifest, *_ = candidate + with (root / "populace_us_2030.h5").open("ab") as stream: + stream.write(b"changed") + with pytest.raises(ValueError, match="sha256"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +def test_source_enrichment_cannot_skip_annual_checks(candidate): + release, root, manifest, _, _, save = candidate + manifest["release_type"] = "source_enrichment" + manifest["metadata"]["dataset_years"] = None + save() + with pytest.raises(ReleaseContractError, match="dataset_years"): + validate_release_dir(release, artifact_root=root) + + +def test_missing_acceptance_does_not_certify_candidate(candidate): + release, root, manifest, *_ = candidate + del manifest["metadata"]["annual_projection_acceptance"] + with pytest.raises(ValueError, match="acceptance"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +@pytest.mark.parametrize( + "path", ["../outside.h5", "/absolute.h5", "releases/../outside.h5"] +) +def test_noncanonical_artifact_paths_refuse(candidate, path): + release, root, manifest, *_ = candidate + manifest["artifacts"]["populace_us_2030"]["path"] = path + with pytest.raises(ValueError, match="relative path"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +@pytest.mark.parametrize( + "path", ["releases/release-id/nested/2030.h5", "releases/another/2030.h5"] +) +def test_release_paths_must_match_the_publishers_upload_layout(candidate, path): + release, root, manifest, *_ = candidate + manifest["artifacts"]["populace_us_2030"]["path"] = path + with pytest.raises(ValueError, match="bare filenames|another release"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +def test_bare_release_local_path_cannot_publish_at_wrong_location(candidate): + release, root, manifest, *_ = candidate + source = root / "populace_us_2030.h5" + (release / source.name).write_bytes(source.read_bytes()) + with pytest.raises(ValueError, match="exact release prefix"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +@pytest.mark.parametrize( + "problem", + ["other_family", "invented_parent", "wrong_model", "wrong_acceptance_model"], +) +def test_annual_evidence_must_belong_to_certified_base_and_runtime(candidate, problem): + release, root, manifest, evidence, acceptance, save = candidate + if problem == "other_family": + manifest["default_datasets"]["national"] = "another_base" + elif problem == "invented_parent": + evidence["base"]["parent_release"] = "invented-not-a-release" + elif problem == "wrong_model": + manifest["build"]["built_with_model_package"]["version"] = "2.2.1" + else: + acceptance["model"]["source_tree_sha256"] = "c" * 64 + save() + with pytest.raises(ValueError, match="certified national base|model|runtime"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + +@pytest.mark.parametrize("specifier", [None, "", " ", ","]) +def test_empty_compatibility_claim_cannot_admit_unvalidated_model(candidate, specifier): + release, root, manifest, *_ = candidate + manifest["build"]["built_with_model_package"]["version"] = "2.2.1" + manifest["compatible_model_packages"] = [ + {"name": "policyengine-us", "specifier": specifier} + ] + with pytest.raises(ValueError, match="nonempty specifiers"): + validate_annual_projection_extension(release, manifest, artifact_root=root) diff --git a/packages/microcosm-data/tests/test_release.py b/packages/microcosm-data/tests/test_release.py index 527f3a38b..cd76274c8 100644 --- a/packages/microcosm-data/tests/test_release.py +++ b/packages/microcosm-data/tests/test_release.py @@ -647,6 +647,94 @@ def artifact_root(tmp_path: Path) -> Path: return directory +@pytest.fixture +def annual_release(release_dir, artifact_root): + from .test_annual_projections import _h5, add_annual_extension + + dataset = artifact_root / "populace_us_2024.h5" + _h5(dataset, 2024) + sha = _sha256(dataset) + path = release_dir / "build_manifest.json" + build = json.loads(path.read_text()) + build["dataset"]["sha256"] = sha + path.write_text(json.dumps(build)) + path = release_dir / "release_manifest.json" + manifest = json.loads(path.read_text()) + manifest["artifacts"]["populace_us_2024"]["sha256"] = sha + path.write_text(json.dumps(manifest)) + return add_annual_extension(release_dir, artifact_root) + + +def test_annual_cut_passes_real_release_contract( + release_dir, artifact_root, annual_release +): + prepared = release_module.prepare_release( + release_dir, + artifact_root=artifact_root, + tag_name=annual_release, + update_latest=False, + ) + assert prepared.tag == annual_release + assert {"annual_manifest.json", "annual_acceptance.json"}.issubset( + prepared.filenames + ) + assert {"annual_2030.h5", "populace_us_2024.h5"}.issubset(prepared.root_artifacts) + + +def test_annual_cut_uploads_exact_artifacts_without_latest( + hub, release_dir, artifact_root, annual_release +): + publish_release( + release_dir, + "policyengine/populace-us", + api=hub, + artifact_root=artifact_root, + tag_name=annual_release, + update_latest=False, + ) + uploads = dict(hub.uploads) + assert uploads["annual_2030.h5"] == (artifact_root / "annual_2030.h5").read_bytes() + assert ( + uploads["populace_us_2024.h5"] + == (artifact_root / "populace_us_2024.h5").read_bytes() + ) + for name in ( + "annual_manifest.json", + "annual_acceptance.json", + "projection_2030.json", + "release_manifest.json", + ): + assert ( + uploads[f"releases/{release_dir.name}/{name}"] + == (release_dir / name).read_bytes() + ) + assert hub.tags == [{"tag": annual_release, "revision": "commit-1"}] + assert LATEST_POINTER_PATH not in uploads + assert LATEST_EVIDENCE_POINTER_PATH not in uploads + + +@pytest.mark.parametrize( + "option", ["latest", "wrong_tag", "absent_annual", "invalid_base"] +) +def test_annual_cut_preserves_publisher_and_base_guards( + release_dir, artifact_root, annual_release, option +): + path = release_dir / "release_manifest.json" + manifest = json.loads(path.read_text()) + if option == "absent_annual": + del manifest["metadata"] + path.write_text(json.dumps(manifest)) + elif option == "invalid_base": + (release_dir / "calibration_diagnostics.json").write_text("{}") + with pytest.raises((ReleaseContractError, ValueError)): + release_module.prepare_release( + release_dir, + artifact_root=artifact_root, + tag_name=release_dir.name if option == "wrong_tag" else annual_release, + update_latest=option == "latest", + ) + + def test_pointer_payload_names_every_contract_file() -> None: payload = latest_pointer_payload(RELEASE_ID, updated_at="2026-06-11T13:53:15+00:00") assert payload["schema_version"] == LATEST_POINTER_SCHEMA_VERSION @@ -1118,7 +1206,9 @@ def test_per_cut_artifact_revision_publishes_matching_tag_without_latest( for artifact in manifest["artifacts"].values(): artifact["revision"] = cut_tag manifest_path.write_text(json.dumps(manifest)) - monkeypatch.setattr(release_module, "validate_release_dir", lambda _path: None) + monkeypatch.setattr( + release_module, "validate_release_dir", lambda _path, **_kwargs: None + ) publish_release( release_dir, @@ -1145,7 +1235,9 @@ def test_per_cut_artifact_revision_refuses_dangling_default_tag( for artifact in manifest["artifacts"].values(): artifact["revision"] = cut_tag manifest_path.write_text(json.dumps(manifest)) - monkeypatch.setattr(release_module, "validate_release_dir", lambda _path: None) + monkeypatch.setattr( + release_module, "validate_release_dir", lambda _path, **_kwargs: None + ) with pytest.raises(ValueError, match="uniform per-cut artifact revision"): publish_release( @@ -1172,7 +1264,9 @@ def test_unreadable_artifact_revisions_refuse_instead_of_vanishing( for artifact in manifest["artifacts"].values(): artifact["revision"] = 123 manifest_path.write_text(json.dumps(manifest)) - monkeypatch.setattr(release_module, "validate_release_dir", lambda _path: None) + monkeypatch.setattr( + release_module, "validate_release_dir", lambda _path, **_kwargs: None + ) with pytest.raises(ValueError, match="missing or non-string revisions"): publish_release( @@ -1195,7 +1289,9 @@ def test_empty_artifact_revisions_refuse_publication( manifest = json.loads(manifest_path.read_text()) manifest["artifacts"] = {} manifest_path.write_text(json.dumps(manifest)) - monkeypatch.setattr(release_module, "validate_release_dir", lambda _path: None) + monkeypatch.setattr( + release_module, "validate_release_dir", lambda _path, **_kwargs: None + ) with pytest.raises(ValueError, match="declares no artifact revisions"): publish_release( @@ -1223,7 +1319,9 @@ def test_per_cut_tag_refuses_latest_promotion( for artifact in manifest["artifacts"].values(): artifact["revision"] = cut_tag manifest_path.write_text(json.dumps(manifest)) - monkeypatch.setattr(release_module, "validate_release_dir", lambda _path: None) + monkeypatch.setattr( + release_module, "validate_release_dir", lambda _path, **_kwargs: None + ) with pytest.raises(ValueError, match="inspect-only"): publish_release( diff --git a/packages/microcosm-data/tests/test_source_enrichment.py b/packages/microcosm-data/tests/test_source_enrichment.py index 333a37f93..07c6c6751 100644 --- a/packages/microcosm-data/tests/test_source_enrichment.py +++ b/packages/microcosm-data/tests/test_source_enrichment.py @@ -37,6 +37,7 @@ def candidate(tmp_path, monkeypatch): } ) with pd.HDFStore(parent, "w") as store: + store.put("_time_period", pd.Series([2024]), format="table") store.put("person", people, format="table", data_columns=True) for entity in ("household", "spm_unit", "tax_unit", "family", "marital_unit"): store.put( @@ -383,6 +384,79 @@ def actual_test(*args, **kwargs): return output, calls +def test_annual_cut_preserves_source_enrichment_qualification( + candidate, tmp_path, monkeypatch +): + from microcosm.data.release import prepare_release + + from .test_annual_projections import add_annual_extension + + output, calls = _qualify_candidate(candidate, tmp_path, monkeypatch) + _, parent, root = candidate + tag = add_annual_extension(output, root) + previous_calls = len(calls) + prepared = prepare_release( + output, + artifact_root=root, + parent_h5=parent, + compatibility_wheels=(tmp_path / "country.whl",), + tag_name=tag, + update_latest=False, + ) + assert len(calls) == previous_calls + 1 + assert calls[-1]["require_wheels"] is True + assert { + "annual_manifest.json", + "annual_acceptance.json", + "projection_2030.json", + enrichment.COMPATIBILITY_FILE, + }.issubset(prepared.filenames) + assert set(prepared.root_artifacts) == {"populace_us_2024.h5", "annual_2030.h5"} + + +@pytest.mark.parametrize( + "problem", ["base_path", "extra_microdata", "missing_parent", "absent_annual"] +) +def test_annual_source_enrichment_retains_original_constraints( + candidate, tmp_path, monkeypatch, problem +): + from .test_annual_projections import add_annual_extension + + output, _ = _qualify_candidate(candidate, tmp_path, monkeypatch) + _, parent, root = candidate + add_annual_extension(output, root) + path = output / "release_manifest.json" + manifest = json.loads(path.read_text()) + if problem == "base_path": + key, artifact = next( + (key, value) + for key, value in manifest["artifacts"].items() + if value["path"] == enrichment.SOURCE_EVIDENCE_FILE + ) + manifest["artifacts"][key]["path"] = ( + f"releases/{output.name}/{artifact['path']}" + ) + elif problem == "extra_microdata": + manifest["artifacts"]["extra"] = { + **manifest["artifacts"]["dataset"], + "path": "extra.h5", + } + shutil.copyfile(root / "populace_us_2024.h5", output / "extra.h5") + elif problem == "missing_parent": + (output / "parent_build_manifest.json").unlink() + else: + del manifest["metadata"] + path.write_text(json.dumps(manifest)) + with pytest.raises(ReleaseContractError): + enrichment.validate_source_enrichment_candidate( + output, + parent_h5=parent, + artifact_root=root, + require_compatibility=True, + compatibility_wheels=(tmp_path / "country.whl",), + ) + + def test_certification_writes_new_bundle_and_preflight_replays( candidate, tmp_path, monkeypatch ): From 668160356e747a47913ef5b95d383ea77d6c114a Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 19 Sep 2026 18:08:18 -0400 Subject: [PATCH 4/5] Require complete annual input history in release families --- docs/us-annual-static-aging.md | 4 +- .../src/microcosm/data/annual_projections.py | 11 +- .../tests/test_annual_projections.py | 108 ++++++++++++------ packages/microcosm-data/tests/test_release.py | 6 +- .../tests/test_source_enrichment.py | 4 +- 5 files changed, 88 insertions(+), 45 deletions(-) diff --git a/docs/us-annual-static-aging.md b/docs/us-annual-static-aging.md index 538b4aaf6..554d8eacc 100644 --- a/docs/us-annual-static-aging.md +++ b/docs/us-annual-static-aging.md @@ -84,7 +84,9 @@ publication decision; this module supplies no certification override. paths at the repository root; give both reports and every per-year projection receipt their exact `releases//filename.json` paths. The gate verifies each per-year receipt's hash and years against - the candidate manifest. Pin all artifacts to one + the candidate manifest and requires every year from the base through the + last declared year, so policy lookbacks retain their annual inputs. + Pin all artifacts to one `-annual--` tag and their exact hashes. Complete source-enrichment certification before this step: that operation writes base-tag compatibility metadata. diff --git a/packages/microcosm-data/src/microcosm/data/annual_projections.py b/packages/microcosm-data/src/microcosm/data/annual_projections.py index 25e3786c6..d6f19bba1 100644 --- a/packages/microcosm-data/src/microcosm/data/annual_projections.py +++ b/packages/microcosm-data/src/microcosm/data/annual_projections.py @@ -308,10 +308,17 @@ def validate_annual_projection_extension( years = _object(mapping, f"annual family {family!r}") if not years or years.get(str(base_year)) != base_key: raise ValueError("annual family must include its unchanged base year") + if any( + not isinstance(year, str) or not _YEAR.fullmatch(year) for year in years + ): + raise ValueError("annual year keys must be four-digit decimal strings") + declared_years = {int(year) for year in years} + if declared_years != set(range(base_year, max(declared_years) + 1)): + raise ValueError( + "annual family must cover every year from its base through its last year" + ) seen: set[str] = set() for year_text, key in years.items(): - if not isinstance(year_text, str) or not _YEAR.fullmatch(year_text): - raise ValueError("annual year keys must be four-digit decimal strings") if not isinstance(key, str) or key in seen or int(year_text) < base_year: raise ValueError( "annual artifacts must be unique and no earlier than their source" diff --git a/packages/microcosm-data/tests/test_annual_projections.py b/packages/microcosm-data/tests/test_annual_projections.py index 50388052b..9a29dd7fd 100644 --- a/packages/microcosm-data/tests/test_annual_projections.py +++ b/packages/microcosm-data/tests/test_annual_projections.py @@ -54,7 +54,7 @@ def candidate(tmp_path): root = tmp_path / "artifacts" root.mkdir() mapping = { - "populace_us_2024": {"2024": "populace_us_2024", "2030": "populace_us_2030"} + "populace_us_2024": {"2024": "populace_us_2024", "2025": "populace_us_2025"} } manifest = { "metadata": {"dataset_years": mapping}, @@ -100,7 +100,7 @@ def candidate(tmp_path): "model": dict(evidence["model"]), "runtime": dict(evidence["runtime"]["versions"]), } - for year in (2024, 2030): + for year in (2024, 2025): key = f"populace_us_{year}" path = root / f"{key}.h5" _h5(path, year) @@ -142,15 +142,15 @@ def candidate(tmp_path): }, } evidence["base"]["sha256"] = manifest["artifacts"]["populace_us_2024"]["sha256"] - projection_path = release / "projection_2030.json" + projection_path = release / "projection_2025.json" projection_path.write_text( - json.dumps({"base_year": 2024, "year": 2030, "factors": {}}) + json.dumps({"base_year": 2024, "year": 2025, "factors": {}}) ) - evidence["artifacts"]["populace_us_2030"]["projection_receipt"] = { + evidence["artifacts"]["populace_us_2025"]["projection_receipt"] = { "path": projection_path.name, "sha256": _sha(projection_path), } - manifest["artifacts"]["projection_2030"] = { + manifest["artifacts"]["projection_2025"] = { "path": f"releases/{release.name}/{projection_path.name}", "sha256": _sha(projection_path), "revision": f"{release.name}-annual-20260919T220000Z-a1b2c3d4", @@ -184,11 +184,11 @@ def add_annual_extension(release, root): manifest = json.loads((release / "release_manifest.json").read_text()) base_key = manifest["default_datasets"]["national"] base = manifest["artifacts"][base_key] - projected = root / "annual_2030.h5" + projected = root / "annual_2025.h5" shutil.copyfile(root / base["path"], projected) with h5py.File(projected, "r+") as store: row = store["_time_period/table"][:] - row["values"] = 2030 + row["values"] = 2025 store["_time_period/table"][:] = row rows = { entity: len(store[f"{entity}/table"]) @@ -210,7 +210,7 @@ def add_annual_extension(release, root): "commit": "a" * 40, "source_tree_sha256": "b" * 64, } - mapping = {base_key: {"2024": base_key, "2030": "annual_2030"}} + mapping = {base_key: {"2024": base_key, "2025": "annual_2025"}} evidence = { "schema_version": 1, "kind": "us_annual_static_aging_candidate", @@ -234,7 +234,7 @@ def add_annual_extension(release, root): "runtime": runtime, "years": {}, } - manifest["artifacts"]["annual_2030"] = { + manifest["artifacts"]["annual_2025"] = { "kind": "microdata", "path": projected.name, "sha256": _sha(projected), @@ -258,15 +258,15 @@ def add_annual_extension(release, root): ) }, } - projection_path = release / "projection_2030.json" + projection_path = release / "projection_2025.json" projection_path.write_text( - json.dumps({"base_year": 2024, "year": 2030, "factors": {}}) + json.dumps({"base_year": 2024, "year": 2025, "factors": {}}) ) - evidence["artifacts"]["annual_2030"]["projection_receipt"] = { + evidence["artifacts"]["annual_2025"]["projection_receipt"] = { "path": projection_path.name, "sha256": _sha(projection_path), } - manifest["artifacts"]["projection_2030"] = { + manifest["artifacts"]["projection_2025"] = { "kind": "diagnostics", "path": f"releases/{release.name}/{projection_path.name}", "sha256": _sha(projection_path), @@ -308,15 +308,49 @@ def test_accepted_files_match_year_identity_and_hash(candidate): release, root, manifest, *_ = candidate result = validate_annual_projection_extension(release, manifest, artifact_root=root) assert result.revision == f"{release.name}-annual-20260919T220000Z-a1b2c3d4" - assert result.projected_artifacts == {"populace_us_2030"} + assert result.projected_artifacts == {"populace_us_2025"} assert set(result.additional_artifacts) == { "annual_manifest", "annual_acceptance", - "populace_us_2030", - "projection_2030", + "populace_us_2025", + "projection_2025", } +def test_annual_family_rejects_missing_middle_year(candidate): + release, root, manifest, evidence, acceptance, save = candidate + old_key, key = "populace_us_2025", "populace_us_2026" + mapping = manifest["metadata"]["dataset_years"]["populace_us_2024"] + del mapping["2025"] + mapping["2026"] = key + path = root / f"{key}.h5" + _h5(path, 2026) + sha = _sha(path) + artifact = manifest["artifacts"][key] = manifest["artifacts"].pop(old_key) + artifact.update(path=path.name, sha256=sha) + record = evidence["artifacts"][key] = evidence["artifacts"].pop(old_key) + record.update(year=2026, sha256=sha) + receipt_path = release / "projection_2026.json" + receipt_path.write_text( + json.dumps({"base_year": 2024, "year": 2026, "factors": {}}) + ) + record["projection_receipt"] = { + "path": receipt_path.name, + "sha256": _sha(receipt_path), + } + receipt_artifact = manifest["artifacts"]["projection_2026"] = manifest[ + "artifacts" + ].pop("projection_2025") + receipt_artifact.update( + path=f"releases/{release.name}/{receipt_path.name}", sha256=_sha(receipt_path) + ) + accepted = acceptance["years"]["2026"] = acceptance["years"].pop("2025") + accepted.update(dataset=key, sha256=sha) + save() + with pytest.raises(ValueError, match="cover every year"): + validate_annual_projection_extension(release, manifest, artifact_root=root) + + @pytest.mark.parametrize( "revision", [ @@ -329,19 +363,19 @@ def test_accepted_files_match_year_identity_and_hash(candidate): ) def test_annual_extension_requires_immutable_uniform_cut(candidate, revision): release, root, manifest, *_ = candidate - manifest["artifacts"]["populace_us_2030"]["revision"] = revision + manifest["artifacts"]["populace_us_2025"]["revision"] = revision with pytest.raises(ValueError, match="revision|annual cut"): validate_annual_projection_extension(release, manifest, artifact_root=root) def test_nested_root_artifact_cannot_be_shadowed_by_release_file(candidate): release, root, manifest, *_ = candidate - relative = "annual/populace_us_2030.h5" + relative = "annual/populace_us_2025.h5" (root / "annual").mkdir() (release / "annual").mkdir() - shutil.copyfile(root / "populace_us_2030.h5", root / relative) + shutil.copyfile(root / "populace_us_2025.h5", root / relative) shutil.copyfile(root / relative, release / relative) - manifest["artifacts"]["populace_us_2030"]["path"] = relative + manifest["artifacts"]["populace_us_2025"]["path"] = relative with pytest.raises(ValueError, match="exact release prefix"): validate_annual_projection_extension(release, manifest, artifact_root=root) @@ -361,13 +395,13 @@ def test_nested_root_artifact_cannot_be_shadowed_by_release_file(candidate): ) def test_per_year_projection_receipt_must_be_delivered(candidate, problem): release, root, manifest, evidence, _, save = candidate - record = evidence["artifacts"]["populace_us_2030"] - artifact = manifest["artifacts"]["projection_2030"] - path = release / "projection_2030.json" + record = evidence["artifacts"]["populace_us_2025"] + artifact = manifest["artifacts"]["projection_2025"] + path = release / "projection_2025.json" if problem == "missing_reference": del record["projection_receipt"] elif problem == "undeclared": - del manifest["artifacts"]["projection_2030"] + del manifest["artifacts"]["projection_2025"] elif problem == "missing_file": path.unlink() elif problem == "bare_path": @@ -392,13 +426,13 @@ def test_per_year_projection_receipt_must_be_delivered(candidate, problem): def test_incomplete_or_stale_acceptance_refuses(candidate, problem): release, root, manifest, _, acceptance, save = candidate if problem == "missing_year": - del acceptance["years"]["2030"] + del acceptance["years"]["2025"] elif problem == "failed_runtime": - acceptance["years"]["2030"]["checks"]["runtime"] = "pending" + acceptance["years"]["2025"]["checks"]["runtime"] = "pending" elif problem == "changed_sha": - acceptance["years"]["2030"]["sha256"] = "0" * 64 + acceptance["years"]["2025"]["sha256"] = "0" * 64 else: - acceptance["years"]["2035"] = acceptance["years"]["2030"] + acceptance["years"]["2035"] = acceptance["years"]["2025"] save() with pytest.raises(ValueError, match="acceptance"): validate_annual_projection_extension(release, manifest, artifact_root=root) @@ -407,11 +441,11 @@ def test_incomplete_or_stale_acceptance_refuses(candidate, problem): @pytest.mark.parametrize("problem", ["year", "person", "negative_weight", "nan_weight"]) def test_native_content_checks_do_not_trust_receipt_claims(candidate, problem): release, root, manifest, evidence, acceptance, save = candidate - key = "populace_us_2030" + key = "populace_us_2025" path = root / f"{key}.h5" _h5( path, - 2035 if problem == "year" else 2030, + 2035 if problem == "year" else 2025, person_id=2 if problem == "person" else 1, weight=-1 if problem == "negative_weight" @@ -422,7 +456,7 @@ def test_native_content_checks_do_not_trust_receipt_claims(candidate, problem): sha = _sha(path) manifest["artifacts"][key]["sha256"] = sha evidence["artifacts"][key]["sha256"] = sha - acceptance["years"]["2030"]["sha256"] = sha + acceptance["years"]["2025"]["sha256"] = sha save() with pytest.raises(ValueError, match="year|person_id|weights"): validate_annual_projection_extension(release, manifest, artifact_root=root) @@ -430,7 +464,7 @@ def test_native_content_checks_do_not_trust_receipt_claims(candidate, problem): def test_changed_h5_bytes_refuse(candidate): release, root, manifest, *_ = candidate - with (root / "populace_us_2030.h5").open("ab") as stream: + with (root / "populace_us_2025.h5").open("ab") as stream: stream.write(b"changed") with pytest.raises(ValueError, match="sha256"): validate_annual_projection_extension(release, manifest, artifact_root=root) @@ -457,24 +491,24 @@ def test_missing_acceptance_does_not_certify_candidate(candidate): ) def test_noncanonical_artifact_paths_refuse(candidate, path): release, root, manifest, *_ = candidate - manifest["artifacts"]["populace_us_2030"]["path"] = path + manifest["artifacts"]["populace_us_2025"]["path"] = path with pytest.raises(ValueError, match="relative path"): validate_annual_projection_extension(release, manifest, artifact_root=root) @pytest.mark.parametrize( - "path", ["releases/release-id/nested/2030.h5", "releases/another/2030.h5"] + "path", ["releases/release-id/nested/2025.h5", "releases/another/2025.h5"] ) def test_release_paths_must_match_the_publishers_upload_layout(candidate, path): release, root, manifest, *_ = candidate - manifest["artifacts"]["populace_us_2030"]["path"] = path + manifest["artifacts"]["populace_us_2025"]["path"] = path with pytest.raises(ValueError, match="bare filenames|another release"): validate_annual_projection_extension(release, manifest, artifact_root=root) def test_bare_release_local_path_cannot_publish_at_wrong_location(candidate): release, root, manifest, *_ = candidate - source = root / "populace_us_2030.h5" + source = root / "populace_us_2025.h5" (release / source.name).write_bytes(source.read_bytes()) with pytest.raises(ValueError, match="exact release prefix"): validate_annual_projection_extension(release, manifest, artifact_root=root) diff --git a/packages/microcosm-data/tests/test_release.py b/packages/microcosm-data/tests/test_release.py index cd76274c8..2a0027863 100644 --- a/packages/microcosm-data/tests/test_release.py +++ b/packages/microcosm-data/tests/test_release.py @@ -678,7 +678,7 @@ def test_annual_cut_passes_real_release_contract( assert {"annual_manifest.json", "annual_acceptance.json"}.issubset( prepared.filenames ) - assert {"annual_2030.h5", "populace_us_2024.h5"}.issubset(prepared.root_artifacts) + assert {"annual_2025.h5", "populace_us_2024.h5"}.issubset(prepared.root_artifacts) def test_annual_cut_uploads_exact_artifacts_without_latest( @@ -693,7 +693,7 @@ def test_annual_cut_uploads_exact_artifacts_without_latest( update_latest=False, ) uploads = dict(hub.uploads) - assert uploads["annual_2030.h5"] == (artifact_root / "annual_2030.h5").read_bytes() + assert uploads["annual_2025.h5"] == (artifact_root / "annual_2025.h5").read_bytes() assert ( uploads["populace_us_2024.h5"] == (artifact_root / "populace_us_2024.h5").read_bytes() @@ -701,7 +701,7 @@ def test_annual_cut_uploads_exact_artifacts_without_latest( for name in ( "annual_manifest.json", "annual_acceptance.json", - "projection_2030.json", + "projection_2025.json", "release_manifest.json", ): assert ( diff --git a/packages/microcosm-data/tests/test_source_enrichment.py b/packages/microcosm-data/tests/test_source_enrichment.py index 07c6c6751..d068f59b1 100644 --- a/packages/microcosm-data/tests/test_source_enrichment.py +++ b/packages/microcosm-data/tests/test_source_enrichment.py @@ -408,10 +408,10 @@ def test_annual_cut_preserves_source_enrichment_qualification( assert { "annual_manifest.json", "annual_acceptance.json", - "projection_2030.json", + "projection_2025.json", enrichment.COMPATIBILITY_FILE, }.issubset(prepared.filenames) - assert set(prepared.root_artifacts) == {"populace_us_2024.h5", "annual_2030.h5"} + assert set(prepared.root_artifacts) == {"populace_us_2024.h5", "annual_2025.h5"} @pytest.mark.parametrize( From e7bccc359bf06345cdd7107c10174b7ffdb4612f Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 19 Sep 2026 21:43:38 -0400 Subject: [PATCH 5/5] Register annual HDF serializer boundaries --- .../build/frame_serializer_registry.py | 11 ++++ .../tests/test_frame_serializer_registry.py | 50 +++++++++++++++++-- .../src/microcosm/data/annual_projections.py | 5 +- 3 files changed, 61 insertions(+), 5 deletions(-) diff --git a/packages/microcosm-build/src/microcosm/build/frame_serializer_registry.py b/packages/microcosm-build/src/microcosm/build/frame_serializer_registry.py index 2367fa7b0..5451afcd9 100644 --- a/packages/microcosm-build/src/microcosm/build/frame_serializer_registry.py +++ b/packages/microcosm-build/src/microcosm/build/frame_serializer_registry.py @@ -111,6 +111,17 @@ class HdfWriteExclusion: version_owner="PolicyEngineUSAdapter payload contract", nullable_boolean_storage="numpy_bool_or_object_pd_na_v1", ), + FrameSerializerSpec( + serializer_id="us_annual_static_aging", + writer=HdfWriteSite( + "packages/microcosm-build/src/microcosm/build/us_annual_static_aging.py", + "_write_year", + ), + backend="pandas.HDFStore table with direct fields", + routes=("US annual static-aging candidate",), + version_owner="schema-1 annual static-aging candidate native layout", + nullable_boolean_storage="numpy_bool_missing_rejected_v1", + ), FrameSerializerSpec( serializer_id="legacy_us_two_spine", writer=HdfWriteSite( diff --git a/packages/microcosm-build/tests/test_frame_serializer_registry.py b/packages/microcosm-build/tests/test_frame_serializer_registry.py index 11edf2aeb..46bf4ca79 100644 --- a/packages/microcosm-build/tests/test_frame_serializer_registry.py +++ b/packages/microcosm-build/tests/test_frame_serializer_registry.py @@ -69,6 +69,7 @@ def _dtype_family_table( missing_values = { "mixed": [True, pd.NA, False], "all_missing": [pd.NA, pd.NA, pd.NA], + "complete": [True, False, True], }[nullable_case] return pd.DataFrame( { @@ -350,12 +351,36 @@ def _round_trip_fiscal_checkpoint( ) +def _round_trip_us_annual_static_aging( + tmp_path: Path, nullable_case: str +) -> BooleanRoundTrip: + pytest.importorskip("policyengine_us") + from microcosm.build.us_annual_static_aging import _write_year + + source = _dtype_family_table(nullable_case) + before = source.copy(deep=True) + frame = _us_frame(source) + path = tmp_path / "annual.h5" + try: + _write_year( + path, {entity: frame.table(entity) for entity in frame.entities}, 2025 + ) + finally: + pd.testing.assert_frame_equal( + source, before, check_exact=True, check_dtype=True + ) + with pd.HDFStore(path, "r") as store: + loaded = read_frame_table(store, "person") + return _semantic_observation(source, before, loaded) + + ROUND_TRIP_ADAPTERS: dict[str, RoundTripAdapter] = { "frame_checkpoint": _round_trip_frame_checkpoint, "nullable_us_h5": _round_trip_nullable_us_h5, "uk_single_year_h5": _round_trip_uk_single_year, "axiom_entity_tables": _round_trip_axiom, "policyengine_us_single_year": _round_trip_policyengine_us, + "us_annual_static_aging": _round_trip_us_annual_static_aging, "legacy_us_two_spine": _round_trip_legacy_us, "acs_local_lean_checkpoint": _round_trip_acs_lean, "fiscal_target_frame_checkpoint": _round_trip_fiscal_checkpoint, @@ -428,10 +453,10 @@ def test_registry_classifies_every_writable_production_hdf_site() -> None: assert _discover_writable_hdf_sites() == classified -def test_registry_has_exactly_eight_unique_frame_table_serializers() -> None: - assert len(FRAME_TABLE_SERIALIZERS) == 8 - assert len({spec.serializer_id for spec in FRAME_TABLE_SERIALIZERS}) == 8 - assert len({spec.writer.key for spec in FRAME_TABLE_SERIALIZERS}) == 8 +def test_registry_has_exactly_nine_unique_frame_table_serializers() -> None: + assert len(FRAME_TABLE_SERIALIZERS) == 9 + assert len({spec.serializer_id for spec in FRAME_TABLE_SERIALIZERS}) == 9 + assert len({spec.writer.key for spec in FRAME_TABLE_SERIALIZERS}) == 9 def test_round_trip_adapter_registry_exactly_matches_serializer_registry() -> None: @@ -451,6 +476,13 @@ def test_registered_serializer_round_trips_nullable_boolean_dtype_family( nullable_case: str, tmp_path: Path, ) -> None: + if serializer.nullable_boolean_storage == "numpy_bool_missing_rejected_v1": + with pytest.raises( + ValueError, match="Annual native layout requires table-format" + ): + ROUND_TRIP_ADAPTERS[serializer.serializer_id](tmp_path, nullable_case) + assert not (tmp_path / "annual.h5").exists() + return observation = ROUND_TRIP_ADAPTERS[serializer.serializer_id](tmp_path, nullable_case) # Serializers may materialize a boundary copy, never rewrite the source. @@ -503,6 +535,16 @@ def test_registered_serializer_round_trips_nullable_boolean_dtype_family( assert all(value is pd.NA for value in missing_scalars) +def test_annual_serializer_preserves_supported_complete_boolean_columns(tmp_path): + observation = _round_trip_us_annual_static_aging(tmp_path, "complete") + for column in (NATIVE_COLUMN, COMPLETE_COLUMN, MISSING_COLUMN): + assert observation.loaded[column].dtype == np.dtype(np.bool_) + np.testing.assert_array_equal( + observation.loaded[column], + observation.source[column].to_numpy(dtype=np.bool_), + ) + + def test_policyengine_us_adapter_owns_its_registered_hdf_boundary() -> None: (spec,) = ( candidate diff --git a/packages/microcosm-data/src/microcosm/data/annual_projections.py b/packages/microcosm-data/src/microcosm/data/annual_projections.py index d6f19bba1..f0a929d9b 100644 --- a/packages/microcosm-data/src/microcosm/data/annual_projections.py +++ b/packages/microcosm-data/src/microcosm/data/annual_projections.py @@ -110,7 +110,10 @@ def _native_year(h5: h5py.File) -> int: def _check_native_identity( base: Path, projected: Path, year: int, record: Mapping ) -> None: - with h5py.File(base) as original, h5py.File(projected) as annual: + with ( + h5py.File(base, mode="r") as original, + h5py.File(projected, mode="r") as annual, + ): if _native_year(annual) != year: raise ValueError(f"annual H5 stored year does not match {year}") rows = _object(record.get("rows"), f"{year} row counts")