From aa21c8bd489d829bd264144b9aae47fd97eb36e7 Mon Sep 17 00:00:00 2001 From: Vahid Ahmadi Date: Wed, 16 Sep 2026 10:57:57 +0100 Subject: [PATCH 1/7] Estimate variance from replicate weights Recomputing a statistic once per replicate weight vector and scaling the spread by the factor for the replication scheme gives a variance estimate that needs no analytic formula. That makes it work for every estimator here, including the Gini coefficient and quantiles, where the analytic variance is awkward enough that users leave for R. Supports jackknife, BRR, Fay's BRR, bootstrap and the successive-difference scheme used for the ACS and CPS. Valid only for replicate weights as published with a survey: weights calibrated to external targets no longer correspond to the original replication scheme, and the docstrings say so. Closes #319 Co-Authored-By: Claude Opus 5 (1M context) --- build/lib/microdf/__init__.py | 26 + build/lib/microdf/microdataframe.py | 1021 +++++++++++++++++ build/lib/microdf/microseries.py | 983 ++++++++++++++++ build/lib/microdf/replication.py | 123 ++ build/lib/microdf/tests/conftest.py | 9 + .../microdf/tests/test_aggregation_errors.py | 57 + .../tests/test_dataframe_weight_storage.py | 61 + .../tests/test_microseries_dataframe.py | 817 +++++++++++++ .../tests/test_nullify_weights_index.py | 10 + .../tests/test_pandas3_compatibility.py | 242 ++++ .../tests/test_quantile_missing_values.py | 134 +++ build/lib/microdf/tests/test_replication.py | 110 ++ build/lib/microdf/tests/test_serialization.py | 143 +++ .../microdf/tests/test_version_metadata.py | 8 + .../replicate-weight-variance.added.md | 1 + microdf/__init__.py | 4 + microdf/microseries.py | 31 + microdf/replication.py | 123 ++ microdf/tests/test_replication.py | 110 ++ uv.lock | 20 +- 20 files changed, 4023 insertions(+), 10 deletions(-) create mode 100644 build/lib/microdf/__init__.py create mode 100644 build/lib/microdf/microdataframe.py create mode 100644 build/lib/microdf/microseries.py create mode 100644 build/lib/microdf/replication.py create mode 100644 build/lib/microdf/tests/conftest.py create mode 100644 build/lib/microdf/tests/test_aggregation_errors.py create mode 100644 build/lib/microdf/tests/test_dataframe_weight_storage.py create mode 100644 build/lib/microdf/tests/test_microseries_dataframe.py create mode 100644 build/lib/microdf/tests/test_nullify_weights_index.py create mode 100644 build/lib/microdf/tests/test_pandas3_compatibility.py create mode 100644 build/lib/microdf/tests/test_quantile_missing_values.py create mode 100644 build/lib/microdf/tests/test_replication.py create mode 100644 build/lib/microdf/tests/test_serialization.py create mode 100644 build/lib/microdf/tests/test_version_metadata.py create mode 100644 changelog.d/replicate-weight-variance.added.md create mode 100644 microdf/replication.py create mode 100644 microdf/tests/test_replication.py diff --git a/build/lib/microdf/__init__.py b/build/lib/microdf/__init__.py new file mode 100644 index 00000000..2e2d83a6 --- /dev/null +++ b/build/lib/microdf/__init__.py @@ -0,0 +1,26 @@ +from importlib.metadata import PackageNotFoundError, version + +from .microdataframe import MicroDataFrame, MicroDataFrameGroupBy +from .microseries import MicroSeries, MicroSeriesGroupBy +from .replication import replicate_standard_error, replicate_variance + +name = "microdf" + +# Read the version from package metadata so it can't drift from +# pyproject.toml (the automated bump only touches pyproject). +try: + __version__ = version("microdf-python") +except PackageNotFoundError: # pragma: no cover - running from a source tree + __version__ = "unknown" + +__all__ = [ + # microseries.py + "MicroSeries", + "MicroSeriesGroupBy", + # microdataframe.py + "MicroDataFrame", + "MicroDataFrameGroupBy", + # replication.py + "replicate_variance", + "replicate_standard_error", +] diff --git a/build/lib/microdf/microdataframe.py b/build/lib/microdf/microdataframe.py new file mode 100644 index 00000000..707c3bd7 --- /dev/null +++ b/build/lib/microdf/microdataframe.py @@ -0,0 +1,1021 @@ +import copy +import logging +import warnings +from functools import wraps +from typing import Callable, List, Optional, Union + +import numpy as np +import pandas as pd + +from microdf.microseries import MicroSeries, MicroSeriesGroupBy + +logger = logging.getLogger(__name__) + + +class _MicroLocIndexer: + """Custom loc indexer that returns MicroDataFrame with proper weights.""" + + def __init__(self, mdf: "MicroDataFrame"): + self._mdf = mdf + # Get the parent's loc indexer + self._parent_loc = pd.DataFrame.loc.fget(mdf) + + def __getitem__(self, key): + # Use the parent DataFrame's loc indexer + result = self._parent_loc[key] + + if isinstance(result, pd.DataFrame): + # Get the filtered weights based on the result's index + new_weights = self._mdf.weights.reindex(result.index) + return MicroDataFrame(result, weights=new_weights) + elif isinstance(result, pd.Series): + # Single row or column selected + if result.name in self._mdf.columns: + # Column was selected - return MicroSeries with all weights + return MicroSeries(result, weights=self._mdf.weights) + else: + # Row was selected - return as-is (scalar values for each col) + return result + else: + # Scalar value + return result + + def __setitem__(self, key, value): + self._parent_loc[key] = value + self._mdf._link_all_weights() + + def __getattr__(self, name): + """Delegate unknown attributes to the parent loc indexer.""" + return getattr(self._parent_loc, name) + + +class _MicroILocIndexer: + """Custom iloc indexer that returns MicroDataFrame with proper weights.""" + + def __init__(self, mdf: "MicroDataFrame"): + self._mdf = mdf + # Get the parent's iloc indexer + self._parent_iloc = pd.DataFrame.iloc.fget(mdf) + + def __getitem__(self, key): + # Use the parent DataFrame's iloc indexer + result = self._parent_iloc[key] + + if isinstance(result, pd.DataFrame): + # Get the filtered weights based on the result's index + new_weights = self._mdf.weights.iloc[ + self._mdf.index.get_indexer(result.index) + ] + new_weights = pd.Series(new_weights.values, index=result.index) + return MicroDataFrame(result, weights=new_weights) + elif isinstance(result, pd.Series): + # Single row or column selected + if isinstance(key, tuple) and len(key) == 2: + # df.iloc[:, col_idx] - column selection + row_key = key[0] + if isinstance(row_key, slice) and row_key == slice(None): + # All rows selected for a column + return MicroSeries(result, weights=self._mdf.weights) + # Check if this is a column (result index matches mdf index) + if result.index.equals(self._mdf.index): + return MicroSeries(result, weights=self._mdf.weights) + # Row selection - return as-is + return result + else: + # Scalar value + return result + + def __setitem__(self, key, value): + self._parent_iloc[key] = value + self._mdf._link_all_weights() + + def __getattr__(self, name): + """Delegate unknown attributes to the parent iloc indexer.""" + return getattr(self._parent_iloc, name) + + +class MicroDataFrame(pd.DataFrame): + # Declare weight state as pandas metadata. pandas includes + # _metadata attributes in the pickle state, so weights now survive + # pickling, to_pickle/read_pickle and copy.deepcopy instead of + # vanishing and leaving an AttributeError on the next aggregation. + # Retain the column name for set_weights(..., preserve_old=True). + _metadata = pd.DataFrame._metadata + ["weights", "weights_col"] + + def __init__(self, *args, weights=None, **kwargs): + """A DataFrame-inheriting class for weighted microdata. + + Weights can be provided at initialisation, or using set_weights or + set_weight_col. + + :param weights: Array of weights. + :type weights: np.array + """ + super().__init__(*args, **kwargs) + self.weights = None + self.set_weights(weights) + self._link_all_weights() + self.override_df_functions() + + def __finalize__(self, other, method=None, **kwargs) -> "MicroDataFrame": + """Retain copied weights when pandas finalizes a renamed result.""" + copied_weights = getattr(self, "weights", None) if method == "rename" else None + super().__finalize__(other, method=method, **kwargs) + if copied_weights is not None: + # rename already called copy(); metadata propagation must not + # replace those weights with the source's mutable Series. + self.weights = copied_weights + return self + + def __setstate__(self, state) -> None: + """Restore a pickled MicroDataFrame. + + The weighted aggregations are installed as per-instance closures by + ``override_df_functions``, which only runs in ``__init__`` — a path + unpickling skips. Without reinstalling them, ``mdf.sum()`` on an + unpickled frame silently fell through to the unweighted pandas + implementation. + """ + super().__setstate__(state) + if getattr(self, "weights", None) is None: + self._link_all_weights() + self.override_df_functions() + + @property + def loc(self) -> _MicroLocIndexer: + """Label-based indexer that preserves MicroDataFrame type and weights. + + :return: Custom loc indexer for MicroDataFrame + """ + return _MicroLocIndexer(self) + + @property + def iloc(self) -> _MicroILocIndexer: + """Integer-based indexer that preserves MicroDataFrame type and + weights. + + :return: Custom iloc indexer for MicroDataFrame + """ + return _MicroILocIndexer(self) + + def override_df_functions(self) -> None: + """Override DataFrame functions to work with weighted operations.""" + for name in MicroSeries.FUNCTIONS: + if name in MicroSeries.SCALAR_FUNCTIONS: + setattr(self, name, self._create_scalar_function(name)) + elif name in MicroSeries.VECTOR_FUNCTIONS: + setattr(self, name, self._create_vector_function(name)) + elif name in MicroSeries.AGNOSTIC_FUNCTIONS: + setattr(self, name, self._create_agnostic_function(name)) + + def _create_scalar_function(self, name: str) -> Callable: + """Create a scalar function that returns a Series of results. + + :param name: Name of the function to create + :return: Function that applies the operation to all columns + """ + + def fn(*args, **kwargs) -> pd.Series: + results = {} + for col in self.columns: + if pd.api.types.is_numeric_dtype(self[col]): + try: + results[col] = getattr(self[col], name)(*args, **kwargs) + except TypeError as exc: + # Skip columns whose dtype can't take this aggregation. + # Deliberately narrow: catching every Exception here also + # swallowed real errors (e.g. the ValueError from + # gini(negatives=...)) and returned a silently truncated + # result instead of raising. + logger.debug("skipping column %s in %s: %s", col, name, exc) + return pd.Series(results) + + return fn + + def _create_vector_function(self, name: str) -> Callable: + """Create a vector function that returns a DataFrame of results. + + :param name: Name of the function to create + :return: Function that applies the operation to all columns + """ + + def fn(*args, **kwargs) -> pd.DataFrame: + results = [] + columns = [] + for col in self.columns: + if pd.api.types.is_numeric_dtype(self[col]): + try: + result = getattr(self[col], name)(*args, **kwargs) + results.append(result) + columns.append(col) + except TypeError as exc: + # Skip columns whose dtype can't take this aggregation. + # Deliberately narrow: catching every Exception here also + # swallowed real errors (e.g. the ValueError from + # gini(negatives=...)) and returned a silently truncated + # result instead of raising. + logger.debug("skipping column %s in %s: %s", col, name, exc) + + if results: + df = pd.DataFrame(results) + df.index = columns + return df + else: + return pd.DataFrame() + + return fn + + def _create_agnostic_function(self, name: str) -> Callable: + """Create a function that can be either scalar or vector based on + input. + + :param name: Name of the function to create + :return: Function that applies the operation to all columns + """ + + def fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: + # Check if first argument is array-like + is_array = len(args) > 0 and hasattr(args[0], "__len__") + + if is_array: + # Use vector function behavior + results = [] + columns = [] + for col in self.columns: + if pd.api.types.is_numeric_dtype(self[col]): + try: + result = getattr(self[col], name)(*args, **kwargs) + results.append(result) + columns.append(col) + except TypeError as exc: + # Skip columns whose dtype can't take this aggregation. + # Deliberately narrow: catching every Exception here also + # swallowed real errors (e.g. the ValueError from + # gini(negatives=...)) and returned a silently truncated + # result instead of raising. + logger.debug("skipping column %s in %s: %s", col, name, exc) + + if results: + df = pd.DataFrame(results) + df.index = columns + return df + else: + return pd.DataFrame() + else: + # Use scalar function behavior + results = {} + for col in self.columns: + if pd.api.types.is_numeric_dtype(self[col]): + try: + results[col] = getattr(self[col], name)(*args, **kwargs) + except TypeError as exc: + # Skip columns whose dtype can't take this aggregation. + # Deliberately narrow: catching every Exception here also + # swallowed real errors (e.g. the ValueError from + # gini(negatives=...)) and returned a silently truncated + # result instead of raising. + logger.debug("skipping column %s in %s: %s", col, name, exc) + return pd.Series(results) + + return fn + + def get_args_as_micro_series(*kwarg_names: tuple) -> Callable: + """Decorator for auto-parsing column names into MicroSeries objects. + + If given, kwarg_names limits arguments checked to keyword arguments + specified. + + :param arg_names: argument names to restrict to. + :type arg_names: str + """ + + def arg_series_decorator(fn) -> Callable: + @wraps(fn) + def series_function( + self, *args, **kwargs + ) -> Union[pd.Series, pd.DataFrame]: + new_args = [] + new_kwargs = {} + if len(kwarg_names) == 0: + for value in args: + if isinstance(value, str): + if value not in self.columns: + raise Exception("Column not found") + new_args += [self[value]] + else: + new_args += [value] + for name, value in kwargs.items(): + if isinstance(value, str) and ( + len(kwarg_names) == 0 or name in kwarg_names + ): + if value not in self.columns: + raise Exception("Column not found") + new_kwargs[name] = self[value] + else: + new_kwargs[name] = value + return fn(self, *new_args, **new_kwargs) + + return series_function + + return arg_series_decorator + + def __setitem__(self, *args, **kwargs) -> None: + super().__setitem__(*args, **kwargs) + self._link_all_weights() + + def _link_weights(self, column) -> None: + # In pandas 3.0+, we can't modify column classes in-place due to CoW. + # Instead, we rely on __getitem__ to wrap columns as MicroSeries on + # access. This method is kept for backward compatibility but is now + # a no-op. + pass + + def _link_all_weights(self) -> None: + if self.weights is None: + if len(self) > 0: + self.set_weights(np.ones((len(self)))) + # In pandas 3.0+, columns are wrapped as MicroSeries on access via + # __getitem__, not stored as MicroSeries internally. + + def set_weights( + self, + weights: Union[np.ndarray, str], + preserve_old: Optional[bool] = False, + ) -> None: + """Sets the weights for the MicroDataFrame. + + If a string is received, it will be assumed to be the column name of + the weight column. + + :param weights: Array of weights. + :param preserve_old: If True, keeps the old weights as a column when + new weights are provided. + :type weights: np.array + """ + if preserve_old and self.weights_col is not None: + self["old_" + self.weights_col] = self.weights + + if isinstance(weights, str): + self.weights_col = weights + # Keep stored weights independent from edits to the source column. + self.weights = pd.Series( + np.array(self[weights], copy=True), + index=self.index, + dtype=float, + ) + self._link_all_weights() + elif weights is not None: + if len(weights) != len(self): + raise ValueError( + f"Length of weights ({len(weights)}) does not match " + f"length of DataFrame ({len(self)})." + ) + self.weights_col = None + # Align weights to self.index. Without this, weighted ops + # (self[col].multiply(self.weights) in .sum()) align on + # label, so any non-default index silently produces all-NaN + # and aggregations collapse to 0. If a Series is passed in, + # strip its index so we position-align to self.index. + if isinstance(weights, pd.Series): + weights = weights.values + with warnings.catch_warnings(): + warnings.filterwarnings("ignore", category=UserWarning) + self.weights = pd.Series( + np.asarray(weights), index=self.index, dtype=float + ) + self._link_all_weights() + + def set_weight_col(self, column: str, preserve_old: Optional[bool] = False) -> None: + """Sets the weights for the MicroDataFrame by specifying the name of + the weight column. + + .. deprecated:: 1.0.2 + Use :meth:`set_weights` with a string argument instead. + This method will be removed in a future version. + + :param column: Name of the column to use as weights. + :param preserve_old: If True, keeps the old weights as a column when + new weights are provided. + :type column: str + """ + import warnings + + warnings.warn( + "set_weight_col is deprecated and will be removed in a " + "future version. Use set_weights(column_name) instead.", + DeprecationWarning, + stacklevel=2, + ) + + if preserve_old and self.weights_col is not None: + self["old_" + self.weights_col] = self.weights + + # Delegate to set_weights: it validates length and builds an + # index-aligned float Series rather than a bare ndarray. + self.set_weights(column) + + def nullify_weights(self) -> None: + """Set all weights to 1, effectively making the DataFrame unweighted. + + This is useful for comparing weighted and unweighted statistics or when + you want to temporarily ignore weights. + """ + # Route through set_weights so self.weights stays an index-aligned + # float Series. Assigning a bare ndarray here broke every caller + # that treats it as a Series (equals(), reindex() in __getitem__). + self.set_weights(np.ones(len(self))) + + def __getitem__( + self, key: Union[str, List] + ) -> Union[MicroSeries, "MicroDataFrame"]: + # Let pandas handle the initial slicing + result = super().__getitem__(key) + + # If the result is a DataFrame, re-synchronize the weights + if isinstance(result, pd.DataFrame): + new_weights = self.weights.reindex(result.index) + return MicroDataFrame(result, weights=new_weights) + + # If the result is a Series (single column), wrap as MicroSeries + if isinstance(result, pd.Series): + return MicroSeries(result, weights=self.weights) + + # Otherwise, the result is a scalar, so just return it + return result + + def catch_series_relapse(self) -> None: + # In pandas 3.0+, we don't need to track series class changes since + # __getitem__ always wraps columns as MicroSeries on access. + pass + + def __setattr__(self, key, value) -> None: + super().__setattr__(key, value) + # No need to call catch_series_relapse in pandas 3.0+ since we wrap + # on access rather than store MicroSeries internally. + + def reset_index( + self, + level: Optional[int] = None, + drop: Optional[bool] = False, + inplace: Optional[bool] = False, + col_level: Optional[int] = 0, + col_fill: Optional[str] = "", + allow_duplicates: Optional[bool] = None, + names: Optional[List[str]] = None, + ) -> Union["MicroDataFrame", None]: + """Reset the index of the MicroDataFrame. + + This method supports all parameters of pandas DataFrame.reset_index(), + including the 'inplace' parameter. + + :param level: Only remove the given levels from the index. Removes all + levels by default. + :param drop: Do not try to insert index into dataframe columns. This + resets the index to the default integer index. + :param inplace: Modify the DataFrame in place (do not create a new + object). + :param col_level: If the columns have multiple levels, determines which + level the labels are inserted into. + :param col_fill: If the columns have multiple levels, determines how + the other levels are named. + :param allow_duplicates: Allow duplicate column labels to be created. + :param names: Using the given string, rename the DataFrame column which + contains the index data. + :return: MicroDataFrame with reset index or None if inplace=True. + """ + if inplace: + # Snapshot weight *values* positionally — the index is about + # to change and reset_index preserves row order. + weight_values = np.asarray(self.weights.values, dtype=float) + super().reset_index( + level=level, + drop=drop, + inplace=True, + col_level=col_level, + col_fill=col_fill, + allow_duplicates=allow_duplicates, + names=names, + ) + self.weights = pd.Series(weight_values, index=self.index, dtype=float) + self._link_all_weights() + return None + else: + res = super().reset_index( + level=level, + drop=drop, + inplace=False, + col_level=col_level, + col_fill=col_fill, + allow_duplicates=allow_duplicates, + names=names, + ) + out = MicroDataFrame(res, weights=self.weights.values) + # Ensure weights align to res.index (reset_index changes the + # index but preserves row order, so pass values positionally). + out.weights = pd.Series( + np.asarray(self.weights.values, dtype=float), + index=out.index, + dtype=float, + ) + return out + + def copy(self, deep: Optional[bool] = True) -> "MicroDataFrame": + res = super().copy(deep) + # super().copy() corrupts self's column types to plain Series. + # Restore them in O(N) instead of O(N²) by calling + # _link_all_weights once rather than per-column __setitem__. + self._link_all_weights() + res = MicroDataFrame(res, weights=self.weights.copy(deep)) + return res + + def drop( + self, + labels=None, + axis=0, + index=None, + columns=None, + level=None, + inplace=False, + errors="raise", + ): + """Drop specified labels from rows or columns. + + This method supports all parameters of pandas DataFrame.drop(), + including the 'inplace' parameter. + + :param labels: Index or column labels to drop. + :param axis: Whether to drop labels from the index (0 or 'index') or + columns (1 or 'columns'). + :param index: Alternative to specifying axis (labels, axis=0 is + equivalent to index=labels). + :param columns: Alternative to specifying axis (labels, axis=1 is + equivalent to columns=labels). + :param level: For MultiIndex, level from which the labels will be + removed. + :param inplace: If False, return a copy. Otherwise, do operation + inplace and return None. + :param errors: If 'ignore', suppress error and only existing labels are + dropped. + :return: MicroDataFrame or None if inplace=True. + """ + row_drop = axis in (0, "index") or index is not None + if inplace: + # Snapshot the pre-drop weights keyed by the pre-drop index so + # we can reindex to the surviving rows after the drop. + pre_drop_weights = pd.Series(self.weights.values, index=self.index.copy()) + # Perform in-place drop on the parent DataFrame + super().drop( + labels=labels, + axis=axis, + index=index, + columns=columns, + level=level, + inplace=True, + errors=errors, + ) + if row_drop: + surviving = pre_drop_weights.reindex(self.index) + self.weights = pd.Series( + surviving.values, index=self.index, dtype=float + ) + else: + self.weights = pd.Series( + pre_drop_weights.values, index=self.index, dtype=float + ) + self._link_all_weights() + return None + else: + res = super().drop( + labels=labels, + axis=axis, + index=index, + columns=columns, + level=level, + inplace=False, + errors=errors, + ) + if row_drop: + # Row drop: keep only the weights for surviving rows, + # in the order of the resulting DataFrame. + pre_drop_weights = pd.Series(self.weights.values, index=self.index) + new_weights = pre_drop_weights.reindex(res.index).values + else: + new_weights = self.weights.values + out = MicroDataFrame(res, weights=new_weights) + # Guard against the set_weights path building weights with a + # default RangeIndex, which would misalign against res.index + # and silently zero weighted aggregations. + out.weights = pd.Series(new_weights, index=out.index, dtype=float) + return out + + def merge( + self, + right, + how="inner", + on=None, + left_on=None, + right_on=None, + left_index=False, + right_index=False, + sort=False, + suffixes=("_x", "_y"), + copy=True, + indicator=False, + validate=None, + ): + """Merge DataFrame or named Series objects with a database-style join. + + This method overrides pandas DataFrame.merge() to return a + MicroDataFrame. + + :param right: Object to merge with. + :param how: Type of merge to be performed. + :param on: Column or index level names to join on. + :param left_on: Column or index level names to join on in the left + DataFrame. + :param right_on: Column or index level names to join on in the right + DataFrame. + :param left_index: Use the index from the left DataFrame as the join + key(s). + :param right_index: Use the index from the right DataFrame as the join + key(s). + :param sort: Sort the join keys lexicographically in the result + DataFrame. + :param suffixes: A length-2 sequence where each element is optionally a + string indicating the suffix to add to overlapping column names. + :param copy: If False, avoid copy if possible. + :param indicator: If True, adds a column to output DataFrame called + "_merge". + :param validate: If specified, checks if merge is of specified type. + :return: MicroDataFrame with merged data. + """ + # Attach the left weights as a temporary column so pandas' merge + # propagates them onto every surviving output row (including + # many-to-many row duplications, inner-join filtering, and + # left-with-missing NaNs). We then strip the column back off. + tmp = "__microdf_weights__" + # Avoid clobbering if this exact name is already used. + while tmp in self.columns or tmp in right.columns: + tmp += "_" + left_df = pd.DataFrame(self).copy() + left_df[tmp] = np.asarray(self.weights.values, dtype=float) + res = left_df.merge( + right, + how=how, + on=on, + left_on=left_on, + right_on=right_on, + left_index=left_index, + right_index=right_index, + sort=sort, + suffixes=suffixes, + copy=copy, + indicator=indicator, + validate=validate, + ) + # Pull out the propagated weights. Rows with no left match in a + # right/outer join get NaN weight — fill with 0 so they don't + # poison later aggregations (a user who needs a different + # convention can override afterwards). + merged_weights = res[tmp].fillna(0).to_numpy(dtype=float) + res = res.drop(columns=[tmp]) + out = MicroDataFrame(res, weights=merged_weights) + # Ensure the weights Series aligns with res.index regardless of + # the default-RangeIndex behavior of set_weights. + out.weights = pd.Series(merged_weights, index=out.index, dtype=float) + return out + + def __getattr__(self, name): + """Allow accessing columns as attributes (e.g., df.column_name). + + This enables more intuitive column access while preserving MicroSeries + functionality when accessing columns. + + :param name: Attribute name to access + :return: MicroSeries if the attribute is a column, otherwise delegates + to parent + """ + if name in self.columns: + return self[name] + return super().__getattr__(name) + + def equals(self, other: "MicroDataFrame") -> bool: + equal_values = super().equals(other) + equal_weights = self.weights.equals(other.weights) + return equal_values and equal_weights + + @get_args_as_micro_series() + def groupby(self, by: Union[str, List], *args, **kwargs) -> "MicroDataFrameGroupBy": + """Returns a GroupBy object with MicroSeriesGroupBy objects for each + column. + + :param by: column to group by + :type by: Union[str, List] + + return: DataFrameGroupBy object with columns using weights + rtype: DataFrameGroupBy + """ + # Build the groupby on a *copy* that carries a ``__tmp_weights`` + # column. We used to set this column on ``self`` directly, which + # permanently leaked the weight column onto the caller's + # DataFrame — any later ``df.sum()`` or ``list(df.columns)`` + # would then include it. + staged = pd.DataFrame(self).copy() + staged["__tmp_weights"] = np.asarray(self.weights.values, dtype=float) + gb = staged.groupby(by, *args, **kwargs) + weights = copy.deepcopy(gb["__tmp_weights"]) + for col in staged.columns: # df.groupby(...)[col]s use weights + res = gb[col] + res.__class__ = MicroSeriesGroupBy + res._init() + res.weights = weights + setattr(gb, col, res) + gb.__class__ = MicroDataFrameGroupBy + gb._init(by) + return gb + + @get_args_as_micro_series() + def poverty_rate(self, income: str, threshold: str) -> float: + """Calculate poverty rate, i.e., the population share with income below + their poverty threshold. + + :param income: Column indicating income. + :type income: str + :param threshold: Column indicating threshold. + :type threshold: str + :return: Poverty rate between zero and one. + :rtype: float + """ + pov = income < threshold + return pov.sum() / pov.count() + + @get_args_as_micro_series() + def deep_poverty_rate(self, income: str, threshold: str) -> float: + """Calculate deep poverty rate, i.e., the population share with income + below half their poverty threshold. + + :param income: Column indicating income. + :type income: str + :param threshold: Column indicating threshold. + :type threshold: str + :return: Deep poverty rate between zero and one. + :rtype: float + """ + pov = income < (threshold / 2) + return pov.sum() / pov.count() + + @get_args_as_micro_series() + def poverty_gap(self, income: str, threshold: str) -> float: + """Calculate poverty gap, i.e., the total gap between income and + poverty thresholds for all people in poverty. + + :param income: Column indicating income. + :type income: str + :param threshold: Column indicating threshold. + :type threshold: str + :return: Poverty gap. + :rtype: float + """ + gaps = (threshold - income)[threshold > income] + return gaps.sum() + + @get_args_as_micro_series() + def deep_poverty_gap(self, income: str, threshold: str) -> float: + """Calculate deep poverty gap, i.e., the total gap between income and + half of poverty thresholds for all people in deep poverty. + + :param income: Column indicating income. + :type income: str + :param threshold: Column indicating threshold. + :type threshold: str + :return: Deep poverty gap. + :rtype: float + """ + deep_threshold = threshold / 2 + gaps = (deep_threshold - income)[deep_threshold > income] + return gaps.sum() + + @get_args_as_micro_series() + def squared_poverty_gap(self, income: str, threshold: str) -> float: + """Calculate squared poverty gap, i.e., the total squared gap between + income and poverty thresholds for all people in poverty. Also known as + the poverty severity index. + + :param income: Column indicating income. + :type income: str + :param threshold: Column indicating threshold. + :type threshold: str + :return: Squared poverty gap. + :rtype: float + """ + gaps = (threshold - income)[threshold > income] + squared_gaps = gaps**2 + return squared_gaps.sum() + + @get_args_as_micro_series() + def poverty_count( + self, + income: Union[MicroSeries, str], + threshold: Union[MicroSeries, str], + ) -> int: + """Calculates the number of entities with income below a poverty + threshold. + + :param income: income array or column name + :type income: Union[MicroSeries, str] + + :param threshold: threshold array or column name + :type threshold: Union[MicroSeries, str] + + return: number of entities in poverty + rtype: int + """ + in_poverty = income < threshold + return in_poverty.sum() + + def astype( + self, + dtype, + copy: Optional[bool] = True, + errors: Optional[str] = "raise", + ) -> "MicroDataFrame": + """Convert MicroDataFrame to specified data type while preserving + weights. + + :param dtype: Data type to convert to. Can be numpy dtype, Python type, + or dict. + :param copy: Whether to make a copy of the data (default True). + :param errors: How to handle conversion errors (default "raise"). + :return: New MicroDataFrame with converted data types and preserved + weights. + """ + converted_df = super().astype(dtype, copy=copy, errors=errors) + return MicroDataFrame( + converted_df, weights=self.weights.copy() if copy else self.weights + ) + + def __repr__(self) -> str: + df = pd.DataFrame(self) + df["weight"] = self.weights + return df[[df.columns[-1]] + list(df.columns[:-1])].__repr__() + + +class MicroDataFrameGroupBy(pd.core.groupby.generic.DataFrameGroupBy): + def _init(self, by: Union[str, List]): + self._by = by + self.columns = list(self.obj.columns) + if isinstance(by, list): + for column in by: + self.columns.remove(column) + elif isinstance(by, str): + self.columns.remove(by) + self.columns.remove("__tmp_weights") + # Filter to only numeric columns + self.numeric_columns = [ + col for col in self.columns if pd.api.types.is_numeric_dtype(self.obj[col]) + ] + # Store reference to weights groupby for column selection + self._weights_groupby = copy.deepcopy(super().__getitem__("__tmp_weights")) + for fn_name in MicroSeries.SCALAR_FUNCTIONS: + + def get_fn(name): + def fn(*args, **kwargs): + results = {} + for col in self.numeric_columns: + try: + results[col] = getattr(getattr(self, col), name)( + *args, **kwargs + ) + except TypeError as exc: + # Skip columns whose dtype can't take this aggregation. + # Deliberately narrow: catching every Exception here also + # swallowed real errors (e.g. the ValueError from + # gini(negatives=...)) and returned a silently truncated + # result instead of raising. + logger.debug("skipping column %s in %s: %s", col, name, exc) + # Return plain DataFrame - aggregated results don't have + # per-row weights (weights were already applied) + return pd.DataFrame(results) if results else pd.DataFrame() + + return fn + + setattr(self, fn_name, get_fn(fn_name)) + for fn_name in MicroSeries.VECTOR_FUNCTIONS: + + def get_fn(name) -> Callable: + def fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: + results = {} + for col in self.numeric_columns: + try: + results[col] = getattr(getattr(self, col), name)( + *args, **kwargs + ) + except TypeError as exc: + # Skip columns whose dtype can't take this aggregation. + # Deliberately narrow: catching every Exception here also + # swallowed real errors (e.g. the ValueError from + # gini(negatives=...)) and returned a silently truncated + # result instead of raising. + logger.debug("skipping column %s in %s: %s", col, name, exc) + # Return plain DataFrame - aggregated results don't have + # per-row weights (weights were already applied) + return pd.DataFrame(results) if results else pd.DataFrame() + + return fn + + setattr(self, fn_name, get_fn(fn_name)) + + def __getitem__( + self, key: Union[str, List] + ) -> Union["MicroSeriesGroupBy", "MicroDataFrameGroupBy"]: + """Select columns from the groupby object while preserving weights. + + This ensures that operations like groupby(col)["y"].sum() or + groupby(col)[["y"]].sum() use weighted aggregation. + + :param key: Column name or list of column names + :return: MicroSeriesGroupBy for single column, MicroDataFrameGroupBy + for multiple columns + """ + if isinstance(key, str): + # Single column - return MicroSeriesGroupBy + result = super().__getitem__(key) + result.__class__ = MicroSeriesGroupBy + result._init() + result.weights = self._weights_groupby + return result + else: + # Multiple columns - return a new MicroDataFrameGroupBy + # with only the selected columns + result = super().__getitem__(key) + result.__class__ = MicroDataFrameGroupBy + # Re-initialize with the subset of columns + result._by = self._by + result.columns = list(key) if hasattr(key, "__iter__") else [key] + result.numeric_columns = [ + col + for col in result.columns + if pd.api.types.is_numeric_dtype(result.obj[col]) + ] + result._weights_groupby = self._weights_groupby + # Set up the column attributes as MicroSeriesGroupBy + for col in result.columns: + col_gb = super().__getitem__(col) + col_gb.__class__ = MicroSeriesGroupBy + col_gb._init() + col_gb.weights = self._weights_groupby + setattr(result, col, col_gb) + # Set up the scalar and vector functions + for fn_name in MicroSeries.SCALAR_FUNCTIONS: + + def get_scalar_fn(name, res): + def fn(*args, **kwargs): + results = {} + for col in res.numeric_columns: + try: + results[col] = getattr(getattr(res, col), name)( + *args, **kwargs + ) + except TypeError as exc: + # Skip columns whose dtype can't take this aggregation. + # Deliberately narrow: catching every Exception here also + # swallowed real errors (e.g. the ValueError from + # gini(negatives=...)) and returned a silently truncated + # result instead of raising. + logger.debug( + "skipping column %s in %s: %s", col, name, exc + ) + # Return plain DataFrame - aggregated results don't + # have per-row weights (weights were already applied) + return pd.DataFrame(results) if results else pd.DataFrame() + + return fn + + setattr(result, fn_name, get_scalar_fn(fn_name, result)) + for fn_name in MicroSeries.VECTOR_FUNCTIONS: + + def get_vector_fn(name, res): + def fn(*args, **kwargs): + results = {} + for col in res.numeric_columns: + try: + results[col] = getattr(getattr(res, col), name)( + *args, **kwargs + ) + except TypeError as exc: + # Skip columns whose dtype can't take this aggregation. + # Deliberately narrow: catching every Exception here also + # swallowed real errors (e.g. the ValueError from + # gini(negatives=...)) and returned a silently truncated + # result instead of raising. + logger.debug( + "skipping column %s in %s: %s", col, name, exc + ) + # Return plain DataFrame - aggregated results don't + # have per-row weights (weights were already applied) + return pd.DataFrame(results) if results else pd.DataFrame() + + return fn + + setattr(result, fn_name, get_vector_fn(fn_name, result)) + return result diff --git a/build/lib/microdf/microseries.py b/build/lib/microdf/microseries.py new file mode 100644 index 00000000..688f8e77 --- /dev/null +++ b/build/lib/microdf/microseries.py @@ -0,0 +1,983 @@ +import logging +import warnings +from functools import wraps +from typing import Callable, List, Optional, Union + +import numpy as np +import pandas as pd + +logger = logging.getLogger(__name__) + + +def _weighted_top_share( + values: np.ndarray, weights: np.ndarray, top_x_pct: float +) -> float: + """Share of the sum held by the top ``top_x_pct`` of weight. + + Sort by value ascending, cumulate the weight, pick the slice from the top + that covers exactly ``top_x_pct`` of total weight, and distribute the tied- + at-cutoff row proportionally so constant values return exactly + ``top_x_pct`` rather than 1.0. + """ + if top_x_pct <= 0: + return 0.0 + if top_x_pct >= 1: + return 1.0 + total_weight = weights.sum() + total_sum = float((values * weights).sum()) + if total_weight == 0 or total_sum == 0: + return np.nan + # Ascending sort; the "top" cutoff is the final ``top_x_pct`` of + # cumulative weight. + order = np.argsort(values, kind="mergesort") + v = values[order] + w = weights[order] + # Cumulative weight from the bottom up. + cum_w = np.cumsum(w) + target_bottom_weight = total_weight * (1.0 - top_x_pct) + # searchsorted(cum_w, target, side="right") gives the first index + # whose cumulative weight exceeds the bottom cutoff. + k = int(np.searchsorted(cum_w, target_bottom_weight, side="right")) + # Rows strictly above the cutoff contribute all of their weight. + if k >= len(v): + return 0.0 + top_sum = float((v[k + 1 :] * w[k + 1 :]).sum()) + # Row k straddles the cutoff; include the fraction of its weight + # that lies above the cutoff so ties don't double-count. + partial_weight = cum_w[k] - target_bottom_weight + top_sum += float(v[k] * partial_weight) + return top_sum / total_sum + + +class MicroSeries(pd.Series): + # Declare ``weights`` as pandas metadata. pandas includes + # _metadata attributes in the pickle state, so weights now survive + # pickling, to_pickle/read_pickle and copy.deepcopy instead of + # vanishing and leaving an AttributeError on the next aggregation. + # Keep pandas' own metadata, including the Series name. + _metadata = pd.Series._metadata + ["weights"] + + def __init__(self, *args, weights: np.array = None, **kwargs): + """A Series-inheriting class for weighted microdata. + + Weights can be provided at initialisation, or using set_weights. + + :param weights: Array of weights. + :type weights: np.array + """ + super().__init__(*args, **kwargs) + self.set_weights(weights) + + def __finalize__(self, other, method=None, **kwargs) -> "MicroSeries": + """Retain copied weights when pandas finalizes a renamed result.""" + copied_weights = getattr(self, "weights", None) if method == "rename" else None + super().__finalize__(other, method=method, **kwargs) + if copied_weights is not None: + # rename already called copy(); metadata propagation must not + # replace those weights with the source's mutable Series. + self.weights = copied_weights + return self + + @property + def _values(self): + """Internal access to underlying numpy array without warning.""" + return super().values + + @property + def values(self): + """Access underlying numpy array. + + .. warning:: + Returns a plain numpy array without weights. Operations + like ``.mean()`` on the result will be unweighted. Use + MicroSeries methods directly for weighted calculations + (e.g., ``ms.mean()`` instead of ``ms.values.mean()``). + """ + warnings.warn( + "Accessing .values on a MicroSeries returns a plain numpy " + "array without weights. Operations like .mean() on the " + "result will be unweighted. Use MicroSeries methods " + "directly for weighted calculations (e.g., ms.mean() " + "instead of ms.values.mean()).", + UserWarning, + stacklevel=2, + ) + return super().values + + def to_numpy(self, *args, **kwargs): + """Convert to numpy array. + + .. warning:: + Returns a plain numpy array without weights. Operations + like ``.mean()`` on the result will be unweighted. Use + MicroSeries methods directly for weighted calculations. + """ + warnings.warn( + "Calling .to_numpy() on a MicroSeries returns a plain " + "numpy array without weights. Operations like .mean() on " + "the result will be unweighted. Use MicroSeries methods " + "directly for weighted calculations.", + UserWarning, + stacklevel=2, + ) + return super().to_numpy(*args, **kwargs) + + def scalar_function(fn: Callable) -> Callable: + """Decorator marking ``fn`` as returning a scalar (float).""" + fn._rtype = float + return fn + + def vector_function(fn: Callable) -> Callable: + """Decorator marking ``fn`` as returning a pandas Series.""" + fn._rtype = pd.Series + return fn + + def set_weights( + self, weights: np.array, preserve_old: Optional[bool] = False + ) -> None: + """Sets the weight values. + + :param weights: Array of weights. + :param preserve_old: If True, keeps the old weights as a column when + new weights are provided. + :type weights: np.array. + """ + if weights is None: + if len(self) > 0: + self.weights = pd.Series( + np.ones_like(self._values), + index=self.index, + dtype=float, + ) + else: + if len(weights) != len(self): + raise ValueError( + f"Length of weights ({len(weights)}) does not match " + f"length of DataFrame ({len(self)})." + ) + + if preserve_old and self.weights is not None: + self["old_weights"] = self.weights + + # Align weights to self.index so element-wise operations such + # as self.multiply(self.weights) (used by .sum(), .weight()) + # don't silently produce all-NaN when the caller uses a + # non-default index. If a pandas Series is passed in, strip + # its index first so we position-align rather than label-align. + if isinstance(weights, pd.Series): + weights = weights.values + self.weights = pd.Series(np.asarray(weights), index=self.index, dtype=float) + + def nullify_weights(self) -> None: + """Set all weights to 1, effectively making the Series unweighted. + + This is useful for comparing weighted and unweighted statistics or when + you want to temporarily ignore weights. + """ + # Index the ones against self.index: weighted ops are label-aligned + # (self.multiply(self.weights) in .sum()/.weight()), so a default + # RangeIndex here silently produces all-NaN and collapses every + # aggregation to 0 whenever the caller uses a non-default index. + self.weights = pd.Series(np.ones(len(self)), index=self.index, dtype=float) + + @vector_function + def weight(self) -> pd.Series: + """Calculates the weighted value of the MicroSeries. + + :returns: A Series multiplying the MicroSeries by its weight. + :rtype: pd.Series + """ + return self.multiply(self.weights) + + @scalar_function + def sum(self) -> float: + """Calculates the weighted sum of the MicroSeries. + + :returns: The weighted sum. + :rtype: float + """ + return self.multiply(self.weights).sum() + + @scalar_function + def count(self, skipna: bool = True) -> float: + """Calculates the weighted count of the MicroSeries. + + By default skips NaN values (matching pandas ``Series.count``). + + :param skipna: Exclude NaN values (default True). If False, the + weighted count of every row is returned. + :type skipna: bool + :returns: The weighted count. + :rtype: float + """ + weights = np.asarray(self.weights.values, dtype=float) + if not skipna: + return float(weights.sum()) + mask = ~pd.isna(self._values) + return float(weights[mask].sum()) + + @scalar_function + def mean(self, skipna: bool = True) -> float: + """Calculates the weighted mean of the MicroSeries. + + :param skipna: Exclude NA/null values. If True (default), NaN values + are excluded. If False, returns NaN if any value is NaN. + :type skipna: bool + :returns: The weighted mean. + :rtype: float + """ + values = self._values + weights = self.weights + + if skipna: + # Create mask for non-NaN values + mask = ~pd.isna(values) + if not mask.any(): + # All values are NaN + return np.nan + values = values[mask] + weights = weights[mask] + + # If skipna=False and there are any NaN values, return NaN + if not skipna and pd.isna(values).any(): + return np.nan + + return np.average(values, weights=weights) + + def _weighted_variance(self, ddof: int = 1, skipna: bool = True) -> float: + """Frequency-weighted variance. + + Uses ``sum(w * (x - wmean)**2) / (sum(w) - ddof)``. With + ``ddof=0`` this is the population variance; with ``ddof=1`` it + is Bessel-corrected assuming the weights are frequency counts — + matching ``np.var(..., ddof=ddof)`` on a replicated sample. + """ + values = np.asarray(self._values, dtype=float) + weights = np.asarray(self.weights.values, dtype=float) + if skipna: + mask = ~np.isnan(values) + values = values[mask] + weights = weights[mask] + elif np.isnan(values).any(): + return np.nan + total_w = weights.sum() + if total_w == 0 or total_w - ddof <= 0: + return np.nan + mean = np.average(values, weights=weights) + return float((weights * (values - mean) ** 2).sum() / (total_w - ddof)) + + @scalar_function + def var(self, ddof: int = 1, skipna: bool = True) -> float: + """Calculates the weighted variance of the MicroSeries. + + Treats weights as frequency counts (``sum(w) - ddof`` in the + denominator) so that with integer weights the result matches + ``np.var`` on the replicated sample. + + :param ddof: Delta degrees of freedom (default 1). + :param skipna: Exclude NaN values (default True). + :returns: The weighted variance. + :rtype: float + """ + return self._weighted_variance(ddof=ddof, skipna=skipna) + + @scalar_function + def std(self, ddof: int = 1, skipna: bool = True) -> float: + """Calculates the weighted standard deviation of the MicroSeries. + + :param ddof: Delta degrees of freedom (default 1). + :param skipna: Exclude NaN values (default True). + :returns: The weighted standard deviation. + :rtype: float + """ + v = self._weighted_variance(ddof=ddof, skipna=skipna) + return float(np.sqrt(v)) if np.isfinite(v) else v + + def cov(self, other, *args, **kwargs): + """Pandas ``cov`` — **unweighted**. + + MicroSeries does not yet compute weighted covariance. Emits a + ``UserWarning`` so callers aren't silently given an unweighted number + after ``.sum()`` and ``.mean()`` worked as expected. See issue tracker + for a weighted implementation. + """ + warnings.warn( + "MicroSeries.cov() falls through to pandas and is " + "unweighted. Use MicroSeries.var()/std() for weighted " + "second moments, or compute covariance manually with the " + "weights.", + UserWarning, + stacklevel=2, + ) + return super().cov(other, *args, **kwargs) + + def corr(self, other, *args, **kwargs): + """Pandas ``corr`` — **unweighted**. + + MicroSeries does not yet compute weighted correlation. Emits a + ``UserWarning`` so callers aren't silently given an unweighted number. + See issue tracker for a weighted implementation. + """ + warnings.warn( + "MicroSeries.corr() falls through to pandas and is " + "unweighted. Compute correlation manually with the weights " + "if you need the survey-weighted value.", + UserWarning, + stacklevel=2, + ) + return super().corr(other, *args, **kwargs) + + def quantile(self, q: np.array, skipna: bool = True) -> pd.Series: + """Calculates weighted quantiles of the MicroSeries. + + Uses the inverse CDF method: the q-th quantile is the smallest + value where the cumulative weight proportion >= q. This matches + the default behavior of R's survey::svyquantile. + + :param q: Quantile(s) to calculate, must be in [0, 1]. + :type q: float or np.array + :param skipna: Exclude NaN values (default True). NaN sorts to the + end of the array, so leaving NaN rows in would let their weight + inflate the cumulative distribution and push the cutoff upward. + If False, NaN is returned whenever any value is NaN. + :type skipna: bool + + :return: Weighted quantile value(s). + :rtype: float or pd.Series + """ + values = np.array(self._values) + quantiles = np.atleast_1d(q) + sample_weight = np.array(self.weights) + assert np.all(quantiles >= 0) and np.all(quantiles <= 1), ( + "quantiles should be in [0, 1]" + ) + na_mask = pd.isna(values) + if not skipna and na_mask.any(): + return ( + np.nan + if np.array(q).shape == () + else pd.Series(np.full(len(quantiles), np.nan), index=quantiles) + ) + # Drop zero-weight rows before sorting. Without this, q=0 (and + # internal plateaus of zero weight) picked a value with 0 weight + # that should have been skipped by the inverse CDF. E.g. + # MicroSeries([10, 20, 30], weights=[0, 1, 1]).quantile(0) + # returned 10 instead of 20. + # Drop NaN rows for the same reason: NaN sorts last, so its weight + # would inflate the cumulative distribution and push the cutoff up + # (median of [1, nan, 3] returned 3.0 instead of 1.0). + nonzero = (sample_weight > 0) & ~na_mask + if not nonzero.any(): + return ( + np.nan + if np.array(q).shape == () + else pd.Series(np.full(len(quantiles), np.nan), index=quantiles) + ) + values = values[nonzero] + sample_weight = sample_weight[nonzero] + sorter = np.argsort(values) + values = values[sorter] + sample_weight = sample_weight[sorter] + cumsum = np.cumsum(sample_weight) + cumsum_normalized = cumsum / cumsum[-1] + result = np.array( + [ + values[min(np.searchsorted(cumsum_normalized, qi), len(values) - 1)] + for qi in quantiles + ] + ) + if np.array(q).shape == (): + return result[0] + return pd.Series(result, index=quantiles) + + @scalar_function + def median(self, skipna: bool = True) -> float: + """Calculates the weighted median of the MicroSeries. + + :param skipna: Exclude NaN values (default True). + :type skipna: bool + :returns: The weighted median of a DataFrame's column. + :rtype: float + """ + return self.quantile(0.5, skipna=skipna) + + def replicate_standard_error( + self, + statistic: Callable, + replicate_weights, + method: str = "jackknife", + fay_k: Optional[float] = None, + ) -> float: + """Standard error of ``statistic`` from a set of replicate weights. + + Recomputes the statistic once per replicate and scales the spread by + the factor appropriate to how the replicates were built, so it works + for any statistic including the Gini coefficient and quantiles. + + Only valid for replicate weights as published with a survey. Weights + calibrated to external targets no longer correspond to the original + replication scheme. + + :param statistic: Callable taking a MicroSeries, e.g. + ``lambda s: s.median()``. + :param replicate_weights: Array or frame of shape ``(len(self), R)``. + :param method: ``jackknife``, ``brr``, ``bootstrap``, + ``successive-difference`` or ``fay``. + :param fay_k: Fay's perturbation constant, for ``method="fay"``. + :returns: The estimated standard error. + """ + from microdf.replication import replicate_standard_error + + return replicate_standard_error( + self, statistic, replicate_weights, method, fay_k + ) + + @scalar_function + def gini(self, negatives: Optional[str] = None) -> float: + """Calculates Gini index. + + :param negatives: An optional string indicating how to treat + negative values of x: + 'zero' replaces negative values with zeroes. + 'shift' subtracts the minimum value from all values of x, + when this minimum is negative. That is, it adds the absolute + minimum value. + Defaults to None, which leaves negative values as they are. + :type negatives: str + :returns: Gini index. + :rtype: float + """ + x = np.array(self).astype("float") + w = np.asarray(self.weights.values, dtype=float) + if negatives == "zero": + x = np.where(x < 0, 0.0, x) + elif negatives == "shift": + if len(x) > 0 and np.amin(x) < 0: + x = x - np.amin(x) + elif negatives is not None: + raise ValueError( + f"Unknown negatives option {negatives!r}; expected " + "'zero', 'shift', or None." + ) + + if len(x) == 0: + return np.nan + if np.any(x < 0): + # The Lorenz-based formula assumes non-negative values; with + # negatives it can return values outside [0, 1]. + warnings.warn( + "gini() called on data containing negative values; the " + "result is not guaranteed to lie in [0, 1]. Pass " + "negatives='zero' or negatives='shift' to handle them.", + UserWarning, + stacklevel=2, + ) + + # Short-circuit degenerate cases so we don't divide by zero. + total = float((x * w).sum()) + if total == 0: + return 0.0 + + sorter = np.argsort(x, kind="mergesort") + sorted_x = x[sorter] + sorted_w = w[sorter] + cumw = np.cumsum(sorted_w) + cumxw = np.cumsum(sorted_x * sorted_w) + # Trapezoidal approximation of the area under the Lorenz curve. + return float( + np.sum(cumxw[1:] * cumw[:-1] - cumxw[:-1] * cumw[1:]) + / (cumxw[-1] * cumw[-1]) + ) + + @scalar_function + def top_x_pct_share(self, top_x_pct: float) -> float: + """Calculates top x% share. + + Uses a cumulative-weight sort so that rows tied at the cutoff + contribute proportionally rather than all-or-nothing. With + constant values this correctly returns ``top_x_pct`` itself. + + :param top_x_pct: Decimal between 0 and 1 of the top %, e.g. 0.1, + 0.001. + :type top_x_pct: float + :returns: The weighted share held by the top x%. + :rtype: float + """ + return _weighted_top_share( + np.asarray(self._values, dtype=float), + np.asarray(self.weights.values, dtype=float), + float(top_x_pct), + ) + + @scalar_function + def bottom_x_pct_share(self, bottom_x_pct: float) -> float: + """Calculates bottom x% share. + + :param bottom_x_pct: Decimal between 0 and 1 of the bottom %, e.g. 0.1, + 0.001. + :type bottom_x_pct: float + :returns: The weighted share held by the bottom x%. + :rtype: float + """ + return 1 - self.top_x_pct_share(1 - bottom_x_pct) + + @scalar_function + def bottom_50_pct_share(self) -> float: + """Calculates bottom 50% share. + + :returns: The weighted share held by the bottom 50%. + :rtype: float + """ + return self.bottom_x_pct_share(0.5) + + @scalar_function + def top_50_pct_share(self) -> float: + """Calculates top 50% share. + + :returns: The weighted share held by the top 50%. + :rtype: float + """ + return self.top_x_pct_share(0.5) + + @scalar_function + def top_10_pct_share(self) -> float: + """Calculates top 10% share. + + :returns: The weighted share held by the top 10%. + :rtype: float + """ + return self.top_x_pct_share(0.1) + + @scalar_function + def top_1_pct_share(self) -> float: + """Calculates top 1% share. + + :returns: The weighted share held by the top 50%. + :rtype: float + """ + return self.top_x_pct_share(0.01) + + @scalar_function + def top_0_1_pct_share(self) -> float: + """Calculates top 0.1% share. + + :returns: The weighted share held by the top 0.1%. + :rtype: float + """ + return self.top_x_pct_share(0.001) + + @scalar_function + def t10_b50(self) -> float: + """Calculates ratio between the top 10% and bottom 50% shares. + + :returns: The weighted share held by the top 10% divided by the + weighted share held by the bottom 50%. + """ + t10 = self.top_10_pct_share() + b50 = self.bottom_50_pct_share() + return t10 / b50 + + @vector_function + def cumsum(self) -> pd.Series: + logger.warning( + "cumsum() returns cumulative sums of weighted values as a regular " + "pandas Series. The original weights have already been applied " + "and cannot be reused with the cumulative results." + ) + return pd.Series(self * self.weights).cumsum() + + @vector_function + def rank(self, pct: Optional[bool] = False) -> pd.Series: + """Weighted rank of each element. + + Each element's rank is the cumulative weight of all values that are + less than or equal to it. Tied values therefore share the same rank, so + downstream bucketing (``decile_rank``, ``quintile_rank``, etc.) lands + tied rows in the same bucket. + + :param pct: If True, divide ranks by the total weight so they lie in + ``(0, 1]``. + :type pct: bool + :returns: MicroSeries of ranks aligned to ``self``. + :rtype: MicroSeries + """ + weights_sum = np.asarray(self.weights.values, dtype=float).sum() + if weights_sum == 0: + raise ZeroDivisionError( + "Cannot calculate rank with zero total weight. " + "All weights in the MicroSeries are zero, which would " + "result in division by zero." + ) + + values = np.asarray(self._values) + weights = np.asarray(self.weights.values, dtype=float) + order = np.argsort(values, kind="mergesort") + sorted_values = values[order] + sorted_weights = weights[order] + cum_w = np.cumsum(sorted_weights) + # Max rank semantics: every tied group gets the cumulative + # weight at the *end* of the group, so ties share one rank. + # searchsorted(side='right') on the sorted values finds the + # index just past each tied block in sort order. + group_end = np.searchsorted(sorted_values, sorted_values, side="right") - 1 + sorted_ranks = cum_w[group_end] + # Invert the sort to put ranks back into the caller's order. + inverse_order = np.argsort(order, kind="mergesort") + ranks = sorted_ranks[inverse_order] + if pct: + ranks = ranks / weights_sum + ranks = np.where(ranks > 1.0, 1.0, ranks) + return MicroSeries(ranks, index=self.index, weights=self.weights) + + @vector_function + def decile_rank(self, negatives_in_zero: Optional[bool] = False): + """Calculate decile ranks (1-10) with optional zero decile for + negatives. + + :param negatives_in_zero: If True, negative values are assigned to + decile 0. If False (default), all values are ranked 1-10. + :type negatives_in_zero: bool + :returns: MicroSeries with decile ranks + :rtype: MicroSeries + """ + if negatives_in_zero: + negative_mask = self < 0 + if negative_mask.any(): + non_negative_values = self[~negative_mask] + if len(non_negative_values) > 0: + non_neg_ranks = non_negative_values.rank(pct=True) + deciles = np.minimum(np.ceil(non_neg_ranks * 10), 10) + else: + deciles = np.array([]) + + result = np.zeros(len(self)) + result[negative_mask] = 0 + if len(deciles) > 0: + result[~negative_mask] = deciles + + return MicroSeries(result, weights=self.weights) + + # Default behavior: rank all values 1-10 + return MicroSeries( + np.minimum(np.ceil(self.rank(pct=True) * 10), 10), + weights=self.weights, + ) + + @vector_function + def quintile_rank(self) -> "MicroSeries": + return MicroSeries( + np.minimum(np.ceil(self.rank(pct=True) * 5), 5), + weights=self.weights, + ) + + @vector_function + def quartile_rank(self) -> "MicroSeries": + return MicroSeries( + np.minimum(np.ceil(self.rank(pct=True) * 4), 4), + weights=self.weights, + ) + + @vector_function + def percentile_rank(self) -> "MicroSeries": + return MicroSeries( + np.minimum(np.ceil(self.rank(pct=True) * 100), 100), + weights=self.weights, + ) + + def groupby(self, *args, **kwargs) -> "MicroSeriesGroupBy": + gb = super().groupby(*args, **kwargs) + gb.__class__ = MicroSeriesGroupBy + gb._init() + gb.weights = pd.Series(self.weights).groupby(*args, **kwargs) + return gb + + def copy(self, deep: Optional[bool] = True): + res = super().copy(deep) + res = MicroSeries(res, weights=self.weights.copy(deep)) + return res + + def clip( + self, + lower: Optional[float] = None, + upper: Optional[float] = None, + axis: Optional[int] = None, + inplace: Optional[bool] = False, + *args, + **kwargs, + ) -> "MicroSeries": + res = super().clip( + lower=lower, + upper=upper, + axis=axis, + inplace=inplace, + *args, + **kwargs, + ) + if not inplace: + return MicroSeries(res, weights=self.weights) + return self + + def round(self, decimals: Optional[int] = 0, *args, **kwargs) -> "MicroSeries": + res = super().round(decimals=decimals, *args, **kwargs) + return MicroSeries(res, weights=self.weights) + + def equals(self, other: "MicroSeries") -> bool: + equal_values = super().equals(other) + equal_weights = self.weights.equals(other.weights) + return equal_values and equal_weights + + def __getitem__( + self, key: Union[str, int, slice, List, np.ndarray] + ) -> Union["MicroSeries", pd.Series]: + result = super().__getitem__(key) + if isinstance(result, pd.Series): + weights = self.weights.__getitem__(key) + return MicroSeries(result, weights=weights) + return result + + def __getattr__(self, name: str) -> "MicroSeries": + return MicroSeries(super().__getattr__(name), weights=self.weights) + + # operators + + def __add__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__add__(other), weights=self.weights) + + def __sub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__sub__(other), weights=self.weights) + + def __mul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__mul__(other), weights=self.weights) + + def __floordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__floordiv__(other), weights=self.weights) + + def __truediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__truediv__(other), weights=self.weights) + + def __mod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__mod__(other), weights=self.weights) + + def __pow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__pow__(other), weights=self.weights) + + def __xor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__xor__(other), weights=self.weights) + + def __and__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__and__(other), weights=self.weights) + + def __or__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__or__(other), weights=self.weights) + + def __invert__(self) -> "MicroSeries": + return MicroSeries(super().__invert__(), weights=self.weights) + + def __radd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__radd__(other), weights=self.weights) + + def __rsub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__rsub__(other), weights=self.weights) + + def __rmul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__rmul__(other), weights=self.weights) + + def __rfloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__rfloordiv__(other), weights=self.weights) + + def __rtruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__rtruediv__(other), weights=self.weights) + + def __rmod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__rmod__(other), weights=self.weights) + + def __rpow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__rpow__(other), weights=self.weights) + + def __rand__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__rand__(other), weights=self.weights) + + def __ror__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__ror__(other), weights=self.weights) + + def __rxor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__rxor__(other), weights=self.weights) + + def sqrt(self) -> "MicroSeries": + sqrt_values = np.sqrt(self._values) + return MicroSeries(sqrt_values, index=self.index, weights=self.weights) + + # comparators + + def __lt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__lt__(other), weights=self.weights) + + def __le__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__le__(other), weights=self.weights) + + def __eq__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__eq__(other), weights=self.weights) + + def __ne__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__ne__(other), weights=self.weights) + + def __ge__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__ge__(other), weights=self.weights) + + def __gt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__gt__(other), weights=self.weights) + + # assignment operators + + def __iadd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__iadd__(other), weights=self.weights) + + def __isub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__isub__(other), weights=self.weights) + + def __imul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__imul__(other), weights=self.weights) + + def __ifloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__ifloordiv__(other), weights=self.weights) + + def __idiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__idiv__(other), weights=self.weights) + + def __itruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__itruediv__(other), weights=self.weights) + + def __imod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__imod__(other), weights=self.weights) + + def __ipow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": + return MicroSeries(super().__ipow__(other), weights=self.weights) + + # other + + def __neg__(self) -> "MicroSeries": + return MicroSeries(super().__neg__(), weights=self.weights) + + def __pos__(self) -> "MicroSeries": + return MicroSeries(super().__pos__(), weights=self.weights) + + def astype( + self, + dtype, + copy: Optional[bool] = True, + errors: Optional[str] = "raise", + ) -> "MicroSeries": + """Convert MicroSeries to specified data type while preserving weights. + + :param dtype: Data type to convert to. Can be numpy dtype or Python + type. + :param copy: Whether to make a copy of the data (default True). + :param errors: How to handle conversion errors (default "raise"). + :return: New MicroSeries with converted data type and preserved + weights. + """ + converted_series = super().astype(dtype, copy=copy, errors=errors) + return MicroSeries( + converted_series, + weights=self.weights.copy() if copy else self.weights, + ) + + def __repr__(self) -> str: + return pd.DataFrame( + dict(value=self._values, weight=self.weights.values) + ).__repr__() + + +MicroSeries.SCALAR_FUNCTIONS = [ + fn + for fn in dir(MicroSeries) + if "_rtype" in dir(getattr(MicroSeries, fn)) + and getattr(getattr(MicroSeries, fn), "_rtype") == float +] +MicroSeries.VECTOR_FUNCTIONS = [ + fn + for fn in dir(MicroSeries) + if "_rtype" in dir(getattr(MicroSeries, fn)) + and getattr(getattr(MicroSeries, fn), "_rtype") == pd.Series +] +MicroSeries.AGNOSTIC_FUNCTIONS = ["quantile"] +MicroSeries.FUNCTIONS = sum( + [ + MicroSeries.SCALAR_FUNCTIONS, + MicroSeries.VECTOR_FUNCTIONS, + MicroSeries.AGNOSTIC_FUNCTIONS, + ], + [], +) + + +class MicroSeriesGroupBy(pd.core.groupby.generic.SeriesGroupBy): + def _init(self): + def _weighted_agg(name) -> Callable: + def via_micro_series(row, *args, **kwargs): + return getattr(MicroSeries(row.a, weights=row.w), name)(*args, **kwargs) + + fn = getattr(MicroSeries, name) + + @wraps(fn) + def _weighted_agg_fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: + arrays = self.apply(np.array) + weights = self.weights.apply(np.array) + df = pd.DataFrame(dict(a=arrays, w=weights)) + is_array = len(args) > 0 and hasattr(args[0], "__len__") + if ( + name in MicroSeries.SCALAR_FUNCTIONS + or name in MicroSeries.AGNOSTIC_FUNCTIONS + and not is_array + ): + result = df.agg( + lambda row: via_micro_series(row, *args, **kwargs), + axis=1, + ) + elif ( + name in MicroSeries.VECTOR_FUNCTIONS + or name in MicroSeries.AGNOSTIC_FUNCTIONS + and is_array + ): + if name in MicroSeries.AGNOSTIC_FUNCTIONS and not df.empty: + # Concatenate values without keys: concat rejects missing + # MultiIndex keys even when groupby(dropna=False) retains + # them. Reuse the grouping levels and codes so missing + # labels keep the same representation as scalar results. + results = [ + via_micro_series(row, *args, **kwargs) + for _, row in df.iterrows() + ] + result = pd.concat(results) + group_index = ( + df.index + if isinstance(df.index, pd.MultiIndex) + else pd.MultiIndex.from_arrays([df.index]) + ) + quantile_codes, quantile_levels = result.index.factorize( + sort=False + ) + result.index = pd.MultiIndex( + levels=[*group_index.levels, quantile_levels], + codes=[ + codes.repeat(len(results[0])) + for codes in group_index.codes + ] + + [quantile_codes], + names=[*df.index.names, result.index.name], + # Existing group codes are valid; checking would + # rewrite their retained missing labels to -1. + verify_integrity=False, + ) + return result + result = df.apply( + lambda row: via_micro_series(row, *args, **kwargs), + axis=1, + ) + return result.stack() + return result + + return _weighted_agg_fn + + for fn_name in MicroSeries.FUNCTIONS: + setattr(self, fn_name, _weighted_agg(fn_name)) diff --git a/build/lib/microdf/replication.py b/build/lib/microdf/replication.py new file mode 100644 index 00000000..0bb128e0 --- /dev/null +++ b/build/lib/microdf/replication.py @@ -0,0 +1,123 @@ +"""Variance estimation from replicate weights. + +Many survey products publish a set of replicate weight vectors alongside the +main weight. Recomputing a statistic once per replicate and measuring the +spread gives a variance estimate that requires no analytic formula, which is +what makes it usable for statistics such as the Gini coefficient or a +quantile where the analytic variance is awkward. + +The scale factor depends on how the replicates were constructed, so the +method must be named rather than guessed. + +Note that this is only valid for replicate weights as published with a +survey. Weights that have been calibrated or reweighted to external targets +no longer correspond to the original replication scheme, and applying these +estimators to them does not describe the variance of the resulting estimator. +""" + +from typing import Callable, Optional, Union + +import numpy as np +import pandas as pd + +# Scale applied to the sum of squared deviations from the full-sample +# estimate. R is the number of replicates. +METHOD_FACTORS = { + # Delete-a-group jackknife: (R - 1) / R. + "jackknife": lambda r: (r - 1) / r, + # Balanced repeated replication: 1 / R. + "brr": lambda r: 1 / r, + # Bootstrap replicates: 1 / R. + "bootstrap": lambda r: 1 / r, + # Successive difference replication, as used for the ACS and CPS: 4 / R. + "successive-difference": lambda r: 4 / r, +} + + +def _fay_factor(r: int, fay_k: float) -> float: + """Scale for Fay's variant of BRR, which perturbs rather than deletes.""" + if not 0 <= fay_k < 1: + raise ValueError(f"fay_k must be in [0, 1), got {fay_k}") + return 1 / (r * (1 - fay_k) ** 2) + + +def replicate_variance( + series, + statistic: Callable, + replicate_weights: Union[np.ndarray, pd.DataFrame], + method: str = "jackknife", + fay_k: Optional[float] = None, +) -> float: + """Variance of ``statistic`` estimated from replicate weights. + + :param series: A MicroSeries. Its own weights give the point estimate. + :param statistic: Callable taking a MicroSeries and returning a float, + for example ``lambda s: s.gini()``. + :param replicate_weights: Array or frame of shape ``(len(series), R)``. + :param method: One of ``jackknife``, ``brr``, ``bootstrap``, + ``successive-difference``, or ``fay`` (which requires ``fay_k``). + :param fay_k: Fay's perturbation constant, required when + ``method="fay"``. + :returns: The estimated variance of the statistic. + """ + from microdf.microseries import MicroSeries + + weights = np.asarray(replicate_weights, dtype=float) + if weights.ndim != 2: + raise ValueError( + f"replicate_weights must be 2-dimensional, got shape {weights.shape}" + ) + if weights.shape[0] != len(series): + raise ValueError( + f"replicate_weights has {weights.shape[0]} rows but the series has " + f"{len(series)}" + ) + + n_replicates = weights.shape[1] + if n_replicates < 2: + raise ValueError("At least two replicate weights are required") + + if method == "fay": + if fay_k is None: + raise ValueError("method='fay' requires fay_k") + factor = _fay_factor(n_replicates, fay_k) + elif method in METHOD_FACTORS: + if fay_k is not None: + raise ValueError("fay_k applies only to method='fay'") + factor = METHOD_FACTORS[method](n_replicates) + else: + known = ", ".join(sorted([*METHOD_FACTORS, "fay"])) + raise ValueError(f"Unknown method {method!r}; expected one of {known}") + + point = float(statistic(series)) + values = np.asarray(series, dtype=float) + index = series.index + + deviations = [] + for column in range(n_replicates): + replicate = MicroSeries( + values, weights=weights[:, column], index=index + ) + deviations.append(float(statistic(replicate)) - point) + + return factor * float(np.sum(np.square(deviations))) + + +def replicate_standard_error( + series, + statistic: Callable, + replicate_weights: Union[np.ndarray, pd.DataFrame], + method: str = "jackknife", + fay_k: Optional[float] = None, +) -> float: + """Standard error of ``statistic``, the square root of its variance. + + Takes the same arguments as :func:`replicate_variance`. + """ + return float( + np.sqrt( + replicate_variance( + series, statistic, replicate_weights, method, fay_k + ) + ) + ) diff --git a/build/lib/microdf/tests/conftest.py b/build/lib/microdf/tests/conftest.py new file mode 100644 index 00000000..44b0369e --- /dev/null +++ b/build/lib/microdf/tests/conftest.py @@ -0,0 +1,9 @@ +import os + +import pytest + + +@pytest.fixture(scope="session") +def tests_path() -> str: + """""" + return os.path.abspath(os.path.dirname(__file__)) diff --git a/build/lib/microdf/tests/test_aggregation_errors.py b/build/lib/microdf/tests/test_aggregation_errors.py new file mode 100644 index 00000000..bc8dd98e --- /dev/null +++ b/build/lib/microdf/tests/test_aggregation_errors.py @@ -0,0 +1,57 @@ +import microdf as mdf +import numpy as np +import pandas as pd +import pytest + + +def test_aggregation_surfaces_real_errors(): + """A genuine argument error must raise, not be swallowed per column.""" + df = mdf.MicroDataFrame(pd.DataFrame({"x": [1, 2, 3]}), weights=[1, 1, 1]) + with pytest.raises(ValueError, match="Unknown negatives option"): + df.gini(negatives="bogus") + + +def test_aggregation_still_skips_non_numeric_columns(): + """Narrowing the guard must not change which columns aggregate.""" + df = mdf.MicroDataFrame( + pd.DataFrame( + { + "x": [1, 2, 3], + "s": ["a", "b", "c"], + "dt": pd.to_datetime(["2020-01-01"] * 3), + } + ), + weights=[1, 2, 3], + ) + assert list(df.sum().index) == ["x"] + assert df.sum()["x"] == 14 + assert list(df.mean().index) == ["x"] + + +def test_gini_shift_accepts_nonnegative_and_negative_columns(): + frame = mdf.MicroDataFrame( + {"positive": [1, 2, 3], "negative": [-1, 1, 3]}, weights=[1, 1, 1] + ) + # Shift leaves [1, 2, 3] unchanged and changes [-1, 1, 3] to [0, 2, 4]. + # The pairwise-difference Ginis are 2/9 and 4/9, respectively. + expected = pd.Series({"positive": 2 / 9, "negative": 4 / 9}) + pd.testing.assert_series_equal(frame.gini(negatives="shift"), expected) + + +@pytest.mark.parametrize("selected", [False, True]) +def test_grouped_gini_shift_accepts_positive_groups(selected): + frame = mdf.MicroDataFrame( + {"group": ["a", "a", "a", "b", "b", "b"], "value": [1, 2, 3, -1, 1, 3]}, + weights=[1, 1, 1, 1, 1, 1], + ) + grouped = frame.groupby("group") + if selected: + grouped = grouped[["value"]] + expected = pd.DataFrame( + {"value": [2 / 9, 4 / 9]}, index=pd.Index(["a", "b"], name="group") + ) + pd.testing.assert_frame_equal(grouped.gini(negatives="shift"), expected) + + +def test_gini_shift_accepts_empty_series(): + assert np.isnan(mdf.MicroSeries([], weights=[]).gini(negatives="shift")) diff --git a/build/lib/microdf/tests/test_dataframe_weight_storage.py b/build/lib/microdf/tests/test_dataframe_weight_storage.py new file mode 100644 index 00000000..fc766096 --- /dev/null +++ b/build/lib/microdf/tests/test_dataframe_weight_storage.py @@ -0,0 +1,61 @@ +import warnings + +import numpy as np +import pandas as pd +import pytest + +import microdf as mdf + + +def test_weights_stay_a_series_after_nullify(): + """nullify_weights must leave weights as an index-aligned Series.""" + df = mdf.MicroDataFrame(pd.DataFrame({"x": [1, 2, 3]}), weights=[4, 5, 6]) + df.nullify_weights() + assert isinstance(df.weights, pd.Series) + assert list(df.weights.index) == list(df.index) + assert df.equals(df) + assert df.sum()["x"] == 6 + + +def test_weights_stay_a_series_after_set_weight_col(): + """The deprecated set_weight_col must also produce a Series.""" + df = mdf.MicroDataFrame(pd.DataFrame({"x": [1, 2, 3], "w": [1.0, 2.0, 3.0]})) + with warnings.catch_warnings(): + warnings.simplefilter("ignore", DeprecationWarning) + df.set_weight_col("w") + assert isinstance(df.weights, pd.Series) + assert df.weights_col == "w" + assert df.equals(df) + assert df.sum()["x"] == 14 + + +@pytest.mark.parametrize("dtype", ["int64", "float64"]) +def test_weight_column_edits_do_not_change_stored_weights(dtype): + """Selecting a weight column takes a copy of its current values.""" + df = mdf.MicroDataFrame( + {"x": [10, 20], "w": np.array([1, 2], dtype=dtype)}, + index=[10, 20], + ) + with pytest.warns(DeprecationWarning): + df.set_weight_col("w") + + df.loc[10, "w"] = 100 + + np.testing.assert_array_equal(df.weights, [1, 2]) + assert df.sum()["x"] == 10 * 1 + 20 * 2 + + +@pytest.mark.parametrize("dtype", ["int64", "float64"]) +def test_stored_weight_edits_do_not_change_weight_column(dtype): + """Changing stored weights leaves the source column values intact.""" + df = mdf.MicroDataFrame( + {"x": [10, 20], "w": np.array([1, 2], dtype=dtype)}, + index=[10, 20], + ) + with pytest.warns(DeprecationWarning): + df.set_weight_col("w") + + df.weights.iloc[0] = 100 + + np.testing.assert_array_equal(df["w"], [1, 2]) + assert df.sum()["x"] == 10 * 100 + 20 * 2 diff --git a/build/lib/microdf/tests/test_microseries_dataframe.py b/build/lib/microdf/tests/test_microseries_dataframe.py new file mode 100644 index 00000000..a73e8bb5 --- /dev/null +++ b/build/lib/microdf/tests/test_microseries_dataframe.py @@ -0,0 +1,817 @@ +import warnings + +import numpy as np +import pandas as pd +import pytest + +import microdf as mdf +from microdf.microdataframe import MicroDataFrame +from microdf.microseries import MicroSeries + + +def test_df_init() -> None: + arr = np.array([0, 1, 1]) + w = np.array([3, 0, 9]) + df = mdf.MicroDataFrame({"a": arr}, weights=w) + assert df.a.mean() == np.average(arr, weights=w) + + df = mdf.MicroDataFrame() + df["a"] = arr + df.set_weights(w) + assert df.a.mean() == np.average(arr, weights=w) + + df = mdf.MicroDataFrame() + df["a"] = arr + df["w"] = w + df.set_weight_col("w") + assert df.a.mean() == np.average(arr, weights=w) + + # Test set_weights with string (column name) + df2 = mdf.MicroDataFrame() + df2["a"] = arr + df2["w"] = w + df2.set_weights("w") # Using string column name instead of set_weight_col + assert df2.a.mean() == np.average(arr, weights=w) + assert np.array_equal(df2.weights.values, w) + + +def test_handles_empty_index() -> None: + arr = np.array([0, 1, 1]) + w = np.array([3, 0, 9]) + df = mdf.MicroDataFrame({"a": arr}, weights=w) + + empty_index = pd.Index([]) + df[empty_index] # Implicit assert; checking for ValueError + + +def test_series_getitem() -> None: + arr = np.array([0, 1, 1]) + w = np.array([3, 0, 9]) + s = mdf.MicroSeries(arr, weights=w) + assert s[[1, 2]].sum() == np.sum(arr[[1, 2]] * w[[1, 2]]) + + assert s[1:3].sum() == np.sum(arr[1:3] * w[1:3]) + + +def test_sum() -> None: + arr = np.array([0, 1, 1]) + w = np.array([3, 0, 9]) + series = mdf.MicroSeries(arr, weights=w) + assert series.sum() == (arr * w).sum() + + arr = np.linspace(-20, 100, 100) + w = np.linspace(1, 3, 100) + series = mdf.MicroSeries(arr) + series.set_weights(w) + assert series.sum() == (arr * w).sum() + + # Verify that an error is thrown when passing weights of different size + # from the values. + w = np.linspace(1, 3, 101) + series = mdf.MicroSeries(arr) + try: + series.set_weights(w) + assert False + except Exception: + pass + + +def test_mean() -> None: + arr = np.array([3, 0, 2]) + w = np.array([4, 1, 1]) + series = mdf.MicroSeries(arr, weights=w) + assert series.mean() == np.average(arr, weights=w) + + arr = np.linspace(-20, 100, 100) + w = np.linspace(1, 3, 100) + series = mdf.MicroSeries(arr) + series.set_weights(w) + assert series.mean() == np.average(arr, weights=w) + + w = np.linspace(1, 3, 101) + series = mdf.MicroSeries(arr) + try: + series.set_weights(w) + assert False + except Exception: + pass + + +def test_mean_skipna() -> None: + # Test skipna=True (default) - should skip NaN values + arr = np.array([3.0, np.nan, 2.0]) + w = np.array([4.0, 1.0, 1.0]) + series = mdf.MicroSeries(arr, weights=w) + + # skipna=True should exclude NaN and its weight + expected = np.average([3.0, 2.0], weights=[4.0, 1.0]) + assert series.mean(skipna=True) == expected + assert series.mean() == expected # Default should be skipna=True + + # Test skipna=False - should return NaN if any value is NaN + assert np.isnan(series.mean(skipna=False)) + + # Test with all NaN values + arr_all_nan = np.array([np.nan, np.nan, np.nan]) + w_all_nan = np.array([1.0, 2.0, 3.0]) + series_all_nan = mdf.MicroSeries(arr_all_nan, weights=w_all_nan) + assert np.isnan(series_all_nan.mean(skipna=True)) + assert np.isnan(series_all_nan.mean(skipna=False)) + + # Test with no NaN values - skipna should not affect result + arr_no_nan = np.array([3.0, 5.0, 2.0]) + w_no_nan = np.array([4.0, 1.0, 1.0]) + series_no_nan = mdf.MicroSeries(arr_no_nan, weights=w_no_nan) + expected_no_nan = np.average(arr_no_nan, weights=w_no_nan) + assert series_no_nan.mean(skipna=True) == expected_no_nan + assert series_no_nan.mean(skipna=False) == expected_no_nan + + +def test_poverty_count() -> None: + arr = np.array([10000, 20000, 50000]) + w = np.array([1123, 1144, 2211]) + df = pd.DataFrame() + df["income"] = arr + df["threshold"] = 16000 + df = MicroDataFrame(df, weights=w) + assert df.poverty_count("income", "threshold") == w[0] + + +def test_median() -> None: + # 1, 2, 3, 4, *4*, 4, 5, 5, 5 + arr = np.array([1, 2, 3, 4, 5]) + w = np.array([1, 1, 1, 3, 3]) + series = mdf.MicroSeries(arr, weights=w) + assert series.median() == 4 + + +def test_weighted_quantile_skewed() -> None: + # 99% of the population has 0 income, 1% has 1M + # The median should be 0, not an interpolated value + series = mdf.MicroSeries([0, 1_000_000], weights=[99, 1]) + assert series.median() == 0 + assert series.quantile(0.5) == 0 + # 99th percentile is still 0 since exactly 99% have 0 + assert series.quantile(0.99) == 0 + # Only quantile > 0.99 gives 1M + assert series.quantile(1.0) == 1_000_000 + # Test multiple quantiles + result = series.quantile([0.1, 0.5, 0.99, 1.0]) + assert result[0.1] == 0 + assert result[0.5] == 0 + assert result[0.99] == 0 + assert result[1.0] == 1_000_000 + + +def test_weighted_quantile_boundaries() -> None: + # Test q=0 returns minimum, q=1 returns maximum + series = mdf.MicroSeries([10, 20, 30], weights=[1, 1, 1]) + assert series.quantile(0.0) == 10 + assert series.quantile(1.0) == 30 + + +def test_weighted_quantile_equal_weights() -> None: + # With equal weights, should match "replicated" interpretation + # Values: 1, 2, 3 each with weight 2 -> like [1,1,2,2,3,3] + series = mdf.MicroSeries([1, 2, 3], weights=[2, 2, 2]) + # cumsum_normalized = [2/6, 4/6, 6/6] = [0.333, 0.667, 1.0] + # median (0.5): smallest where cumsum >= 0.5 -> index 1 -> value 2 + assert series.median() == 2 + # 0.25 quantile: smallest where cumsum >= 0.25 -> index 0 -> value 1 + assert series.quantile(0.25) == 1 + # 0.75 quantile: smallest where cumsum >= 0.75 -> index 2 -> value 3 + assert series.quantile(0.75) == 3 + + +def test_weighted_quantile_unsorted_input() -> None: + # Ensure sorting works correctly + series = mdf.MicroSeries([30, 10, 20], weights=[1, 2, 1]) + # Sorted: values [10, 20, 30], weights [2, 1, 1] + # cumsum_normalized = [0.5, 0.75, 1.0] + assert series.quantile(0.0) == 10 + assert series.quantile(0.5) == 10 # cumsum[0]=0.5 >= 0.5 + assert series.quantile(0.6) == 20 # cumsum[1]=0.75 >= 0.6 + assert series.quantile(1.0) == 30 + + +def test_unweighted_groupby() -> None: + df = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}) + assert (df.groupby("x").z.sum().values == np.array([5.0, 6.0])).all() + + +def test_multiple_groupby() -> None: + df = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}) + assert (df.groupby(["x", "y"]).z.sum() == np.array([5, 6])).all() + + +def test_set_index() -> None: + d = mdf.MicroDataFrame(dict(x=[1, 2, 3]), weights=[4, 5, 6]) + assert d.x.__class__ == MicroSeries + d.index = [1, 2, 3] + assert d.x.__class__ == MicroSeries + + +def test_reset_index() -> None: + d = mdf.MicroDataFrame(dict(x=[1, 2, 3]), weights=[4, 5, 6]) + assert d.reset_index().__class__ == MicroDataFrame + + +def test_cumsum() -> None: + s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) + assert np.array_equal(s.cumsum().values, [4, 14, 32]) + + s = mdf.MicroSeries([2, 1, 3], weights=[5, 4, 6]) + assert np.array_equal(s.cumsum().values, [10, 14, 32]) + + s = mdf.MicroSeries([3, 1, 2], weights=[6, 4, 5]) + assert np.array_equal(s.cumsum().values, [18, 22, 32]) + + +def test_rank() -> None: + s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) + assert np.array_equal(s.rank().values, [4, 9, 15]) + + s = mdf.MicroSeries([3, 1, 2], weights=[6, 4, 5]) + assert np.array_equal(s.rank().values, [15, 4, 9]) + + s = mdf.MicroSeries([2, 1, 3], weights=[5, 4, 6]) + assert np.array_equal(s.rank().values, [9, 4, 15]) + + +def test_percentile_rank() -> None: + s = mdf.MicroSeries([4, 2, 3, 1], weights=[20, 40, 20, 20]) + assert np.array_equal(s.percentile_rank().values, [100, 60, 80, 20]) + + +def test_quartile_rank() -> None: + s = mdf.MicroSeries([4, 2, 3], weights=[25, 50, 25]) + assert np.array_equal(s.quartile_rank().values, [4, 2, 3]) + + +def test_quintile_rank() -> None: + s = mdf.MicroSeries([4, 2, 3], weights=[20, 60, 20]) + assert np.array_equal(s.quintile_rank().values, [5, 3, 4]) + + +def test_decile_rank() -> None: + s = mdf.MicroSeries( + [5, 4, 3, 2, 1, 6, 7, 8, 9], + weights=[10, 20, 10, 10, 10, 10, 10, 10, 10], + ) + assert np.array_equal(s.decile_rank().values, [6, 5, 3, 2, 1, 7, 8, 9, 10]) + + +def test_copy_equals() -> None: + d = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8]) + d_copy = d.copy() + d_copy_diff_weights = d_copy.copy() + d_copy_diff_weights.weights *= 2 + assert d.equals(d_copy) + assert not d.equals(d_copy_diff_weights) + # Same for a MicroSeries. + assert d.x.equals(d_copy.x) + assert not d.x.equals(d_copy_diff_weights.x) + + +def test_subset() -> None: + df = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8]) + df_no_z = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4]}, weights=[7, 8]) + assert df[["x", "y"]].equals(df_no_z) + df_no_z_diff_weights = df_no_z.copy() + df_no_z_diff_weights.weights += 1 + assert not df[["x", "y"]].equals(df_no_z_diff_weights) + + +def test_value_subset() -> None: + d = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [1, 2, 2]}, weights=[4, 5, 6]) + d2 = d[d.y > 1] + assert d2.y.shape == d2.weights.shape + + +def test_bitwise_ops_return_microseries() -> None: + s1 = mdf.MicroSeries([True, False, True], weights=[1, 2, 3]) + s2 = mdf.MicroSeries([False, False, True], weights=[1, 2, 3]) + and_result = s1 & s2 + or_result = s1 | s2 + assert isinstance(and_result, mdf.MicroSeries) + assert isinstance(or_result, mdf.MicroSeries) + expected_and = mdf.MicroSeries([False, False, True], weights=[1, 2, 3]) + expected_or = mdf.MicroSeries([True, False, True], weights=[1, 2, 3]) + assert and_result.equals(expected_and) + assert or_result.equals(expected_or) + + +def test_additional_ops_return_microseries() -> None: + s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) + radd = 1 + s + xor = s ^ mdf.MicroSeries([0, 1, 0], weights=[4, 5, 6]) + inv = ~mdf.MicroSeries([True, False], weights=[1, 1]) + assert isinstance(radd, mdf.MicroSeries) + assert isinstance(xor, mdf.MicroSeries) + assert isinstance(inv, mdf.MicroSeries) + + +def test_reset_index_inplace() -> None: + df = pd.DataFrame( + {"A": [1, 2, 3, 4], "B": [5, 6, 7, 8]}, index=["a", "b", "c", "d"] + ) + weights = np.array([0.1, 0.2, 0.3, 0.4]) + mdf = MicroDataFrame(df, weights=weights) + + # Test 1: reset_index with inplace=False (default) + mdf_copy = mdf.copy() + result = mdf_copy.reset_index() + assert list(mdf_copy.index) == ["a", "b", "c", "d"] + assert list(result.index) == [0, 1, 2, 3] + assert "index" in result.columns + assert list(result["index"]) == ["a", "b", "c", "d"] + np.testing.assert_array_equal(result.weights.values, weights) + + # Test 2: reset_index with inplace=True + mdf_copy = mdf.copy() + result = mdf_copy.reset_index(inplace=True) + assert result is None + assert list(mdf_copy.index) == [0, 1, 2, 3] + assert "index" in mdf_copy.columns + assert list(mdf_copy["index"]) == ["a", "b", "c", "d"] + np.testing.assert_array_equal(mdf_copy.weights.values, weights) + assert isinstance(mdf_copy["A"], MicroSeries) + assert isinstance(mdf_copy["B"], MicroSeries) + assert isinstance(mdf_copy["index"], MicroSeries) + + # Test 3: reset_index with drop=True + mdf_copy = mdf.copy() + mdf_copy.reset_index(drop=True, inplace=True) + assert list(mdf_copy.index) == [0, 1, 2, 3] + assert "index" not in mdf_copy.columns + assert list(mdf_copy.columns) == ["A", "B"] + np.testing.assert_array_equal(mdf_copy.weights.values, weights) + + # Test 4: Multi-level index + arrays = [["bar", "bar", "baz", "baz"], ["one", "two", "one", "two"]] + multi_index = pd.MultiIndex.from_arrays(arrays, names=["first", "second"]) + df_multi = pd.DataFrame({"A": [1, 2, 3, 4], "B": [5, 6, 7, 8]}, index=multi_index) + mdf_multi = MicroDataFrame(df_multi, weights=weights) + result = mdf_multi.reset_index(level="first") + assert "first" in result.columns + assert result.index.name == "second" + np.testing.assert_array_equal(result.weights.values, weights) + + # Reset all levels in place + mdf_multi.reset_index(inplace=True) + assert "first" in mdf_multi.columns + assert "second" in mdf_multi.columns + assert list(mdf_multi.index) == [0, 1, 2, 3] + np.testing.assert_array_equal(mdf_multi.weights.values, weights) + + +def test_loc_preserves_weights() -> None: + """Test that .loc[] returns MicroDataFrame with proper weights (issue + #265).""" + df = mdf.MicroDataFrame({"one": [1, 1, 1, 1, 1]}, weights=[10, 20, 30, 40, 50]) + + # Filter all rows (should get same weights) + filtered = df.loc[df.one == 1] + assert isinstance(filtered, MicroDataFrame) + assert filtered.one.sum() == 150.0 # Weighted sum + + # Partial filter + df2 = mdf.MicroDataFrame({"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50]) + subset = df2.loc[df2.x > 2] + assert isinstance(subset, MicroDataFrame) + assert subset.x.sum() == 500.0 # 3*30 + 4*40 + 5*50 = 500 + np.testing.assert_array_equal(subset.weights.values, [30.0, 40.0, 50.0]) + + +def test_iloc_preserves_weights() -> None: + """Test that .iloc[] returns MicroDataFrame with proper weights.""" + df = mdf.MicroDataFrame({"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50]) + + # Select rows by position + subset = df.iloc[2:5] + assert isinstance(subset, MicroDataFrame) + assert subset.x.sum() == 500.0 # 3*30 + 4*40 + 5*50 = 500 + np.testing.assert_array_equal(subset.weights.values, [30.0, 40.0, 50.0]) + + +def test_groupby_column_selection() -> None: + """Test that groupby column selection preserves weights (issue #193).""" + d = mdf.MicroDataFrame(dict(g=["a", "a", "b"], y=[1, 2, 3]), weights=[4, 5, 6]) + + # Test single column string selection + result_str = d.groupby("g")["y"].sum() + assert result_str["a"] == 14.0 # 1*4 + 2*5 = 14 + assert result_str["b"] == 18.0 # 3*6 = 18 + + # Test list column selection + result_list = d.groupby("g")[["y"]].sum() + assert result_list.loc["a", "y"] == 14.0 + assert result_list.loc["b", "y"] == 18.0 + + # Aggregated results should be plain DataFrame (no spurious weight column) + result_all = d.groupby("g").sum() + assert "weight" not in result_all.columns + assert list(result_all.columns) == ["y"] + + +def test_values_warns() -> None: + """Accessing .values on a MicroSeries should emit a UserWarning.""" + ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + _ = ms.values + assert len(w) == 1 + assert issubclass(w[0].category, UserWarning) + assert "weights" in str(w[0].message).lower() + + +def test_to_numpy_warns() -> None: + """Calling .to_numpy() on a MicroSeries should emit a UserWarning.""" + ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + _ = ms.to_numpy() + assert len(w) == 1 + assert issubclass(w[0].category, UserWarning) + assert "weights" in str(w[0].message).lower() + + +def test_mean_no_warning() -> None: + """Internal .values usage in .mean() should NOT emit a warning.""" + ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + _ = ms.mean() + user_warnings = [x for x in w if issubclass(x.category, UserWarning)] + assert len(user_warnings) == 0 + + +def test_sum_with_non_default_index() -> None: + """Weighted sum must not silently return 0 with a non-default index. + + Regression test for the bug where ``set_weights`` stored the weights Series + with a default ``RangeIndex`` regardless of ``self.index``. Element-wise + ops like ``self.multiply(self.weights)`` then aligned on label, producing + all-NaN and a silent ``0.0`` from ``.sum()`` while ``.mean()`` (which uses + a positional ndarray) stayed correct. + """ + # MicroSeries with custom integer index. + s = mdf.MicroSeries([1, 2, 3], index=[100, 200, 300], weights=[10, 20, 30]) + assert s.sum() == 140.0 + assert s.weights.index.tolist() == [100, 200, 300] + + # MicroDataFrame with custom integer index. + df = mdf.MicroDataFrame( + {"x": [1, 2, 3]}, index=[100, 200, 300], weights=[10, 20, 30] + ) + assert df.x.sum() == 140.0 + assert df.weights.index.tolist() == [100, 200, 300] + + # MicroDataFrame with string index + set_weights via column name. + df2 = mdf.MicroDataFrame( + {"x": [1, 2, 3], "w": [10, 20, 30]}, + index=["a", "b", "c"], + ) + df2.set_weights("w") + assert df2.x.sum() == 140.0 + assert df2.weights.index.tolist() == ["a", "b", "c"] + + # Passing a Series with its own index should position-align, not + # label-align, so sum does not depend on accidental index alignment. + df3 = mdf.MicroDataFrame({"x": [1, 2, 3]}, index=[100, 200, 300]) + df3.set_weights(pd.Series([10, 20, 30], index=[0, 1, 2])) + assert df3.x.sum() == 140.0 + + +def test_repr_no_warning() -> None: + """Internal .values usage in __repr__ should NOT emit a warning.""" + ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + _ = repr(ms) + user_warnings = [x for x in w if issubclass(x.category, UserWarning)] + assert len(user_warnings) == 0 + + +def test_drop_inplace_aligns_weights() -> None: + """Regression: ``drop(inplace=True)`` must keep weights in sync. + + Previously, ``weights_backup = self.weights.copy()`` was taken *before* + the drop, then reassigned back afterwards — so the weights vector + kept its original length and any subsequent weighted op either + raised a length-mismatch ValueError or silently returned 0. + """ + # Row drop inplace. + df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40]) + df.drop(index=[0, 1], inplace=True) + assert len(df) == len(df.weights) == 2 + assert df.x.sum() == 3 * 30 + 4 * 40 # 250 + + # Row drop non-inplace. + df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40]) + df2 = df.drop(index=[0, 1]) + assert len(df2) == len(df2.weights) == 2 + assert df2.x.sum() == 250 + # Original untouched. + assert len(df) == 4 + assert df.x.sum() == 1 * 10 + 2 * 20 + 3 * 30 + 4 * 40 + + # Column drop (weights length unchanged). + df = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [4, 5, 6]}, weights=[10, 20, 30]) + df.drop(columns=["y"], inplace=True) + assert list(df.columns) == ["x"] + assert df.x.sum() == 1 * 10 + 2 * 20 + 3 * 30 + + # String index row drop. + df = mdf.MicroDataFrame( + {"x": [1, 2, 3, 4]}, + index=["a", "b", "c", "d"], + weights=[10, 20, 30, 40], + ) + df.drop(index=["a", "b"], inplace=True) + assert df.x.sum() == 250 + + # labels= with default axis=0. + df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40]) + df.drop(labels=[0, 1], inplace=True) + assert df.x.sum() == 250 + + +def test_merge_preserves_weights_per_surviving_row() -> None: + """Regression: merge must propagate weights onto the merged rows. + + Previously the implementation passed ``self.weights`` straight to the + MicroDataFrame constructor, so any merge that changed row count (inner + filtering, left-with-missing, many-to-many, outer) raised ``ValueError: + Length of weights (N) does not match length of DataFrame (M)``. + """ + # Inner join filters rows. + left = mdf.MicroDataFrame( + {"k": [1, 2, 3, 4], "v": [10, 20, 30, 40]}, weights=[1, 2, 3, 4] + ) + right = pd.DataFrame({"k": [2, 4], "w": [20, 40]}) + res = left.merge(right, on="k") + assert len(res) == 2 + # k=2 carries weight 2; k=4 carries weight 4. + np.testing.assert_array_equal(sorted(res.weights.values), [2.0, 4.0]) + assert res.v.sum() == 2 * 20 + 4 * 40 + + # Left join with missing from right. + left = mdf.MicroDataFrame( + {"k": [1, 2, 3, 4], "v": [10, 20, 30, 40]}, weights=[1, 2, 3, 4] + ) + right = pd.DataFrame({"k": [2, 4], "w": [100, 200]}) + res = left.merge(right, on="k", how="left") + assert len(res) == 4 + assert res.v.sum() == 300 # 1*10 + 2*20 + 3*30 + 4*40 + + # Many-to-many duplicates left rows; the same weight should ride + # along on each duplicate. + left = mdf.MicroDataFrame({"k": [1, 2], "v": [10, 20]}, weights=[5, 7]) + right = pd.DataFrame({"k": [1, 1, 2], "w": [100, 200, 300]}) + res = left.merge(right, on="k") + assert len(res) == 3 + # v=10 weighted 5 appears twice, v=20 weighted 7 appears once. + assert res.v.sum() == 10 * 5 + 10 * 5 + 20 * 7 + + # Outer join: right-only rows have no left weight. We default to 0 + # so they don't silently poison downstream aggregations. + left = mdf.MicroDataFrame({"k": [1, 2], "v": [10, 20]}, weights=[1, 2]) + right = pd.DataFrame({"k": [2, 3], "w": [20, 30]}) + res = left.merge(right, on="k", how="outer") + # k=1 -> weight 1, k=2 -> weight 2, k=3 -> weight 0 (right-only). + assert sorted(res.weights.values) == [0.0, 1.0, 2.0] + + +def test_groupby_does_not_leak_tmp_weights_column() -> None: + """Regression: groupby used to mutate self by adding __tmp_weights. + + Previously, ``MicroDataFrame.groupby`` set ``self["__tmp_weights"]`` and + never cleaned it up, so ``df.columns`` afterwards included the weight + column and any later ``df.sum()`` or iteration over columns picked it up as + data. + """ + df = mdf.MicroDataFrame({"g": ["a", "a", "b"], "v": [1, 2, 3]}, weights=[1, 2, 3]) + original_cols = list(df.columns) + _ = df.groupby("g").sum() + assert list(df.columns) == original_cols + assert "__tmp_weights" not in df.columns + + # Groupby by a list of columns should also not leak. + df2 = mdf.MicroDataFrame( + {"g1": ["a", "a", "b"], "g2": [1, 1, 2], "v": [1, 2, 3]}, + weights=[1, 2, 3], + ) + orig2 = list(df2.columns) + _ = df2.groupby(["g1", "g2"]).v.sum() + assert list(df2.columns) == orig2 + + # Weighted aggregation is still correct after the fix. + result = df.groupby("g").v.sum() + assert result["a"] == 1 * 1 + 2 * 2 + assert result["b"] == 3 * 3 + + +def test_quantile_skips_zero_weight_rows() -> None: + """Regression: quantile(0) shouldn't pick a zero-weight element. + + Previously, ``np.searchsorted(cumsum_norm, 0, side='left')`` returned 0 + even when that first sorted element had zero weight, so ``MicroSeries([10, + 20, 30], weights=[0, 1, 1]).quantile(0)`` returned 10 instead of 20. The + fix drops zero-weight rows before computing the CDF. + """ + s = mdf.MicroSeries([10, 20, 30], weights=[0, 1, 1]) + assert s.quantile(0.0) == 20 + assert s.quantile(0.5) == 20 + assert s.quantile(1.0) == 30 + + # Internal plateau of zero weight. + s = mdf.MicroSeries([10, 20, 30, 40], weights=[1, 0, 1, 1]) + assert s.quantile(0.0) == 10 + # Post-filter values [10, 30, 40] with equal weights -> cum=[.33,.67,1]. + # 0.4 -> smallest cum >= 0.4 is index 1 -> value 30. + assert s.quantile(0.4) == 30 + # The zero-weight value (20) should never be selected. + for q in np.linspace(0, 1, 21): + assert s.quantile(q) != 20 + + # All zero weights -> NaN (defined behaviour). + s = mdf.MicroSeries([10, 20, 30], weights=[0, 0, 0]) + assert np.isnan(s.quantile(0.5)) + + +def test_top_x_pct_share_handles_ties_and_edges() -> None: + """Regression: top_x_pct_share double-counted threshold ties. + + Old implementation: ``self[self >= threshold].sum() / self.sum()``. + With constant values every call returned 1.0 regardless of the + requested top percent; ``top_x_pct_share(0)`` returned the share of + the max bucket instead of 0. + """ + # Constant values: the top p% should hold exactly p% of the total. + for p in [0.0, 0.01, 0.1, 0.5, 1.0]: + got = mdf.MicroSeries([5] * 10, weights=[1] * 10).top_x_pct_share(p) + assert np.isclose(got, p), f"top={p}, got {got}" + + # Non-constant, equal weights. + s = mdf.MicroSeries(list(range(1, 11)), weights=[1] * 10) + # Sum 1..10 = 55. Top 10% = top 1 row = 10 -> 10/55. + assert np.isclose(s.top_x_pct_share(0.1), 10 / 55) + # Top 50% = rows 6..10 -> 40/55. + assert np.isclose(s.top_x_pct_share(0.5), 40 / 55) + # Top 0% = 0, top 100% = 1. + assert s.top_x_pct_share(0.0) == 0.0 + assert s.top_x_pct_share(1.0) == 1.0 + + # Bottom share complements the top share. + assert np.isclose(s.bottom_x_pct_share(0.1), 1 - s.top_x_pct_share(0.9)) + + # Ties with unequal totals. + s_ties = mdf.MicroSeries([1, 1, 10, 10], weights=[1, 1, 1, 1]) + # Top 50% = the two 10s -> 20/22. + assert np.isclose(s_ties.top_x_pct_share(0.5), 20 / 22) + + # Downstream helpers still work. + assert np.isclose(s_ties.top_10_pct_share(), s_ties.top_x_pct_share(0.1)) + assert np.isclose(s_ties.top_50_pct_share(), s_ties.top_x_pct_share(0.5)) + + +def test_gini_negatives_option_applied() -> None: + """Regression: gini(negatives=...) was silently ignored. + + Both branches of the old implementation sorted ``self`` directly rather + than the local ``x`` that was mutated by the ``negatives`` option, so + ``negatives='zero'`` and ``negatives='shift'`` did nothing. + """ + s = mdf.MicroSeries([-5, 0, 10], weights=[1, 1, 1]) + + # Leaving negatives in place now warns. + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + _ = s.gini() + user_warnings = [x for x in w if issubclass(x.category, UserWarning)] + assert len(user_warnings) == 1 + assert "negative" in str(user_warnings[0].message).lower() + + # 'zero' clamps negatives. Values become [0, 0, 10] with equal + # weights; closed-form gini = 2/3. + assert np.isclose(s.gini(negatives="zero"), 2 / 3) + + # 'shift' adds |min|. Values become [0, 5, 15]; Gini in [0, 1]. + shifted = s.gini(negatives="shift") + assert 0 <= shifted <= 1 + + # All-zero short-circuits to 0 instead of nan/RuntimeWarning. + assert mdf.MicroSeries([0, 0, 0], weights=[1, 2, 3]).gini() == 0.0 + + # Invalid negatives arg raises. + with pytest.raises(ValueError): + mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1]).gini(negatives="bogus") + + +def test_std_var_are_weighted() -> None: + """Regression: std/var used to silently fall through to pandas. + + The old implementation had no override, so a MicroSeries with very uneven + weights returned the unweighted 1.0. Now std and var treat the weights as + frequency counts, matching numpy on the replicated sample. + """ + s = mdf.MicroSeries([1, 2, 3], weights=[100, 1, 1]) + # Unweighted would be 1.0. Weighted std pulls toward the heavy row. + assert s.std() < 1.0 + assert s.var() < 1.0 + + # Integer-replication equivalence. + s = mdf.MicroSeries([1, 2, 3], weights=[2, 3, 1]) + rep = np.array([1, 1, 2, 2, 2, 3]) + assert np.isclose(s.std(), np.std(rep, ddof=1)) + assert np.isclose(s.var(), np.var(rep, ddof=1)) + assert np.isclose(s.var(ddof=0), np.var(rep, ddof=0)) + + # NaN handling. + s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[2, 3, 1]) + assert not np.isnan(s.std()) + assert np.isnan(s.std(skipna=False)) + + # DataFrame dispatch: df.std() / df.var() now return weighted stats. + df = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [10, 20, 30]}, weights=[2, 3, 1]) + np.testing.assert_allclose( + df.std().values, + [ + np.std(rep, ddof=1), + np.std(np.array([10, 10, 20, 20, 20, 30]), ddof=1), + ], + ) + + +def test_cov_corr_warn_when_fallthrough() -> None: + """Regression: cov/corr silently returned unweighted pandas values. + + They still fall through to pandas (a weighted impl is a separate issue) but + now emit a UserWarning so callers aren't misled. + """ + s1 = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1]) + s2 = mdf.MicroSeries([2, 4, 6], weights=[1, 1, 1]) + + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + _ = s1.cov(s2) + msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)] + assert any("unweighted" in m.lower() for m in msgs) + + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + _ = s1.corr(s2) + msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)] + assert any("unweighted" in m.lower() for m in msgs) + + +def test_count_skips_nan_by_default() -> None: + """Regression: ``count()`` included NaN-row weight, contrary to pandas. + + Pandas ``Series.count`` skips NaN; MicroSeries returned the full weight sum + regardless. The fix matches pandas semantics and adds a ``skipna`` kwarg so + callers can opt out. + """ + s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[10, 20, 30]) + assert s.count() == 40.0 + assert s.count(skipna=True) == 40.0 + assert s.count(skipna=False) == 60.0 + + # No NaN: skipna is a no-op. + assert mdf.MicroSeries([1, 2, 3], weights=[2, 3, 4]).count() == 9.0 + + # All NaN: count skips everything. + all_nan = mdf.MicroSeries([np.nan] * 3, weights=[1, 2, 3]) + assert all_nan.count() == 0.0 + assert all_nan.count(skipna=False) == 6.0 + + +def test_rank_ties_share_bucket() -> None: + """Regression: rank used to assign ties to different ranks/buckets. + + Previously ``rank`` returned the running cumulative weight in sort order, + so every row — tied or not — got a distinct value. As a result + ``MicroSeries([5]*5, weights=[1]*5).decile_rank()`` returned ``[2, 4, 6, 8, + 10]`` rather than all 10. With max-rank semantics, tied values share the + cumulative weight at the end of their tie group, so bucketing is stable + under ties. + """ + # All tied: every element lands in the top decile. + s = mdf.MicroSeries([5] * 5, weights=[1] * 5) + np.testing.assert_array_equal(s.rank().values, [5, 5, 5, 5, 5]) + np.testing.assert_array_equal(s.decile_rank().values, [10] * 5) + np.testing.assert_array_equal(s.quintile_rank().values, [5] * 5) + + # Partial ties. + s = mdf.MicroSeries([1, 2, 2, 3], weights=[1, 1, 1, 1]) + np.testing.assert_array_equal(s.rank().values, [1, 3, 3, 4]) + + # pct=True normalizes to (0, 1] and still shares ranks on ties. + s = mdf.MicroSeries([5] * 4, weights=[1] * 4) + np.testing.assert_allclose(s.rank(pct=True).values, [1.0, 1.0, 1.0, 1.0]) + + # Non-ties still match the old cumulative-weight behaviour, so the + # existing ``test_rank`` expectations hold. + s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) + np.testing.assert_array_equal(s.rank().values, [4, 9, 15]) diff --git a/build/lib/microdf/tests/test_nullify_weights_index.py b/build/lib/microdf/tests/test_nullify_weights_index.py new file mode 100644 index 00000000..8d4f9c70 --- /dev/null +++ b/build/lib/microdf/tests/test_nullify_weights_index.py @@ -0,0 +1,10 @@ +import microdf as mdf + + +def test_nullify_weights_non_default_index(): + """nullify_weights must align to the index, not a fresh RangeIndex.""" + s = mdf.MicroSeries([1, 2, 3], index=[10, 11, 12], weights=[1, 2, 3]) + s.nullify_weights() + assert s.sum() == 6 + assert s.mean() == 2 + assert list(s.weights.index) == [10, 11, 12] diff --git a/build/lib/microdf/tests/test_pandas3_compatibility.py b/build/lib/microdf/tests/test_pandas3_compatibility.py new file mode 100644 index 00000000..1413780d --- /dev/null +++ b/build/lib/microdf/tests/test_pandas3_compatibility.py @@ -0,0 +1,242 @@ +"""Tests for pandas 3.0.0 compatibility in microdf. + +These tests verify that microdf works correctly with pandas 3.0.0, +which introduces: +1. PyArrow-backed strings as default (StringDtype) +2. Copy-on-Write by default +3. Changes to how Series subclasses are handled +""" + +import numpy as np +import pandas as pd + +from microdf.microdataframe import MicroDataFrame +from microdf.microseries import MicroSeries + + +class TestMicroSeriesSubclassPreservation: + """Test that MicroSeries subclass is preserved across operations.""" + + def test_microseries_set_weights_after_creation(self): + """Ensure set_weights works on MicroSeries. + + This is the error reported in pandas 3: + AttributeError: 'Series' object has no attribute 'set_weights' + """ + ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 1.0, 1.0])) + assert hasattr(ms, "set_weights") + assert hasattr(ms, "weights") + + # Should be able to call set_weights + ms.set_weights(np.array([2.0, 2.0, 2.0])) + assert np.allclose(ms.weights, [2.0, 2.0, 2.0]) + + def test_microseries_preserved_after_arithmetic(self): + """Arithmetic operations should return MicroSeries, not plain + Series.""" + ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) + + # Addition + result = ms + 1 + assert isinstance(result, MicroSeries), ( + f"Got {type(result)} instead of MicroSeries" + ) + assert hasattr(result, "weights") + assert hasattr(result, "set_weights") + + # Multiplication + result = ms * 2 + assert isinstance(result, MicroSeries), ( + f"Got {type(result)} instead of MicroSeries" + ) + + # Division + result = ms / 2 + assert isinstance(result, MicroSeries), ( + f"Got {type(result)} instead of MicroSeries" + ) + + def test_microseries_preserved_after_comparison(self): + """Comparison operations should return MicroSeries, not plain + Series.""" + ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) + + # Greater than + result = ms > 1 + assert isinstance(result, MicroSeries), ( + f"Got {type(result)} instead of MicroSeries" + ) + assert hasattr(result, "weights") + + # Less than + result = ms < 3 + assert isinstance(result, MicroSeries), ( + f"Got {type(result)} instead of MicroSeries" + ) + + def test_microseries_preserved_after_indexing(self): + """Indexing operations should return MicroSeries, not plain Series.""" + ms = MicroSeries([1, 2, 3, 4, 5], weights=np.array([1.0, 2.0, 3.0, 4.0, 5.0])) + + # Boolean indexing + result = ms[ms > 2] + assert isinstance(result, MicroSeries), ( + f"Got {type(result)} instead of MicroSeries" + ) + assert hasattr(result, "weights") + + # Slice indexing + result = ms[1:3] + assert isinstance(result, MicroSeries), ( + f"Got {type(result)} instead of MicroSeries" + ) + + +class TestMicroDataFrameSubclassPreservation: + """Test that MicroDataFrame column access returns MicroSeries.""" + + def test_microdataframe_column_returns_microseries(self): + """Accessing a column from MicroDataFrame should return MicroSeries.""" + mdf = MicroDataFrame( + {"a": [1, 2, 3], "b": [4, 5, 6]}, weights=np.array([1.0, 2.0, 3.0]) + ) + + # Column access + col = mdf["a"] + assert isinstance(col, MicroSeries), f"Got {type(col)} instead of MicroSeries" + assert hasattr(col, "weights") + assert hasattr(col, "set_weights") + + def test_microdataframe_operations_preserve_type(self): + """Operations on MicroDataFrame columns should preserve MicroSeries + type.""" + mdf = MicroDataFrame( + {"a": [1, 2, 3], "b": [4, 5, 6]}, weights=np.array([1.0, 2.0, 3.0]) + ) + + # Column operations + result = mdf["a"] + mdf["b"] + assert isinstance(result, MicroSeries), ( + f"Got {type(result)} instead of MicroSeries" + ) + assert hasattr(result, "weights") + + +class TestStringDtypeHandling: + """Test that MicroSeries/MicroDataFrame handle pandas 3 string dtypes.""" + + def test_microseries_with_string_data(self): + """MicroSeries should work with string data in pandas 3.""" + # Create with string data + ms = MicroSeries(["a", "b", "c"], weights=np.array([1.0, 2.0, 3.0])) + assert len(ms) == 3 + assert hasattr(ms, "weights") + + def test_microdataframe_with_string_columns(self): + """MicroDataFrame should work with string columns in pandas 3.""" + mdf = MicroDataFrame( + {"names": ["alice", "bob", "charlie"], "values": [1, 2, 3]}, + weights=np.array([1.0, 2.0, 3.0]), + ) + assert len(mdf) == 3 + + # String column access should still work + names = mdf["names"] + assert len(names) == 3 + + +class TestWeightedOperationsWithPandas3: + """Test that weighted operations work correctly with pandas 3.""" + + def test_weighted_sum(self): + """Weighted sum should work correctly.""" + ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) + # Weighted sum: 1*1 + 2*2 + 3*3 = 1 + 4 + 9 = 14 + assert ms.sum() == 14 + + def test_weighted_mean(self): + """Weighted mean should work correctly.""" + ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) + # Weighted mean: (1*1 + 2*2 + 3*3) / (1 + 2 + 3) = 14 / 6 ≈ 2.333 + assert np.isclose(ms.mean(), 14 / 6) + + def test_weighted_count(self): + """Weighted count should return sum of weights.""" + ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) + assert ms.count() == 6.0 + + +class TestCopyOnWriteCompatibility: + """Test compatibility with pandas 3 Copy-on-Write.""" + + def test_microseries_copy_independent(self): + """Copying a MicroSeries should create an independent copy.""" + ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) + ms_copy = ms.copy() + + # Modify original + ms.set_weights(np.array([4.0, 5.0, 6.0])) + + # Copy should be unchanged + assert np.allclose(ms_copy.weights, [1.0, 2.0, 3.0]) + + def test_microdataframe_copy_independent(self): + """Copying a MicroDataFrame should create an independent copy.""" + mdf = MicroDataFrame({"a": [1, 2, 3]}, weights=np.array([1.0, 2.0, 3.0])) + mdf_copy = mdf.copy() + + # Modify original + mdf.set_weights(np.array([4.0, 5.0, 6.0])) + + # Copy should be unchanged + assert np.allclose(mdf_copy.weights, [1.0, 2.0, 3.0]) + + def test_column_set_weights_after_access_regression(self): + """Regression test for pandas 3.0 CoW compatibility. + + In pandas 3.0 with Copy-on-Write, modifying column.__class__ doesn't + persist because each access returns a copy. This test verifies the fix + that wraps columns as MicroSeries on access in __getitem__. + """ + mdf = MicroDataFrame( + {"income": [10000, 20000, 30000]}, + weights=np.array([1.0, 2.0, 3.0]), + ) + + # This was the exact error that occurred: + # AttributeError: 'Series' object has no attribute 'set_weights' + col = mdf["income"] + col.set_weights(np.array([4.0, 5.0, 6.0])) # Would fail before fix + + # Verify the new weights took effect + assert np.allclose(col.weights, [4.0, 5.0, 6.0]) + + +class TestGroupByWithPandas3: + """Test groupby operations with pandas 3.""" + + def test_microseries_groupby_preserves_weights(self): + """GroupBy operations should preserve weights.""" + ms = MicroSeries([1, 2, 3, 4], weights=np.array([1.0, 2.0, 3.0, 4.0])) + groups = pd.Series(["a", "a", "b", "b"]) + + gb = ms.groupby(groups) + # Should be able to call weighted operations + result = gb.sum() + # Group a: 1*1 + 2*2 = 5 + # Group b: 3*3 + 4*4 = 25 + assert result["a"] == 5 + assert result["b"] == 25 + + def test_microdataframe_groupby_preserves_weights(self): + """MicroDataFrame groupby should preserve weights on columns.""" + mdf = MicroDataFrame( + {"group": ["a", "a", "b", "b"], "value": [1, 2, 3, 4]}, + weights=np.array([1.0, 2.0, 3.0, 4.0]), + ) + + gb = mdf.groupby("group") + result = gb.sum() + + # Check that weighted sum was computed + assert "value" in result.columns diff --git a/build/lib/microdf/tests/test_quantile_missing_values.py b/build/lib/microdf/tests/test_quantile_missing_values.py new file mode 100644 index 00000000..94e5677e --- /dev/null +++ b/build/lib/microdf/tests/test_quantile_missing_values.py @@ -0,0 +1,134 @@ +import microdf as mdf +import numpy as np +import pandas as pd +import pytest + + +def test_quantile_skips_nan(): + """NaN weight must not inflate the cumulative distribution. + + Dropping a NaN row should give the same answer as never having had + it: the inverse-CDF quantile of [1, nan, 3] equals that of [1, 3]. + """ + with_nan = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[1, 1, 1]) + without_nan = mdf.MicroSeries([1.0, 3.0], weights=[1, 1]) + assert with_nan.median() == without_nan.median() + assert with_nan.quantile(0.5) == without_nan.quantile(0.5) + + q = [0.25, 0.5, 0.75] + np.testing.assert_array_equal( + mdf.MicroSeries([1.0, np.nan, 3.0, 5.0], weights=[1, 1, 1, 1]).quantile(q), + mdf.MicroSeries([1.0, 3.0, 5.0], weights=[1, 1, 1]).quantile(q), + ) + + +def test_quantile_skipna_false_propagates_nan(): + """Skipna=False returns NaN when any value is NaN, like mean/var.""" + s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[1, 1, 1]) + assert np.isnan(s.quantile(0.5, skipna=False)) + assert np.isnan(s.median(skipna=False)) + assert s.quantile([0.25, 0.75], skipna=False).isna().all() + + +def test_quantile_all_nan_returns_nan(): + s = mdf.MicroSeries([np.nan, np.nan], weights=[1, 1]) + assert np.isnan(s.median()) + + +@pytest.mark.parametrize("skipna", [True, False]) +@pytest.mark.parametrize("q", [-0.1, 1.1, [0.5, 1.1]]) +def test_quantile_validates_bounds_with_missing_values(q, skipna): + series = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[1, 7, 1]) + with pytest.raises(AssertionError, match="quantiles should be in"): + series.quantile(q, skipna=skipna) + + +@pytest.mark.parametrize("skipna", [True, False]) +@pytest.mark.parametrize("multiple_keys", [False, True]) +def test_grouped_quantiles_preserve_missing_groups(skipna, multiple_keys): + series = mdf.MicroSeries( + [1.0, np.nan, 3.0, 5.0, np.nan, np.nan], weights=[1, 1, 1, 1, 1, 1] + ) + groups = ["a", "a", "b", "b", "c", "c"] + keys = [groups, [1, 1, 2, 2, 3, 3]] if multiple_keys else groups + grouped = series.groupby(keys) + quantiles = [0.25, 0.75] + result = grouped.quantile(quantiles, skipna=skipna) + # Scalar calls retain every group. Vector calls must retain the same + # groups, including the partial-NaN and all-NaN groups. + for quantile in quantiles: + pd.testing.assert_series_equal( + result.xs(quantile, level=-1), + grouped.quantile(quantile, skipna=skipna), + ) + first_group = 1.0 if skipna else np.nan + np.testing.assert_allclose( + result.to_numpy(), + [first_group, first_group, 3.0, 5.0, np.nan, np.nan], + equal_nan=True, + ) + + +def test_grouped_quantiles_preserve_repeated_requests(): + series = mdf.MicroSeries([1.0, np.nan, 3.0, 5.0], weights=[1, 1, 1, 1]) + result = series.groupby(["a", "a", "b", "b"]).quantile([0.5, 0.5], skipna=False) + assert result.index.tolist() == [("a", 0.5), ("a", 0.5), ("b", 0.5), ("b", 0.5)] + np.testing.assert_allclose( + result.to_numpy(), [np.nan, np.nan, 3.0, 3.0], equal_nan=True + ) + + +@pytest.mark.parametrize("quantiles", [[0.75, 0.25], [0.5, 0.5], []]) +@pytest.mark.parametrize("skipna", [True, False]) +@pytest.mark.parametrize("sort", [True, False]) +def test_grouped_quantiles_preserve_missing_multiple_keys(quantiles, skipna, sort): + """Missing group keys survive alongside missing values and repeated q.""" + frame = mdf.MicroDataFrame( + { + "region": ["north", "north", None, "south", "south"], + "year": [2024, 2024, 2024, np.nan, 2025], + "income": [10.0, np.nan, 20.0, 30.0, 40.0], + }, + weights=[1, 4, 2, 3, 1], + ) + grouped = frame.groupby(["region", "year"], dropna=False, sort=sort)["income"] + result = grouped.quantile(quantiles, skipna=skipna) + + # Each retained group has one nonmissing value. With skipna=False, + # the north group is NaN because it also contains a missing value. + north = 10.0 if skipna else np.nan + groups = [("north", 2024.0, north)] + if sort: + groups += [ + ("south", 2025.0, 40.0), + ("south", np.nan, 30.0), + (np.nan, 2024.0, 20.0), + ] + else: + groups += [ + (np.nan, 2024.0, 20.0), + ("south", np.nan, 30.0), + ("south", 2025.0, 40.0), + ] + expected_index = pd.MultiIndex.from_tuples( + [(region, year, q) for region, year, _ in groups for q in quantiles], + names=["region", "year", None], + ) + expected_values = [value for _, _, value in groups for _ in quantiles] + assert result.index.names == expected_index.names + if quantiles: + for level in range(3): + pd.testing.assert_index_equal( + result.index.get_level_values(level), + expected_index.get_level_values(level), + ) + assert result.index.nlevels == 3 + np.testing.assert_allclose(result.to_numpy(), expected_values, equal_nan=True) + for q in set(quantiles): + if quantiles.count(q) == 1: + selected = result.xs(q, level=-1) + scalar = grouped.quantile(q, skipna=skipna) + assert selected.index.equals(scalar.index) + np.testing.assert_allclose( + selected.to_numpy(), scalar.to_numpy(), equal_nan=True + ) diff --git a/build/lib/microdf/tests/test_replication.py b/build/lib/microdf/tests/test_replication.py new file mode 100644 index 00000000..28b5bbb0 --- /dev/null +++ b/build/lib/microdf/tests/test_replication.py @@ -0,0 +1,110 @@ +"""Variance estimation from replicate weights. + +Recomputing a statistic once per replicate weight vector and measuring the +spread gives a variance estimate for statistics whose analytic variance is +awkward, such as the Gini coefficient or a quantile. +""" + +import numpy as np +import pytest + +from microdf import MicroSeries, replicate_standard_error, replicate_variance + + +@pytest.fixture +def series_and_replicates(): + rng = np.random.default_rng(0) + values = rng.lognormal(mean=10, sigma=1.0, size=500) + weights = np.full(500, 40.0) + replicates = weights[:, None] * rng.poisson(1.0, size=(500, 200)) + return MicroSeries(values, weights=weights), replicates + + +def test_matches_the_analytic_standard_error_of_a_weighted_mean( + series_and_replicates, +): + """The mean has a closed form, so it is the case we can check exactly.""" + series, replicates = series_and_replicates + + replicate_se = replicate_standard_error( + series, lambda s: s.mean(), replicates, method="bootstrap" + ) + + values = np.asarray(series) + analytic_se = np.sqrt(np.var(values, ddof=1) / len(values)) + + assert replicate_se == pytest.approx(analytic_se, rel=0.15) + + +def test_works_for_statistics_with_no_analytic_variance( + series_and_replicates, +): + """The point of the method: Gini and quantiles come out like anything else.""" + series, replicates = series_and_replicates + + for statistic in (lambda s: s.gini(), lambda s: s.median()): + se = replicate_standard_error( + series, statistic, replicates, method="bootstrap" + ) + assert se > 0 + assert np.isfinite(se) + + +@pytest.mark.parametrize( + "method,expected_factor", + [ + ("jackknife", 199 / 200), + ("brr", 1 / 200), + ("bootstrap", 1 / 200), + ("successive-difference", 4 / 200), + ], +) +def test_each_method_applies_its_own_scale( + series_and_replicates, method, expected_factor +): + """The scale factor is what distinguishes the replication schemes.""" + series, replicates = series_and_replicates + + variance = replicate_variance( + series, lambda s: s.mean(), replicates, method=method + ) + reference = replicate_variance( + series, lambda s: s.mean(), replicates, method="brr" + ) + + assert variance == pytest.approx(reference * expected_factor * 200, rel=1e-9) + + +def test_fay_requires_and_uses_its_constant(series_and_replicates): + series, replicates = series_and_replicates + + with pytest.raises(ValueError, match="requires fay_k"): + replicate_variance(series, lambda s: s.mean(), replicates, method="fay") + + fay = replicate_variance( + series, lambda s: s.mean(), replicates, method="fay", fay_k=0.5 + ) + brr = replicate_variance( + series, lambda s: s.mean(), replicates, method="brr" + ) + + # 1 / (R (1 - k)^2) against 1 / R, so a factor of four at k = 0.5. + assert fay == pytest.approx(brr * 4, rel=1e-9) + + +def test_rejects_input_that_cannot_be_right(series_and_replicates): + series, replicates = series_and_replicates + + with pytest.raises(ValueError, match="2-dimensional"): + replicate_variance(series, lambda s: s.mean(), np.ones(500)) + + with pytest.raises(ValueError, match="rows but the series has"): + replicate_variance(series, lambda s: s.mean(), np.ones((499, 10))) + + with pytest.raises(ValueError, match="At least two"): + replicate_variance(series, lambda s: s.mean(), np.ones((500, 1))) + + with pytest.raises(ValueError, match="Unknown method"): + replicate_variance( + series, lambda s: s.mean(), replicates, method="nonsense" + ) diff --git a/build/lib/microdf/tests/test_serialization.py b/build/lib/microdf/tests/test_serialization.py new file mode 100644 index 00000000..59d4bfdc --- /dev/null +++ b/build/lib/microdf/tests/test_serialization.py @@ -0,0 +1,143 @@ +import copy +import io +import pickle + +import pandas as pd +import pytest + +import microdf as mdf + + +def test_microseries_survives_pickling(): + """Weights must survive a pickle round-trip.""" + import pickle + + s = mdf.MicroSeries([1, 2, 3], index=[7, 8, 9], weights=[1, 2, 3]) + restored = pickle.loads(pickle.dumps(s)) + assert isinstance(restored, mdf.MicroSeries) + assert restored.sum() == 14 + assert list(restored.weights) == [1.0, 2.0, 3.0] + + +def test_microdataframe_survives_pickling(): + """Weights and the weighted aggregations must survive a round-trip.""" + import pickle + + df = mdf.MicroDataFrame( + pd.DataFrame({"x": [1, 2, 3]}, index=[7, 8, 9]), weights=[1, 2, 3] + ) + restored = pickle.loads(pickle.dumps(df)) + assert isinstance(restored, mdf.MicroDataFrame) + assert isinstance(restored.weights, pd.Series) + # Would be 6 (unweighted) if the aggregation overrides were not + # reinstalled after unpickling. + assert restored.sum()["x"] == 14 + + +def test_deepcopy_preserves_weights(): + df = mdf.MicroDataFrame(pd.DataFrame({"x": [1, 2, 3]}), weights=[1, 2, 3]) + assert copy.deepcopy(df).sum()["x"] == 14 + s = mdf.MicroSeries([1, 2, 3], weights=[1, 2, 3]) + assert copy.deepcopy(s).sum() == 14 + + +@pytest.mark.parametrize("use_weight_column", [False, True]) +@pytest.mark.parametrize("use_pandas_pickle", [False, True]) +def test_serialization_preserves_weight_column_state( + use_weight_column, use_pandas_pickle +): + """Replacing restored weights can preserve the original weight column.""" + frame = mdf.MicroDataFrame( + pd.DataFrame({"x": [1, 2, 3], "w": [1, 2, 3]}, index=[7, 8, 9]), + weights="w" if use_weight_column else [1, 2, 3], + ) + if use_pandas_pickle: + buffer = io.BytesIO() + frame.to_pickle(buffer) + buffer.seek(0) + restored = pd.read_pickle(buffer) + else: + restored = pickle.loads(pickle.dumps(frame)) + + assert restored.weights_col == ("w" if use_weight_column else None) + restored.set_weights([3, 2, 1], preserve_old=True) + assert restored.sum()["x"] == 10 + assert restored.index.equals(frame.index) + if use_weight_column: + assert restored["old_w"].tolist() == [1, 2, 3] + else: + assert "old_w" not in restored.columns + + +@pytest.mark.parametrize("operation", ["pickle", "pandas_pickle", "deepcopy"]) +def test_named_microseries_preserves_name_and_weights(operation): + """Serialization retains pandas metadata as well as survey weights.""" + series = mdf.MicroSeries( + [1, 2, 3], index=[7, 8, 9], name="group", weights=[1, 2, 3] + ) + if operation == "deepcopy": + restored = copy.deepcopy(series) + elif operation == "pandas_pickle": + buffer = io.BytesIO() + series.to_pickle(buffer) + buffer.seek(0) + restored = pd.read_pickle(buffer) + else: + restored = pickle.loads(pickle.dumps(series)) + + assert restored.name == "group" + assert restored.index.equals(series.index) + pd.testing.assert_series_equal(restored.weights, series.weights) + assert restored.sum() == 14 + + +@pytest.mark.parametrize("selected", [False, True]) +def test_grouped_aggregation_retains_index_name(selected): + """Copying internal grouped weights must retain the grouping label.""" + frame = mdf.MicroDataFrame( + {"group": ["a", "a", "b", "b"], "value": [1, 2, 3, 4]}, + weights=[1, 2, 3, 4], + ) + grouped = frame.groupby("group") + if selected: + grouped = grouped[["value"]] + expected = pd.DataFrame( + {"value": [5.0, 25.0]}, index=pd.Index(["a", "b"], name="group") + ) + pd.testing.assert_frame_equal(grouped.sum(), expected) + + +@pytest.mark.parametrize("kind", ["frame_columns", "frame_index", "series_index"]) +def test_renamed_weights_are_independent(kind): + """Pandas finalization must retain the renamed result's copied weights.""" + if kind == "series_index": + original = mdf.MicroSeries( + [10, 20], index=[7, 8], name="income", weights=[1, 2] + ) + renamed = original.rename(index={7: 70}) + assert renamed.name == "income" + else: + original = mdf.MicroDataFrame( + {"x": [10, 20], "w": [1, 2]}, index=[7, 8], weights="w" + ) + renamed = ( + original.rename(columns={"x": "income"}) + if kind == "frame_columns" + else original.rename(index={7: 70}) + ) + assert renamed.weights_col == "w" + + renamed.weights.iloc[0] = 100 + pd.testing.assert_series_equal( + original.weights, pd.Series([1.0, 2.0], index=[7, 8]) + ) + original_total = original.sum() if kind == "series_index" else original.sum()["x"] + assert original_total == 10 * 1 + 20 * 2 + + original.weights.iloc[1] = 9 + assert renamed.weights.iloc[1] == 2 + if kind == "frame_columns": + # The renamed frame must still use weighted aggregation after pickle. + restored = pickle.loads(pickle.dumps(renamed)) + assert restored.weights_col == "w" + assert restored.sum()["income"] == 10 * 100 + 20 * 2 diff --git a/build/lib/microdf/tests/test_version_metadata.py b/build/lib/microdf/tests/test_version_metadata.py new file mode 100644 index 00000000..5bd1c874 --- /dev/null +++ b/build/lib/microdf/tests/test_version_metadata.py @@ -0,0 +1,8 @@ +import microdf as mdf + + +def test_version_matches_package_metadata(): + """__version__ must not drift from pyproject.toml.""" + from importlib.metadata import version + + assert mdf.__version__ == version("microdf-python") diff --git a/changelog.d/replicate-weight-variance.added.md b/changelog.d/replicate-weight-variance.added.md new file mode 100644 index 00000000..425cb938 --- /dev/null +++ b/changelog.d/replicate-weight-variance.added.md @@ -0,0 +1 @@ +- Added variance and standard error estimation from replicate weights, supporting jackknife, BRR, Fay's BRR, bootstrap and successive-difference schemes, for any statistic in the package. diff --git a/microdf/__init__.py b/microdf/__init__.py index 95133a63..2e2d83a6 100644 --- a/microdf/__init__.py +++ b/microdf/__init__.py @@ -2,6 +2,7 @@ from .microdataframe import MicroDataFrame, MicroDataFrameGroupBy from .microseries import MicroSeries, MicroSeriesGroupBy +from .replication import replicate_standard_error, replicate_variance name = "microdf" @@ -19,4 +20,7 @@ # microdataframe.py "MicroDataFrame", "MicroDataFrameGroupBy", + # replication.py + "replicate_variance", + "replicate_standard_error", ] diff --git a/microdf/microseries.py b/microdf/microseries.py index 2e2f6df3..1185d2e6 100644 --- a/microdf/microseries.py +++ b/microdf/microseries.py @@ -579,6 +579,37 @@ def median(self, skipna: bool = True) -> float: """ return self.quantile(0.5, skipna=skipna) + def replicate_standard_error( + self, + statistic: Callable, + replicate_weights, + method: str = "jackknife", + fay_k: Optional[float] = None, + ) -> float: + """Standard error of ``statistic`` from a set of replicate weights. + + Recomputes the statistic once per replicate and scales the spread by + the factor appropriate to how the replicates were built, so it works + for any statistic including the Gini coefficient and quantiles. + + Only valid for replicate weights as published with a survey. Weights + calibrated to external targets no longer correspond to the original + replication scheme. + + :param statistic: Callable taking a MicroSeries, e.g. + ``lambda s: s.median()``. + :param replicate_weights: Array or frame of shape ``(len(self), R)``. + :param method: ``jackknife``, ``brr``, ``bootstrap``, + ``successive-difference`` or ``fay``. + :param fay_k: Fay's perturbation constant, for ``method="fay"``. + :returns: The estimated standard error. + """ + from microdf.replication import replicate_standard_error + + return replicate_standard_error( + self, statistic, replicate_weights, method, fay_k + ) + @scalar_function def gini(self, negatives: Optional[str] = None) -> float: """Calculates Gini index. diff --git a/microdf/replication.py b/microdf/replication.py new file mode 100644 index 00000000..0bb128e0 --- /dev/null +++ b/microdf/replication.py @@ -0,0 +1,123 @@ +"""Variance estimation from replicate weights. + +Many survey products publish a set of replicate weight vectors alongside the +main weight. Recomputing a statistic once per replicate and measuring the +spread gives a variance estimate that requires no analytic formula, which is +what makes it usable for statistics such as the Gini coefficient or a +quantile where the analytic variance is awkward. + +The scale factor depends on how the replicates were constructed, so the +method must be named rather than guessed. + +Note that this is only valid for replicate weights as published with a +survey. Weights that have been calibrated or reweighted to external targets +no longer correspond to the original replication scheme, and applying these +estimators to them does not describe the variance of the resulting estimator. +""" + +from typing import Callable, Optional, Union + +import numpy as np +import pandas as pd + +# Scale applied to the sum of squared deviations from the full-sample +# estimate. R is the number of replicates. +METHOD_FACTORS = { + # Delete-a-group jackknife: (R - 1) / R. + "jackknife": lambda r: (r - 1) / r, + # Balanced repeated replication: 1 / R. + "brr": lambda r: 1 / r, + # Bootstrap replicates: 1 / R. + "bootstrap": lambda r: 1 / r, + # Successive difference replication, as used for the ACS and CPS: 4 / R. + "successive-difference": lambda r: 4 / r, +} + + +def _fay_factor(r: int, fay_k: float) -> float: + """Scale for Fay's variant of BRR, which perturbs rather than deletes.""" + if not 0 <= fay_k < 1: + raise ValueError(f"fay_k must be in [0, 1), got {fay_k}") + return 1 / (r * (1 - fay_k) ** 2) + + +def replicate_variance( + series, + statistic: Callable, + replicate_weights: Union[np.ndarray, pd.DataFrame], + method: str = "jackknife", + fay_k: Optional[float] = None, +) -> float: + """Variance of ``statistic`` estimated from replicate weights. + + :param series: A MicroSeries. Its own weights give the point estimate. + :param statistic: Callable taking a MicroSeries and returning a float, + for example ``lambda s: s.gini()``. + :param replicate_weights: Array or frame of shape ``(len(series), R)``. + :param method: One of ``jackknife``, ``brr``, ``bootstrap``, + ``successive-difference``, or ``fay`` (which requires ``fay_k``). + :param fay_k: Fay's perturbation constant, required when + ``method="fay"``. + :returns: The estimated variance of the statistic. + """ + from microdf.microseries import MicroSeries + + weights = np.asarray(replicate_weights, dtype=float) + if weights.ndim != 2: + raise ValueError( + f"replicate_weights must be 2-dimensional, got shape {weights.shape}" + ) + if weights.shape[0] != len(series): + raise ValueError( + f"replicate_weights has {weights.shape[0]} rows but the series has " + f"{len(series)}" + ) + + n_replicates = weights.shape[1] + if n_replicates < 2: + raise ValueError("At least two replicate weights are required") + + if method == "fay": + if fay_k is None: + raise ValueError("method='fay' requires fay_k") + factor = _fay_factor(n_replicates, fay_k) + elif method in METHOD_FACTORS: + if fay_k is not None: + raise ValueError("fay_k applies only to method='fay'") + factor = METHOD_FACTORS[method](n_replicates) + else: + known = ", ".join(sorted([*METHOD_FACTORS, "fay"])) + raise ValueError(f"Unknown method {method!r}; expected one of {known}") + + point = float(statistic(series)) + values = np.asarray(series, dtype=float) + index = series.index + + deviations = [] + for column in range(n_replicates): + replicate = MicroSeries( + values, weights=weights[:, column], index=index + ) + deviations.append(float(statistic(replicate)) - point) + + return factor * float(np.sum(np.square(deviations))) + + +def replicate_standard_error( + series, + statistic: Callable, + replicate_weights: Union[np.ndarray, pd.DataFrame], + method: str = "jackknife", + fay_k: Optional[float] = None, +) -> float: + """Standard error of ``statistic``, the square root of its variance. + + Takes the same arguments as :func:`replicate_variance`. + """ + return float( + np.sqrt( + replicate_variance( + series, statistic, replicate_weights, method, fay_k + ) + ) + ) diff --git a/microdf/tests/test_replication.py b/microdf/tests/test_replication.py new file mode 100644 index 00000000..28b5bbb0 --- /dev/null +++ b/microdf/tests/test_replication.py @@ -0,0 +1,110 @@ +"""Variance estimation from replicate weights. + +Recomputing a statistic once per replicate weight vector and measuring the +spread gives a variance estimate for statistics whose analytic variance is +awkward, such as the Gini coefficient or a quantile. +""" + +import numpy as np +import pytest + +from microdf import MicroSeries, replicate_standard_error, replicate_variance + + +@pytest.fixture +def series_and_replicates(): + rng = np.random.default_rng(0) + values = rng.lognormal(mean=10, sigma=1.0, size=500) + weights = np.full(500, 40.0) + replicates = weights[:, None] * rng.poisson(1.0, size=(500, 200)) + return MicroSeries(values, weights=weights), replicates + + +def test_matches_the_analytic_standard_error_of_a_weighted_mean( + series_and_replicates, +): + """The mean has a closed form, so it is the case we can check exactly.""" + series, replicates = series_and_replicates + + replicate_se = replicate_standard_error( + series, lambda s: s.mean(), replicates, method="bootstrap" + ) + + values = np.asarray(series) + analytic_se = np.sqrt(np.var(values, ddof=1) / len(values)) + + assert replicate_se == pytest.approx(analytic_se, rel=0.15) + + +def test_works_for_statistics_with_no_analytic_variance( + series_and_replicates, +): + """The point of the method: Gini and quantiles come out like anything else.""" + series, replicates = series_and_replicates + + for statistic in (lambda s: s.gini(), lambda s: s.median()): + se = replicate_standard_error( + series, statistic, replicates, method="bootstrap" + ) + assert se > 0 + assert np.isfinite(se) + + +@pytest.mark.parametrize( + "method,expected_factor", + [ + ("jackknife", 199 / 200), + ("brr", 1 / 200), + ("bootstrap", 1 / 200), + ("successive-difference", 4 / 200), + ], +) +def test_each_method_applies_its_own_scale( + series_and_replicates, method, expected_factor +): + """The scale factor is what distinguishes the replication schemes.""" + series, replicates = series_and_replicates + + variance = replicate_variance( + series, lambda s: s.mean(), replicates, method=method + ) + reference = replicate_variance( + series, lambda s: s.mean(), replicates, method="brr" + ) + + assert variance == pytest.approx(reference * expected_factor * 200, rel=1e-9) + + +def test_fay_requires_and_uses_its_constant(series_and_replicates): + series, replicates = series_and_replicates + + with pytest.raises(ValueError, match="requires fay_k"): + replicate_variance(series, lambda s: s.mean(), replicates, method="fay") + + fay = replicate_variance( + series, lambda s: s.mean(), replicates, method="fay", fay_k=0.5 + ) + brr = replicate_variance( + series, lambda s: s.mean(), replicates, method="brr" + ) + + # 1 / (R (1 - k)^2) against 1 / R, so a factor of four at k = 0.5. + assert fay == pytest.approx(brr * 4, rel=1e-9) + + +def test_rejects_input_that_cannot_be_right(series_and_replicates): + series, replicates = series_and_replicates + + with pytest.raises(ValueError, match="2-dimensional"): + replicate_variance(series, lambda s: s.mean(), np.ones(500)) + + with pytest.raises(ValueError, match="rows but the series has"): + replicate_variance(series, lambda s: s.mean(), np.ones((499, 10))) + + with pytest.raises(ValueError, match="At least two"): + replicate_variance(series, lambda s: s.mean(), np.ones((500, 1))) + + with pytest.raises(ValueError, match="Unknown method"): + replicate_variance( + series, lambda s: s.mean(), replicates, method="nonsense" + ) diff --git a/uv.lock b/uv.lock index 0911adbd..a5288ea3 100644 --- a/uv.lock +++ b/uv.lock @@ -99,7 +99,7 @@ resolution-markers = [ "python_full_version < '3.10'", ] dependencies = [ - { name = "colorama", marker = "python_full_version < '3.10' and sys_platform == 'win32'" }, + { name = "colorama", marker = "sys_platform == 'win32'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/b9/2e/0090cbf739cee7d23781ad4b89a9894a41538e4fcf4c31dcdd705b78eb8b/click-8.1.8.tar.gz", hash = "sha256:ed53c9d8990d83c2a27deae68e4ee337473f6330c040a31d4225c9574d16096a", size = 226593, upload-time = "2024-12-21T18:38:44.339Z" } wheels = [ @@ -116,7 +116,7 @@ resolution-markers = [ "python_full_version == '3.10.*'", ] dependencies = [ - { name = "colorama", marker = "python_full_version >= '3.10' and sys_platform == 'win32'" }, + { name = "colorama", marker = "sys_platform == 'win32'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/60/6c/8ca2efa64cf75a977a0d7fac081354553ebe483345c734fb6b6515d96bbc/click-8.2.1.tar.gz", hash = "sha256:27c491cc05d968d271d5a1db13e3b5a184636d9d930f148c50b038f0d0646202", size = 286342, upload-time = "2025-05-20T23:19:49.832Z" } wheels = [ @@ -243,7 +243,7 @@ name = "exceptiongroup" version = "1.3.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "typing-extensions", marker = "python_full_version < '3.11'" }, + { name = "typing-extensions" }, ] sdist = { url = "https://files.pythonhosted.org/packages/0b/9f/a65090624ecf468cdca03533906e7c69ed7588582240cfe7cc9e770b50eb/exceptiongroup-1.3.0.tar.gz", hash = "sha256:b241f5885f560bc56a59ee63ca4c6a8bfa46ae4ad651af316d4e81817bb9fd88", size = 29749, upload-time = "2025-05-10T17:42:51.123Z" } wheels = [ @@ -264,7 +264,7 @@ name = "importlib-metadata" version = "8.7.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "zipp", marker = "python_full_version < '3.10'" }, + { name = "zipp" }, ] sdist = { url = "https://files.pythonhosted.org/packages/76/66/650a33bd90f786193e4de4b3ad86ea60b53c89b669a5c7be931fac31cdb0/importlib_metadata-8.7.0.tar.gz", hash = "sha256:d13b81ad223b890aa16c5471f2ac3056cf76c5f10f82d6f9292f0b415f389000", size = 56641, upload-time = "2025-04-27T15:29:01.736Z" } wheels = [ @@ -276,7 +276,7 @@ name = "importlib-resources" version = "6.5.2" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "zipp", marker = "python_full_version < '3.10'" }, + { name = "zipp" }, ] sdist = { url = "https://files.pythonhosted.org/packages/cf/8c/f834fbf984f691b4f7ff60f50b514cc3de5cc08abfc3295564dd89c5e2e7/importlib_resources-6.5.2.tar.gz", hash = "sha256:185f87adef5bcc288449d98fb4fba07cea78bc036455dd44c5fc4a2fe78fed2c", size = 44693, upload-time = "2025-01-03T18:51:56.698Z" } wheels = [ @@ -402,7 +402,7 @@ wheels = [ [[package]] name = "microdf-python" -version = "1.3.1" +version = "1.3.8" source = { editable = "." } dependencies = [ { name = "numpy", version = "2.0.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.10'" }, @@ -777,10 +777,10 @@ resolution-markers = [ "python_full_version == '3.10.*'", ] dependencies = [ - { name = "certifi", marker = "python_full_version >= '3.10'" }, - { name = "charset-normalizer", marker = "python_full_version >= '3.10'" }, - { name = "idna", marker = "python_full_version >= '3.10'" }, - { name = "urllib3", marker = "python_full_version >= '3.10'" }, + { name = "certifi" }, + { name = "charset-normalizer" }, + { name = "idna" }, + { name = "urllib3" }, ] sdist = { url = "https://files.pythonhosted.org/packages/e1/0a/929373653770d8a0d7ea76c37de6e41f11eb07559b103b1c02cafb3f7cf8/requests-2.32.4.tar.gz", hash = "sha256:27d0316682c8a29834d3264820024b62a36942083d52caf2f14c0591336d3422", size = 135258, upload-time = "2025-06-09T16:43:07.34Z" } wheels = [ From 541ec5045ad7331e2fbcec9e2aa5828b623fc4cb Mon Sep 17 00:00:00 2001 From: Vahid Ahmadi Date: Wed, 16 Sep 2026 11:01:44 +0100 Subject: [PATCH 2/7] Use modern type annotation syntax in the new module Co-Authored-By: Claude Opus 5 (1M context) --- microdf/replication.py | 22 +++++++++------------- 1 file changed, 9 insertions(+), 13 deletions(-) diff --git a/microdf/replication.py b/microdf/replication.py index 0bb128e0..56f72abb 100644 --- a/microdf/replication.py +++ b/microdf/replication.py @@ -15,7 +15,9 @@ estimators to them does not describe the variance of the resulting estimator. """ -from typing import Callable, Optional, Union +from __future__ import annotations + +from typing import Callable import numpy as np import pandas as pd @@ -44,9 +46,9 @@ def _fay_factor(r: int, fay_k: float) -> float: def replicate_variance( series, statistic: Callable, - replicate_weights: Union[np.ndarray, pd.DataFrame], + replicate_weights: np.ndarray | pd.DataFrame, method: str = "jackknife", - fay_k: Optional[float] = None, + fay_k: float | None = None, ) -> float: """Variance of ``statistic`` estimated from replicate weights. @@ -95,9 +97,7 @@ def replicate_variance( deviations = [] for column in range(n_replicates): - replicate = MicroSeries( - values, weights=weights[:, column], index=index - ) + replicate = MicroSeries(values, weights=weights[:, column], index=index) deviations.append(float(statistic(replicate)) - point) return factor * float(np.sum(np.square(deviations))) @@ -106,18 +106,14 @@ def replicate_variance( def replicate_standard_error( series, statistic: Callable, - replicate_weights: Union[np.ndarray, pd.DataFrame], + replicate_weights: np.ndarray | pd.DataFrame, method: str = "jackknife", - fay_k: Optional[float] = None, + fay_k: float | None = None, ) -> float: """Standard error of ``statistic``, the square root of its variance. Takes the same arguments as :func:`replicate_variance`. """ return float( - np.sqrt( - replicate_variance( - series, statistic, replicate_weights, method, fay_k - ) - ) + np.sqrt(replicate_variance(series, statistic, replicate_weights, method, fay_k)) ) From d811bca0433a4e63e94ddd525785e47afa733b34 Mon Sep 17 00:00:00 2001 From: Vahid Ahmadi Date: Wed, 16 Sep 2026 11:04:41 +0100 Subject: [PATCH 3/7] Keep build artefacts out of the repository Co-Authored-By: Claude Opus 5 (1M context) --- .gitignore | 5 +- build/lib/microdf/__init__.py | 26 - build/lib/microdf/microdataframe.py | 1021 ----------------- build/lib/microdf/microseries.py | 983 ---------------- build/lib/microdf/replication.py | 123 -- build/lib/microdf/tests/conftest.py | 9 - .../microdf/tests/test_aggregation_errors.py | 57 - .../tests/test_dataframe_weight_storage.py | 61 - .../tests/test_microseries_dataframe.py | 817 ------------- .../tests/test_nullify_weights_index.py | 10 - .../tests/test_pandas3_compatibility.py | 242 ---- .../tests/test_quantile_missing_values.py | 134 --- build/lib/microdf/tests/test_replication.py | 110 -- build/lib/microdf/tests/test_serialization.py | 143 --- .../microdf/tests/test_version_metadata.py | 8 - 15 files changed, 4 insertions(+), 3745 deletions(-) delete mode 100644 build/lib/microdf/__init__.py delete mode 100644 build/lib/microdf/microdataframe.py delete mode 100644 build/lib/microdf/microseries.py delete mode 100644 build/lib/microdf/replication.py delete mode 100644 build/lib/microdf/tests/conftest.py delete mode 100644 build/lib/microdf/tests/test_aggregation_errors.py delete mode 100644 build/lib/microdf/tests/test_dataframe_weight_storage.py delete mode 100644 build/lib/microdf/tests/test_microseries_dataframe.py delete mode 100644 build/lib/microdf/tests/test_nullify_weights_index.py delete mode 100644 build/lib/microdf/tests/test_pandas3_compatibility.py delete mode 100644 build/lib/microdf/tests/test_quantile_missing_values.py delete mode 100644 build/lib/microdf/tests/test_replication.py delete mode 100644 build/lib/microdf/tests/test_serialization.py delete mode 100644 build/lib/microdf/tests/test_version_metadata.py diff --git a/.gitignore b/.gitignore index 475d3e68..f14b3254 100644 --- a/.gitignore +++ b/.gitignore @@ -16,4 +16,7 @@ docs/_build # Codecov coverage reports *.xml -.coverage \ No newline at end of file +.coverage +# Build artefacts +build/ +*.egg-info/ diff --git a/build/lib/microdf/__init__.py b/build/lib/microdf/__init__.py deleted file mode 100644 index 2e2d83a6..00000000 --- a/build/lib/microdf/__init__.py +++ /dev/null @@ -1,26 +0,0 @@ -from importlib.metadata import PackageNotFoundError, version - -from .microdataframe import MicroDataFrame, MicroDataFrameGroupBy -from .microseries import MicroSeries, MicroSeriesGroupBy -from .replication import replicate_standard_error, replicate_variance - -name = "microdf" - -# Read the version from package metadata so it can't drift from -# pyproject.toml (the automated bump only touches pyproject). -try: - __version__ = version("microdf-python") -except PackageNotFoundError: # pragma: no cover - running from a source tree - __version__ = "unknown" - -__all__ = [ - # microseries.py - "MicroSeries", - "MicroSeriesGroupBy", - # microdataframe.py - "MicroDataFrame", - "MicroDataFrameGroupBy", - # replication.py - "replicate_variance", - "replicate_standard_error", -] diff --git a/build/lib/microdf/microdataframe.py b/build/lib/microdf/microdataframe.py deleted file mode 100644 index 707c3bd7..00000000 --- a/build/lib/microdf/microdataframe.py +++ /dev/null @@ -1,1021 +0,0 @@ -import copy -import logging -import warnings -from functools import wraps -from typing import Callable, List, Optional, Union - -import numpy as np -import pandas as pd - -from microdf.microseries import MicroSeries, MicroSeriesGroupBy - -logger = logging.getLogger(__name__) - - -class _MicroLocIndexer: - """Custom loc indexer that returns MicroDataFrame with proper weights.""" - - def __init__(self, mdf: "MicroDataFrame"): - self._mdf = mdf - # Get the parent's loc indexer - self._parent_loc = pd.DataFrame.loc.fget(mdf) - - def __getitem__(self, key): - # Use the parent DataFrame's loc indexer - result = self._parent_loc[key] - - if isinstance(result, pd.DataFrame): - # Get the filtered weights based on the result's index - new_weights = self._mdf.weights.reindex(result.index) - return MicroDataFrame(result, weights=new_weights) - elif isinstance(result, pd.Series): - # Single row or column selected - if result.name in self._mdf.columns: - # Column was selected - return MicroSeries with all weights - return MicroSeries(result, weights=self._mdf.weights) - else: - # Row was selected - return as-is (scalar values for each col) - return result - else: - # Scalar value - return result - - def __setitem__(self, key, value): - self._parent_loc[key] = value - self._mdf._link_all_weights() - - def __getattr__(self, name): - """Delegate unknown attributes to the parent loc indexer.""" - return getattr(self._parent_loc, name) - - -class _MicroILocIndexer: - """Custom iloc indexer that returns MicroDataFrame with proper weights.""" - - def __init__(self, mdf: "MicroDataFrame"): - self._mdf = mdf - # Get the parent's iloc indexer - self._parent_iloc = pd.DataFrame.iloc.fget(mdf) - - def __getitem__(self, key): - # Use the parent DataFrame's iloc indexer - result = self._parent_iloc[key] - - if isinstance(result, pd.DataFrame): - # Get the filtered weights based on the result's index - new_weights = self._mdf.weights.iloc[ - self._mdf.index.get_indexer(result.index) - ] - new_weights = pd.Series(new_weights.values, index=result.index) - return MicroDataFrame(result, weights=new_weights) - elif isinstance(result, pd.Series): - # Single row or column selected - if isinstance(key, tuple) and len(key) == 2: - # df.iloc[:, col_idx] - column selection - row_key = key[0] - if isinstance(row_key, slice) and row_key == slice(None): - # All rows selected for a column - return MicroSeries(result, weights=self._mdf.weights) - # Check if this is a column (result index matches mdf index) - if result.index.equals(self._mdf.index): - return MicroSeries(result, weights=self._mdf.weights) - # Row selection - return as-is - return result - else: - # Scalar value - return result - - def __setitem__(self, key, value): - self._parent_iloc[key] = value - self._mdf._link_all_weights() - - def __getattr__(self, name): - """Delegate unknown attributes to the parent iloc indexer.""" - return getattr(self._parent_iloc, name) - - -class MicroDataFrame(pd.DataFrame): - # Declare weight state as pandas metadata. pandas includes - # _metadata attributes in the pickle state, so weights now survive - # pickling, to_pickle/read_pickle and copy.deepcopy instead of - # vanishing and leaving an AttributeError on the next aggregation. - # Retain the column name for set_weights(..., preserve_old=True). - _metadata = pd.DataFrame._metadata + ["weights", "weights_col"] - - def __init__(self, *args, weights=None, **kwargs): - """A DataFrame-inheriting class for weighted microdata. - - Weights can be provided at initialisation, or using set_weights or - set_weight_col. - - :param weights: Array of weights. - :type weights: np.array - """ - super().__init__(*args, **kwargs) - self.weights = None - self.set_weights(weights) - self._link_all_weights() - self.override_df_functions() - - def __finalize__(self, other, method=None, **kwargs) -> "MicroDataFrame": - """Retain copied weights when pandas finalizes a renamed result.""" - copied_weights = getattr(self, "weights", None) if method == "rename" else None - super().__finalize__(other, method=method, **kwargs) - if copied_weights is not None: - # rename already called copy(); metadata propagation must not - # replace those weights with the source's mutable Series. - self.weights = copied_weights - return self - - def __setstate__(self, state) -> None: - """Restore a pickled MicroDataFrame. - - The weighted aggregations are installed as per-instance closures by - ``override_df_functions``, which only runs in ``__init__`` — a path - unpickling skips. Without reinstalling them, ``mdf.sum()`` on an - unpickled frame silently fell through to the unweighted pandas - implementation. - """ - super().__setstate__(state) - if getattr(self, "weights", None) is None: - self._link_all_weights() - self.override_df_functions() - - @property - def loc(self) -> _MicroLocIndexer: - """Label-based indexer that preserves MicroDataFrame type and weights. - - :return: Custom loc indexer for MicroDataFrame - """ - return _MicroLocIndexer(self) - - @property - def iloc(self) -> _MicroILocIndexer: - """Integer-based indexer that preserves MicroDataFrame type and - weights. - - :return: Custom iloc indexer for MicroDataFrame - """ - return _MicroILocIndexer(self) - - def override_df_functions(self) -> None: - """Override DataFrame functions to work with weighted operations.""" - for name in MicroSeries.FUNCTIONS: - if name in MicroSeries.SCALAR_FUNCTIONS: - setattr(self, name, self._create_scalar_function(name)) - elif name in MicroSeries.VECTOR_FUNCTIONS: - setattr(self, name, self._create_vector_function(name)) - elif name in MicroSeries.AGNOSTIC_FUNCTIONS: - setattr(self, name, self._create_agnostic_function(name)) - - def _create_scalar_function(self, name: str) -> Callable: - """Create a scalar function that returns a Series of results. - - :param name: Name of the function to create - :return: Function that applies the operation to all columns - """ - - def fn(*args, **kwargs) -> pd.Series: - results = {} - for col in self.columns: - if pd.api.types.is_numeric_dtype(self[col]): - try: - results[col] = getattr(self[col], name)(*args, **kwargs) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - return pd.Series(results) - - return fn - - def _create_vector_function(self, name: str) -> Callable: - """Create a vector function that returns a DataFrame of results. - - :param name: Name of the function to create - :return: Function that applies the operation to all columns - """ - - def fn(*args, **kwargs) -> pd.DataFrame: - results = [] - columns = [] - for col in self.columns: - if pd.api.types.is_numeric_dtype(self[col]): - try: - result = getattr(self[col], name)(*args, **kwargs) - results.append(result) - columns.append(col) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - - if results: - df = pd.DataFrame(results) - df.index = columns - return df - else: - return pd.DataFrame() - - return fn - - def _create_agnostic_function(self, name: str) -> Callable: - """Create a function that can be either scalar or vector based on - input. - - :param name: Name of the function to create - :return: Function that applies the operation to all columns - """ - - def fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: - # Check if first argument is array-like - is_array = len(args) > 0 and hasattr(args[0], "__len__") - - if is_array: - # Use vector function behavior - results = [] - columns = [] - for col in self.columns: - if pd.api.types.is_numeric_dtype(self[col]): - try: - result = getattr(self[col], name)(*args, **kwargs) - results.append(result) - columns.append(col) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - - if results: - df = pd.DataFrame(results) - df.index = columns - return df - else: - return pd.DataFrame() - else: - # Use scalar function behavior - results = {} - for col in self.columns: - if pd.api.types.is_numeric_dtype(self[col]): - try: - results[col] = getattr(self[col], name)(*args, **kwargs) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - return pd.Series(results) - - return fn - - def get_args_as_micro_series(*kwarg_names: tuple) -> Callable: - """Decorator for auto-parsing column names into MicroSeries objects. - - If given, kwarg_names limits arguments checked to keyword arguments - specified. - - :param arg_names: argument names to restrict to. - :type arg_names: str - """ - - def arg_series_decorator(fn) -> Callable: - @wraps(fn) - def series_function( - self, *args, **kwargs - ) -> Union[pd.Series, pd.DataFrame]: - new_args = [] - new_kwargs = {} - if len(kwarg_names) == 0: - for value in args: - if isinstance(value, str): - if value not in self.columns: - raise Exception("Column not found") - new_args += [self[value]] - else: - new_args += [value] - for name, value in kwargs.items(): - if isinstance(value, str) and ( - len(kwarg_names) == 0 or name in kwarg_names - ): - if value not in self.columns: - raise Exception("Column not found") - new_kwargs[name] = self[value] - else: - new_kwargs[name] = value - return fn(self, *new_args, **new_kwargs) - - return series_function - - return arg_series_decorator - - def __setitem__(self, *args, **kwargs) -> None: - super().__setitem__(*args, **kwargs) - self._link_all_weights() - - def _link_weights(self, column) -> None: - # In pandas 3.0+, we can't modify column classes in-place due to CoW. - # Instead, we rely on __getitem__ to wrap columns as MicroSeries on - # access. This method is kept for backward compatibility but is now - # a no-op. - pass - - def _link_all_weights(self) -> None: - if self.weights is None: - if len(self) > 0: - self.set_weights(np.ones((len(self)))) - # In pandas 3.0+, columns are wrapped as MicroSeries on access via - # __getitem__, not stored as MicroSeries internally. - - def set_weights( - self, - weights: Union[np.ndarray, str], - preserve_old: Optional[bool] = False, - ) -> None: - """Sets the weights for the MicroDataFrame. - - If a string is received, it will be assumed to be the column name of - the weight column. - - :param weights: Array of weights. - :param preserve_old: If True, keeps the old weights as a column when - new weights are provided. - :type weights: np.array - """ - if preserve_old and self.weights_col is not None: - self["old_" + self.weights_col] = self.weights - - if isinstance(weights, str): - self.weights_col = weights - # Keep stored weights independent from edits to the source column. - self.weights = pd.Series( - np.array(self[weights], copy=True), - index=self.index, - dtype=float, - ) - self._link_all_weights() - elif weights is not None: - if len(weights) != len(self): - raise ValueError( - f"Length of weights ({len(weights)}) does not match " - f"length of DataFrame ({len(self)})." - ) - self.weights_col = None - # Align weights to self.index. Without this, weighted ops - # (self[col].multiply(self.weights) in .sum()) align on - # label, so any non-default index silently produces all-NaN - # and aggregations collapse to 0. If a Series is passed in, - # strip its index so we position-align to self.index. - if isinstance(weights, pd.Series): - weights = weights.values - with warnings.catch_warnings(): - warnings.filterwarnings("ignore", category=UserWarning) - self.weights = pd.Series( - np.asarray(weights), index=self.index, dtype=float - ) - self._link_all_weights() - - def set_weight_col(self, column: str, preserve_old: Optional[bool] = False) -> None: - """Sets the weights for the MicroDataFrame by specifying the name of - the weight column. - - .. deprecated:: 1.0.2 - Use :meth:`set_weights` with a string argument instead. - This method will be removed in a future version. - - :param column: Name of the column to use as weights. - :param preserve_old: If True, keeps the old weights as a column when - new weights are provided. - :type column: str - """ - import warnings - - warnings.warn( - "set_weight_col is deprecated and will be removed in a " - "future version. Use set_weights(column_name) instead.", - DeprecationWarning, - stacklevel=2, - ) - - if preserve_old and self.weights_col is not None: - self["old_" + self.weights_col] = self.weights - - # Delegate to set_weights: it validates length and builds an - # index-aligned float Series rather than a bare ndarray. - self.set_weights(column) - - def nullify_weights(self) -> None: - """Set all weights to 1, effectively making the DataFrame unweighted. - - This is useful for comparing weighted and unweighted statistics or when - you want to temporarily ignore weights. - """ - # Route through set_weights so self.weights stays an index-aligned - # float Series. Assigning a bare ndarray here broke every caller - # that treats it as a Series (equals(), reindex() in __getitem__). - self.set_weights(np.ones(len(self))) - - def __getitem__( - self, key: Union[str, List] - ) -> Union[MicroSeries, "MicroDataFrame"]: - # Let pandas handle the initial slicing - result = super().__getitem__(key) - - # If the result is a DataFrame, re-synchronize the weights - if isinstance(result, pd.DataFrame): - new_weights = self.weights.reindex(result.index) - return MicroDataFrame(result, weights=new_weights) - - # If the result is a Series (single column), wrap as MicroSeries - if isinstance(result, pd.Series): - return MicroSeries(result, weights=self.weights) - - # Otherwise, the result is a scalar, so just return it - return result - - def catch_series_relapse(self) -> None: - # In pandas 3.0+, we don't need to track series class changes since - # __getitem__ always wraps columns as MicroSeries on access. - pass - - def __setattr__(self, key, value) -> None: - super().__setattr__(key, value) - # No need to call catch_series_relapse in pandas 3.0+ since we wrap - # on access rather than store MicroSeries internally. - - def reset_index( - self, - level: Optional[int] = None, - drop: Optional[bool] = False, - inplace: Optional[bool] = False, - col_level: Optional[int] = 0, - col_fill: Optional[str] = "", - allow_duplicates: Optional[bool] = None, - names: Optional[List[str]] = None, - ) -> Union["MicroDataFrame", None]: - """Reset the index of the MicroDataFrame. - - This method supports all parameters of pandas DataFrame.reset_index(), - including the 'inplace' parameter. - - :param level: Only remove the given levels from the index. Removes all - levels by default. - :param drop: Do not try to insert index into dataframe columns. This - resets the index to the default integer index. - :param inplace: Modify the DataFrame in place (do not create a new - object). - :param col_level: If the columns have multiple levels, determines which - level the labels are inserted into. - :param col_fill: If the columns have multiple levels, determines how - the other levels are named. - :param allow_duplicates: Allow duplicate column labels to be created. - :param names: Using the given string, rename the DataFrame column which - contains the index data. - :return: MicroDataFrame with reset index or None if inplace=True. - """ - if inplace: - # Snapshot weight *values* positionally — the index is about - # to change and reset_index preserves row order. - weight_values = np.asarray(self.weights.values, dtype=float) - super().reset_index( - level=level, - drop=drop, - inplace=True, - col_level=col_level, - col_fill=col_fill, - allow_duplicates=allow_duplicates, - names=names, - ) - self.weights = pd.Series(weight_values, index=self.index, dtype=float) - self._link_all_weights() - return None - else: - res = super().reset_index( - level=level, - drop=drop, - inplace=False, - col_level=col_level, - col_fill=col_fill, - allow_duplicates=allow_duplicates, - names=names, - ) - out = MicroDataFrame(res, weights=self.weights.values) - # Ensure weights align to res.index (reset_index changes the - # index but preserves row order, so pass values positionally). - out.weights = pd.Series( - np.asarray(self.weights.values, dtype=float), - index=out.index, - dtype=float, - ) - return out - - def copy(self, deep: Optional[bool] = True) -> "MicroDataFrame": - res = super().copy(deep) - # super().copy() corrupts self's column types to plain Series. - # Restore them in O(N) instead of O(N²) by calling - # _link_all_weights once rather than per-column __setitem__. - self._link_all_weights() - res = MicroDataFrame(res, weights=self.weights.copy(deep)) - return res - - def drop( - self, - labels=None, - axis=0, - index=None, - columns=None, - level=None, - inplace=False, - errors="raise", - ): - """Drop specified labels from rows or columns. - - This method supports all parameters of pandas DataFrame.drop(), - including the 'inplace' parameter. - - :param labels: Index or column labels to drop. - :param axis: Whether to drop labels from the index (0 or 'index') or - columns (1 or 'columns'). - :param index: Alternative to specifying axis (labels, axis=0 is - equivalent to index=labels). - :param columns: Alternative to specifying axis (labels, axis=1 is - equivalent to columns=labels). - :param level: For MultiIndex, level from which the labels will be - removed. - :param inplace: If False, return a copy. Otherwise, do operation - inplace and return None. - :param errors: If 'ignore', suppress error and only existing labels are - dropped. - :return: MicroDataFrame or None if inplace=True. - """ - row_drop = axis in (0, "index") or index is not None - if inplace: - # Snapshot the pre-drop weights keyed by the pre-drop index so - # we can reindex to the surviving rows after the drop. - pre_drop_weights = pd.Series(self.weights.values, index=self.index.copy()) - # Perform in-place drop on the parent DataFrame - super().drop( - labels=labels, - axis=axis, - index=index, - columns=columns, - level=level, - inplace=True, - errors=errors, - ) - if row_drop: - surviving = pre_drop_weights.reindex(self.index) - self.weights = pd.Series( - surviving.values, index=self.index, dtype=float - ) - else: - self.weights = pd.Series( - pre_drop_weights.values, index=self.index, dtype=float - ) - self._link_all_weights() - return None - else: - res = super().drop( - labels=labels, - axis=axis, - index=index, - columns=columns, - level=level, - inplace=False, - errors=errors, - ) - if row_drop: - # Row drop: keep only the weights for surviving rows, - # in the order of the resulting DataFrame. - pre_drop_weights = pd.Series(self.weights.values, index=self.index) - new_weights = pre_drop_weights.reindex(res.index).values - else: - new_weights = self.weights.values - out = MicroDataFrame(res, weights=new_weights) - # Guard against the set_weights path building weights with a - # default RangeIndex, which would misalign against res.index - # and silently zero weighted aggregations. - out.weights = pd.Series(new_weights, index=out.index, dtype=float) - return out - - def merge( - self, - right, - how="inner", - on=None, - left_on=None, - right_on=None, - left_index=False, - right_index=False, - sort=False, - suffixes=("_x", "_y"), - copy=True, - indicator=False, - validate=None, - ): - """Merge DataFrame or named Series objects with a database-style join. - - This method overrides pandas DataFrame.merge() to return a - MicroDataFrame. - - :param right: Object to merge with. - :param how: Type of merge to be performed. - :param on: Column or index level names to join on. - :param left_on: Column or index level names to join on in the left - DataFrame. - :param right_on: Column or index level names to join on in the right - DataFrame. - :param left_index: Use the index from the left DataFrame as the join - key(s). - :param right_index: Use the index from the right DataFrame as the join - key(s). - :param sort: Sort the join keys lexicographically in the result - DataFrame. - :param suffixes: A length-2 sequence where each element is optionally a - string indicating the suffix to add to overlapping column names. - :param copy: If False, avoid copy if possible. - :param indicator: If True, adds a column to output DataFrame called - "_merge". - :param validate: If specified, checks if merge is of specified type. - :return: MicroDataFrame with merged data. - """ - # Attach the left weights as a temporary column so pandas' merge - # propagates them onto every surviving output row (including - # many-to-many row duplications, inner-join filtering, and - # left-with-missing NaNs). We then strip the column back off. - tmp = "__microdf_weights__" - # Avoid clobbering if this exact name is already used. - while tmp in self.columns or tmp in right.columns: - tmp += "_" - left_df = pd.DataFrame(self).copy() - left_df[tmp] = np.asarray(self.weights.values, dtype=float) - res = left_df.merge( - right, - how=how, - on=on, - left_on=left_on, - right_on=right_on, - left_index=left_index, - right_index=right_index, - sort=sort, - suffixes=suffixes, - copy=copy, - indicator=indicator, - validate=validate, - ) - # Pull out the propagated weights. Rows with no left match in a - # right/outer join get NaN weight — fill with 0 so they don't - # poison later aggregations (a user who needs a different - # convention can override afterwards). - merged_weights = res[tmp].fillna(0).to_numpy(dtype=float) - res = res.drop(columns=[tmp]) - out = MicroDataFrame(res, weights=merged_weights) - # Ensure the weights Series aligns with res.index regardless of - # the default-RangeIndex behavior of set_weights. - out.weights = pd.Series(merged_weights, index=out.index, dtype=float) - return out - - def __getattr__(self, name): - """Allow accessing columns as attributes (e.g., df.column_name). - - This enables more intuitive column access while preserving MicroSeries - functionality when accessing columns. - - :param name: Attribute name to access - :return: MicroSeries if the attribute is a column, otherwise delegates - to parent - """ - if name in self.columns: - return self[name] - return super().__getattr__(name) - - def equals(self, other: "MicroDataFrame") -> bool: - equal_values = super().equals(other) - equal_weights = self.weights.equals(other.weights) - return equal_values and equal_weights - - @get_args_as_micro_series() - def groupby(self, by: Union[str, List], *args, **kwargs) -> "MicroDataFrameGroupBy": - """Returns a GroupBy object with MicroSeriesGroupBy objects for each - column. - - :param by: column to group by - :type by: Union[str, List] - - return: DataFrameGroupBy object with columns using weights - rtype: DataFrameGroupBy - """ - # Build the groupby on a *copy* that carries a ``__tmp_weights`` - # column. We used to set this column on ``self`` directly, which - # permanently leaked the weight column onto the caller's - # DataFrame — any later ``df.sum()`` or ``list(df.columns)`` - # would then include it. - staged = pd.DataFrame(self).copy() - staged["__tmp_weights"] = np.asarray(self.weights.values, dtype=float) - gb = staged.groupby(by, *args, **kwargs) - weights = copy.deepcopy(gb["__tmp_weights"]) - for col in staged.columns: # df.groupby(...)[col]s use weights - res = gb[col] - res.__class__ = MicroSeriesGroupBy - res._init() - res.weights = weights - setattr(gb, col, res) - gb.__class__ = MicroDataFrameGroupBy - gb._init(by) - return gb - - @get_args_as_micro_series() - def poverty_rate(self, income: str, threshold: str) -> float: - """Calculate poverty rate, i.e., the population share with income below - their poverty threshold. - - :param income: Column indicating income. - :type income: str - :param threshold: Column indicating threshold. - :type threshold: str - :return: Poverty rate between zero and one. - :rtype: float - """ - pov = income < threshold - return pov.sum() / pov.count() - - @get_args_as_micro_series() - def deep_poverty_rate(self, income: str, threshold: str) -> float: - """Calculate deep poverty rate, i.e., the population share with income - below half their poverty threshold. - - :param income: Column indicating income. - :type income: str - :param threshold: Column indicating threshold. - :type threshold: str - :return: Deep poverty rate between zero and one. - :rtype: float - """ - pov = income < (threshold / 2) - return pov.sum() / pov.count() - - @get_args_as_micro_series() - def poverty_gap(self, income: str, threshold: str) -> float: - """Calculate poverty gap, i.e., the total gap between income and - poverty thresholds for all people in poverty. - - :param income: Column indicating income. - :type income: str - :param threshold: Column indicating threshold. - :type threshold: str - :return: Poverty gap. - :rtype: float - """ - gaps = (threshold - income)[threshold > income] - return gaps.sum() - - @get_args_as_micro_series() - def deep_poverty_gap(self, income: str, threshold: str) -> float: - """Calculate deep poverty gap, i.e., the total gap between income and - half of poverty thresholds for all people in deep poverty. - - :param income: Column indicating income. - :type income: str - :param threshold: Column indicating threshold. - :type threshold: str - :return: Deep poverty gap. - :rtype: float - """ - deep_threshold = threshold / 2 - gaps = (deep_threshold - income)[deep_threshold > income] - return gaps.sum() - - @get_args_as_micro_series() - def squared_poverty_gap(self, income: str, threshold: str) -> float: - """Calculate squared poverty gap, i.e., the total squared gap between - income and poverty thresholds for all people in poverty. Also known as - the poverty severity index. - - :param income: Column indicating income. - :type income: str - :param threshold: Column indicating threshold. - :type threshold: str - :return: Squared poverty gap. - :rtype: float - """ - gaps = (threshold - income)[threshold > income] - squared_gaps = gaps**2 - return squared_gaps.sum() - - @get_args_as_micro_series() - def poverty_count( - self, - income: Union[MicroSeries, str], - threshold: Union[MicroSeries, str], - ) -> int: - """Calculates the number of entities with income below a poverty - threshold. - - :param income: income array or column name - :type income: Union[MicroSeries, str] - - :param threshold: threshold array or column name - :type threshold: Union[MicroSeries, str] - - return: number of entities in poverty - rtype: int - """ - in_poverty = income < threshold - return in_poverty.sum() - - def astype( - self, - dtype, - copy: Optional[bool] = True, - errors: Optional[str] = "raise", - ) -> "MicroDataFrame": - """Convert MicroDataFrame to specified data type while preserving - weights. - - :param dtype: Data type to convert to. Can be numpy dtype, Python type, - or dict. - :param copy: Whether to make a copy of the data (default True). - :param errors: How to handle conversion errors (default "raise"). - :return: New MicroDataFrame with converted data types and preserved - weights. - """ - converted_df = super().astype(dtype, copy=copy, errors=errors) - return MicroDataFrame( - converted_df, weights=self.weights.copy() if copy else self.weights - ) - - def __repr__(self) -> str: - df = pd.DataFrame(self) - df["weight"] = self.weights - return df[[df.columns[-1]] + list(df.columns[:-1])].__repr__() - - -class MicroDataFrameGroupBy(pd.core.groupby.generic.DataFrameGroupBy): - def _init(self, by: Union[str, List]): - self._by = by - self.columns = list(self.obj.columns) - if isinstance(by, list): - for column in by: - self.columns.remove(column) - elif isinstance(by, str): - self.columns.remove(by) - self.columns.remove("__tmp_weights") - # Filter to only numeric columns - self.numeric_columns = [ - col for col in self.columns if pd.api.types.is_numeric_dtype(self.obj[col]) - ] - # Store reference to weights groupby for column selection - self._weights_groupby = copy.deepcopy(super().__getitem__("__tmp_weights")) - for fn_name in MicroSeries.SCALAR_FUNCTIONS: - - def get_fn(name): - def fn(*args, **kwargs): - results = {} - for col in self.numeric_columns: - try: - results[col] = getattr(getattr(self, col), name)( - *args, **kwargs - ) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - # Return plain DataFrame - aggregated results don't have - # per-row weights (weights were already applied) - return pd.DataFrame(results) if results else pd.DataFrame() - - return fn - - setattr(self, fn_name, get_fn(fn_name)) - for fn_name in MicroSeries.VECTOR_FUNCTIONS: - - def get_fn(name) -> Callable: - def fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: - results = {} - for col in self.numeric_columns: - try: - results[col] = getattr(getattr(self, col), name)( - *args, **kwargs - ) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - # Return plain DataFrame - aggregated results don't have - # per-row weights (weights were already applied) - return pd.DataFrame(results) if results else pd.DataFrame() - - return fn - - setattr(self, fn_name, get_fn(fn_name)) - - def __getitem__( - self, key: Union[str, List] - ) -> Union["MicroSeriesGroupBy", "MicroDataFrameGroupBy"]: - """Select columns from the groupby object while preserving weights. - - This ensures that operations like groupby(col)["y"].sum() or - groupby(col)[["y"]].sum() use weighted aggregation. - - :param key: Column name or list of column names - :return: MicroSeriesGroupBy for single column, MicroDataFrameGroupBy - for multiple columns - """ - if isinstance(key, str): - # Single column - return MicroSeriesGroupBy - result = super().__getitem__(key) - result.__class__ = MicroSeriesGroupBy - result._init() - result.weights = self._weights_groupby - return result - else: - # Multiple columns - return a new MicroDataFrameGroupBy - # with only the selected columns - result = super().__getitem__(key) - result.__class__ = MicroDataFrameGroupBy - # Re-initialize with the subset of columns - result._by = self._by - result.columns = list(key) if hasattr(key, "__iter__") else [key] - result.numeric_columns = [ - col - for col in result.columns - if pd.api.types.is_numeric_dtype(result.obj[col]) - ] - result._weights_groupby = self._weights_groupby - # Set up the column attributes as MicroSeriesGroupBy - for col in result.columns: - col_gb = super().__getitem__(col) - col_gb.__class__ = MicroSeriesGroupBy - col_gb._init() - col_gb.weights = self._weights_groupby - setattr(result, col, col_gb) - # Set up the scalar and vector functions - for fn_name in MicroSeries.SCALAR_FUNCTIONS: - - def get_scalar_fn(name, res): - def fn(*args, **kwargs): - results = {} - for col in res.numeric_columns: - try: - results[col] = getattr(getattr(res, col), name)( - *args, **kwargs - ) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug( - "skipping column %s in %s: %s", col, name, exc - ) - # Return plain DataFrame - aggregated results don't - # have per-row weights (weights were already applied) - return pd.DataFrame(results) if results else pd.DataFrame() - - return fn - - setattr(result, fn_name, get_scalar_fn(fn_name, result)) - for fn_name in MicroSeries.VECTOR_FUNCTIONS: - - def get_vector_fn(name, res): - def fn(*args, **kwargs): - results = {} - for col in res.numeric_columns: - try: - results[col] = getattr(getattr(res, col), name)( - *args, **kwargs - ) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug( - "skipping column %s in %s: %s", col, name, exc - ) - # Return plain DataFrame - aggregated results don't - # have per-row weights (weights were already applied) - return pd.DataFrame(results) if results else pd.DataFrame() - - return fn - - setattr(result, fn_name, get_vector_fn(fn_name, result)) - return result diff --git a/build/lib/microdf/microseries.py b/build/lib/microdf/microseries.py deleted file mode 100644 index 688f8e77..00000000 --- a/build/lib/microdf/microseries.py +++ /dev/null @@ -1,983 +0,0 @@ -import logging -import warnings -from functools import wraps -from typing import Callable, List, Optional, Union - -import numpy as np -import pandas as pd - -logger = logging.getLogger(__name__) - - -def _weighted_top_share( - values: np.ndarray, weights: np.ndarray, top_x_pct: float -) -> float: - """Share of the sum held by the top ``top_x_pct`` of weight. - - Sort by value ascending, cumulate the weight, pick the slice from the top - that covers exactly ``top_x_pct`` of total weight, and distribute the tied- - at-cutoff row proportionally so constant values return exactly - ``top_x_pct`` rather than 1.0. - """ - if top_x_pct <= 0: - return 0.0 - if top_x_pct >= 1: - return 1.0 - total_weight = weights.sum() - total_sum = float((values * weights).sum()) - if total_weight == 0 or total_sum == 0: - return np.nan - # Ascending sort; the "top" cutoff is the final ``top_x_pct`` of - # cumulative weight. - order = np.argsort(values, kind="mergesort") - v = values[order] - w = weights[order] - # Cumulative weight from the bottom up. - cum_w = np.cumsum(w) - target_bottom_weight = total_weight * (1.0 - top_x_pct) - # searchsorted(cum_w, target, side="right") gives the first index - # whose cumulative weight exceeds the bottom cutoff. - k = int(np.searchsorted(cum_w, target_bottom_weight, side="right")) - # Rows strictly above the cutoff contribute all of their weight. - if k >= len(v): - return 0.0 - top_sum = float((v[k + 1 :] * w[k + 1 :]).sum()) - # Row k straddles the cutoff; include the fraction of its weight - # that lies above the cutoff so ties don't double-count. - partial_weight = cum_w[k] - target_bottom_weight - top_sum += float(v[k] * partial_weight) - return top_sum / total_sum - - -class MicroSeries(pd.Series): - # Declare ``weights`` as pandas metadata. pandas includes - # _metadata attributes in the pickle state, so weights now survive - # pickling, to_pickle/read_pickle and copy.deepcopy instead of - # vanishing and leaving an AttributeError on the next aggregation. - # Keep pandas' own metadata, including the Series name. - _metadata = pd.Series._metadata + ["weights"] - - def __init__(self, *args, weights: np.array = None, **kwargs): - """A Series-inheriting class for weighted microdata. - - Weights can be provided at initialisation, or using set_weights. - - :param weights: Array of weights. - :type weights: np.array - """ - super().__init__(*args, **kwargs) - self.set_weights(weights) - - def __finalize__(self, other, method=None, **kwargs) -> "MicroSeries": - """Retain copied weights when pandas finalizes a renamed result.""" - copied_weights = getattr(self, "weights", None) if method == "rename" else None - super().__finalize__(other, method=method, **kwargs) - if copied_weights is not None: - # rename already called copy(); metadata propagation must not - # replace those weights with the source's mutable Series. - self.weights = copied_weights - return self - - @property - def _values(self): - """Internal access to underlying numpy array without warning.""" - return super().values - - @property - def values(self): - """Access underlying numpy array. - - .. warning:: - Returns a plain numpy array without weights. Operations - like ``.mean()`` on the result will be unweighted. Use - MicroSeries methods directly for weighted calculations - (e.g., ``ms.mean()`` instead of ``ms.values.mean()``). - """ - warnings.warn( - "Accessing .values on a MicroSeries returns a plain numpy " - "array without weights. Operations like .mean() on the " - "result will be unweighted. Use MicroSeries methods " - "directly for weighted calculations (e.g., ms.mean() " - "instead of ms.values.mean()).", - UserWarning, - stacklevel=2, - ) - return super().values - - def to_numpy(self, *args, **kwargs): - """Convert to numpy array. - - .. warning:: - Returns a plain numpy array without weights. Operations - like ``.mean()`` on the result will be unweighted. Use - MicroSeries methods directly for weighted calculations. - """ - warnings.warn( - "Calling .to_numpy() on a MicroSeries returns a plain " - "numpy array without weights. Operations like .mean() on " - "the result will be unweighted. Use MicroSeries methods " - "directly for weighted calculations.", - UserWarning, - stacklevel=2, - ) - return super().to_numpy(*args, **kwargs) - - def scalar_function(fn: Callable) -> Callable: - """Decorator marking ``fn`` as returning a scalar (float).""" - fn._rtype = float - return fn - - def vector_function(fn: Callable) -> Callable: - """Decorator marking ``fn`` as returning a pandas Series.""" - fn._rtype = pd.Series - return fn - - def set_weights( - self, weights: np.array, preserve_old: Optional[bool] = False - ) -> None: - """Sets the weight values. - - :param weights: Array of weights. - :param preserve_old: If True, keeps the old weights as a column when - new weights are provided. - :type weights: np.array. - """ - if weights is None: - if len(self) > 0: - self.weights = pd.Series( - np.ones_like(self._values), - index=self.index, - dtype=float, - ) - else: - if len(weights) != len(self): - raise ValueError( - f"Length of weights ({len(weights)}) does not match " - f"length of DataFrame ({len(self)})." - ) - - if preserve_old and self.weights is not None: - self["old_weights"] = self.weights - - # Align weights to self.index so element-wise operations such - # as self.multiply(self.weights) (used by .sum(), .weight()) - # don't silently produce all-NaN when the caller uses a - # non-default index. If a pandas Series is passed in, strip - # its index first so we position-align rather than label-align. - if isinstance(weights, pd.Series): - weights = weights.values - self.weights = pd.Series(np.asarray(weights), index=self.index, dtype=float) - - def nullify_weights(self) -> None: - """Set all weights to 1, effectively making the Series unweighted. - - This is useful for comparing weighted and unweighted statistics or when - you want to temporarily ignore weights. - """ - # Index the ones against self.index: weighted ops are label-aligned - # (self.multiply(self.weights) in .sum()/.weight()), so a default - # RangeIndex here silently produces all-NaN and collapses every - # aggregation to 0 whenever the caller uses a non-default index. - self.weights = pd.Series(np.ones(len(self)), index=self.index, dtype=float) - - @vector_function - def weight(self) -> pd.Series: - """Calculates the weighted value of the MicroSeries. - - :returns: A Series multiplying the MicroSeries by its weight. - :rtype: pd.Series - """ - return self.multiply(self.weights) - - @scalar_function - def sum(self) -> float: - """Calculates the weighted sum of the MicroSeries. - - :returns: The weighted sum. - :rtype: float - """ - return self.multiply(self.weights).sum() - - @scalar_function - def count(self, skipna: bool = True) -> float: - """Calculates the weighted count of the MicroSeries. - - By default skips NaN values (matching pandas ``Series.count``). - - :param skipna: Exclude NaN values (default True). If False, the - weighted count of every row is returned. - :type skipna: bool - :returns: The weighted count. - :rtype: float - """ - weights = np.asarray(self.weights.values, dtype=float) - if not skipna: - return float(weights.sum()) - mask = ~pd.isna(self._values) - return float(weights[mask].sum()) - - @scalar_function - def mean(self, skipna: bool = True) -> float: - """Calculates the weighted mean of the MicroSeries. - - :param skipna: Exclude NA/null values. If True (default), NaN values - are excluded. If False, returns NaN if any value is NaN. - :type skipna: bool - :returns: The weighted mean. - :rtype: float - """ - values = self._values - weights = self.weights - - if skipna: - # Create mask for non-NaN values - mask = ~pd.isna(values) - if not mask.any(): - # All values are NaN - return np.nan - values = values[mask] - weights = weights[mask] - - # If skipna=False and there are any NaN values, return NaN - if not skipna and pd.isna(values).any(): - return np.nan - - return np.average(values, weights=weights) - - def _weighted_variance(self, ddof: int = 1, skipna: bool = True) -> float: - """Frequency-weighted variance. - - Uses ``sum(w * (x - wmean)**2) / (sum(w) - ddof)``. With - ``ddof=0`` this is the population variance; with ``ddof=1`` it - is Bessel-corrected assuming the weights are frequency counts — - matching ``np.var(..., ddof=ddof)`` on a replicated sample. - """ - values = np.asarray(self._values, dtype=float) - weights = np.asarray(self.weights.values, dtype=float) - if skipna: - mask = ~np.isnan(values) - values = values[mask] - weights = weights[mask] - elif np.isnan(values).any(): - return np.nan - total_w = weights.sum() - if total_w == 0 or total_w - ddof <= 0: - return np.nan - mean = np.average(values, weights=weights) - return float((weights * (values - mean) ** 2).sum() / (total_w - ddof)) - - @scalar_function - def var(self, ddof: int = 1, skipna: bool = True) -> float: - """Calculates the weighted variance of the MicroSeries. - - Treats weights as frequency counts (``sum(w) - ddof`` in the - denominator) so that with integer weights the result matches - ``np.var`` on the replicated sample. - - :param ddof: Delta degrees of freedom (default 1). - :param skipna: Exclude NaN values (default True). - :returns: The weighted variance. - :rtype: float - """ - return self._weighted_variance(ddof=ddof, skipna=skipna) - - @scalar_function - def std(self, ddof: int = 1, skipna: bool = True) -> float: - """Calculates the weighted standard deviation of the MicroSeries. - - :param ddof: Delta degrees of freedom (default 1). - :param skipna: Exclude NaN values (default True). - :returns: The weighted standard deviation. - :rtype: float - """ - v = self._weighted_variance(ddof=ddof, skipna=skipna) - return float(np.sqrt(v)) if np.isfinite(v) else v - - def cov(self, other, *args, **kwargs): - """Pandas ``cov`` — **unweighted**. - - MicroSeries does not yet compute weighted covariance. Emits a - ``UserWarning`` so callers aren't silently given an unweighted number - after ``.sum()`` and ``.mean()`` worked as expected. See issue tracker - for a weighted implementation. - """ - warnings.warn( - "MicroSeries.cov() falls through to pandas and is " - "unweighted. Use MicroSeries.var()/std() for weighted " - "second moments, or compute covariance manually with the " - "weights.", - UserWarning, - stacklevel=2, - ) - return super().cov(other, *args, **kwargs) - - def corr(self, other, *args, **kwargs): - """Pandas ``corr`` — **unweighted**. - - MicroSeries does not yet compute weighted correlation. Emits a - ``UserWarning`` so callers aren't silently given an unweighted number. - See issue tracker for a weighted implementation. - """ - warnings.warn( - "MicroSeries.corr() falls through to pandas and is " - "unweighted. Compute correlation manually with the weights " - "if you need the survey-weighted value.", - UserWarning, - stacklevel=2, - ) - return super().corr(other, *args, **kwargs) - - def quantile(self, q: np.array, skipna: bool = True) -> pd.Series: - """Calculates weighted quantiles of the MicroSeries. - - Uses the inverse CDF method: the q-th quantile is the smallest - value where the cumulative weight proportion >= q. This matches - the default behavior of R's survey::svyquantile. - - :param q: Quantile(s) to calculate, must be in [0, 1]. - :type q: float or np.array - :param skipna: Exclude NaN values (default True). NaN sorts to the - end of the array, so leaving NaN rows in would let their weight - inflate the cumulative distribution and push the cutoff upward. - If False, NaN is returned whenever any value is NaN. - :type skipna: bool - - :return: Weighted quantile value(s). - :rtype: float or pd.Series - """ - values = np.array(self._values) - quantiles = np.atleast_1d(q) - sample_weight = np.array(self.weights) - assert np.all(quantiles >= 0) and np.all(quantiles <= 1), ( - "quantiles should be in [0, 1]" - ) - na_mask = pd.isna(values) - if not skipna and na_mask.any(): - return ( - np.nan - if np.array(q).shape == () - else pd.Series(np.full(len(quantiles), np.nan), index=quantiles) - ) - # Drop zero-weight rows before sorting. Without this, q=0 (and - # internal plateaus of zero weight) picked a value with 0 weight - # that should have been skipped by the inverse CDF. E.g. - # MicroSeries([10, 20, 30], weights=[0, 1, 1]).quantile(0) - # returned 10 instead of 20. - # Drop NaN rows for the same reason: NaN sorts last, so its weight - # would inflate the cumulative distribution and push the cutoff up - # (median of [1, nan, 3] returned 3.0 instead of 1.0). - nonzero = (sample_weight > 0) & ~na_mask - if not nonzero.any(): - return ( - np.nan - if np.array(q).shape == () - else pd.Series(np.full(len(quantiles), np.nan), index=quantiles) - ) - values = values[nonzero] - sample_weight = sample_weight[nonzero] - sorter = np.argsort(values) - values = values[sorter] - sample_weight = sample_weight[sorter] - cumsum = np.cumsum(sample_weight) - cumsum_normalized = cumsum / cumsum[-1] - result = np.array( - [ - values[min(np.searchsorted(cumsum_normalized, qi), len(values) - 1)] - for qi in quantiles - ] - ) - if np.array(q).shape == (): - return result[0] - return pd.Series(result, index=quantiles) - - @scalar_function - def median(self, skipna: bool = True) -> float: - """Calculates the weighted median of the MicroSeries. - - :param skipna: Exclude NaN values (default True). - :type skipna: bool - :returns: The weighted median of a DataFrame's column. - :rtype: float - """ - return self.quantile(0.5, skipna=skipna) - - def replicate_standard_error( - self, - statistic: Callable, - replicate_weights, - method: str = "jackknife", - fay_k: Optional[float] = None, - ) -> float: - """Standard error of ``statistic`` from a set of replicate weights. - - Recomputes the statistic once per replicate and scales the spread by - the factor appropriate to how the replicates were built, so it works - for any statistic including the Gini coefficient and quantiles. - - Only valid for replicate weights as published with a survey. Weights - calibrated to external targets no longer correspond to the original - replication scheme. - - :param statistic: Callable taking a MicroSeries, e.g. - ``lambda s: s.median()``. - :param replicate_weights: Array or frame of shape ``(len(self), R)``. - :param method: ``jackknife``, ``brr``, ``bootstrap``, - ``successive-difference`` or ``fay``. - :param fay_k: Fay's perturbation constant, for ``method="fay"``. - :returns: The estimated standard error. - """ - from microdf.replication import replicate_standard_error - - return replicate_standard_error( - self, statistic, replicate_weights, method, fay_k - ) - - @scalar_function - def gini(self, negatives: Optional[str] = None) -> float: - """Calculates Gini index. - - :param negatives: An optional string indicating how to treat - negative values of x: - 'zero' replaces negative values with zeroes. - 'shift' subtracts the minimum value from all values of x, - when this minimum is negative. That is, it adds the absolute - minimum value. - Defaults to None, which leaves negative values as they are. - :type negatives: str - :returns: Gini index. - :rtype: float - """ - x = np.array(self).astype("float") - w = np.asarray(self.weights.values, dtype=float) - if negatives == "zero": - x = np.where(x < 0, 0.0, x) - elif negatives == "shift": - if len(x) > 0 and np.amin(x) < 0: - x = x - np.amin(x) - elif negatives is not None: - raise ValueError( - f"Unknown negatives option {negatives!r}; expected " - "'zero', 'shift', or None." - ) - - if len(x) == 0: - return np.nan - if np.any(x < 0): - # The Lorenz-based formula assumes non-negative values; with - # negatives it can return values outside [0, 1]. - warnings.warn( - "gini() called on data containing negative values; the " - "result is not guaranteed to lie in [0, 1]. Pass " - "negatives='zero' or negatives='shift' to handle them.", - UserWarning, - stacklevel=2, - ) - - # Short-circuit degenerate cases so we don't divide by zero. - total = float((x * w).sum()) - if total == 0: - return 0.0 - - sorter = np.argsort(x, kind="mergesort") - sorted_x = x[sorter] - sorted_w = w[sorter] - cumw = np.cumsum(sorted_w) - cumxw = np.cumsum(sorted_x * sorted_w) - # Trapezoidal approximation of the area under the Lorenz curve. - return float( - np.sum(cumxw[1:] * cumw[:-1] - cumxw[:-1] * cumw[1:]) - / (cumxw[-1] * cumw[-1]) - ) - - @scalar_function - def top_x_pct_share(self, top_x_pct: float) -> float: - """Calculates top x% share. - - Uses a cumulative-weight sort so that rows tied at the cutoff - contribute proportionally rather than all-or-nothing. With - constant values this correctly returns ``top_x_pct`` itself. - - :param top_x_pct: Decimal between 0 and 1 of the top %, e.g. 0.1, - 0.001. - :type top_x_pct: float - :returns: The weighted share held by the top x%. - :rtype: float - """ - return _weighted_top_share( - np.asarray(self._values, dtype=float), - np.asarray(self.weights.values, dtype=float), - float(top_x_pct), - ) - - @scalar_function - def bottom_x_pct_share(self, bottom_x_pct: float) -> float: - """Calculates bottom x% share. - - :param bottom_x_pct: Decimal between 0 and 1 of the bottom %, e.g. 0.1, - 0.001. - :type bottom_x_pct: float - :returns: The weighted share held by the bottom x%. - :rtype: float - """ - return 1 - self.top_x_pct_share(1 - bottom_x_pct) - - @scalar_function - def bottom_50_pct_share(self) -> float: - """Calculates bottom 50% share. - - :returns: The weighted share held by the bottom 50%. - :rtype: float - """ - return self.bottom_x_pct_share(0.5) - - @scalar_function - def top_50_pct_share(self) -> float: - """Calculates top 50% share. - - :returns: The weighted share held by the top 50%. - :rtype: float - """ - return self.top_x_pct_share(0.5) - - @scalar_function - def top_10_pct_share(self) -> float: - """Calculates top 10% share. - - :returns: The weighted share held by the top 10%. - :rtype: float - """ - return self.top_x_pct_share(0.1) - - @scalar_function - def top_1_pct_share(self) -> float: - """Calculates top 1% share. - - :returns: The weighted share held by the top 50%. - :rtype: float - """ - return self.top_x_pct_share(0.01) - - @scalar_function - def top_0_1_pct_share(self) -> float: - """Calculates top 0.1% share. - - :returns: The weighted share held by the top 0.1%. - :rtype: float - """ - return self.top_x_pct_share(0.001) - - @scalar_function - def t10_b50(self) -> float: - """Calculates ratio between the top 10% and bottom 50% shares. - - :returns: The weighted share held by the top 10% divided by the - weighted share held by the bottom 50%. - """ - t10 = self.top_10_pct_share() - b50 = self.bottom_50_pct_share() - return t10 / b50 - - @vector_function - def cumsum(self) -> pd.Series: - logger.warning( - "cumsum() returns cumulative sums of weighted values as a regular " - "pandas Series. The original weights have already been applied " - "and cannot be reused with the cumulative results." - ) - return pd.Series(self * self.weights).cumsum() - - @vector_function - def rank(self, pct: Optional[bool] = False) -> pd.Series: - """Weighted rank of each element. - - Each element's rank is the cumulative weight of all values that are - less than or equal to it. Tied values therefore share the same rank, so - downstream bucketing (``decile_rank``, ``quintile_rank``, etc.) lands - tied rows in the same bucket. - - :param pct: If True, divide ranks by the total weight so they lie in - ``(0, 1]``. - :type pct: bool - :returns: MicroSeries of ranks aligned to ``self``. - :rtype: MicroSeries - """ - weights_sum = np.asarray(self.weights.values, dtype=float).sum() - if weights_sum == 0: - raise ZeroDivisionError( - "Cannot calculate rank with zero total weight. " - "All weights in the MicroSeries are zero, which would " - "result in division by zero." - ) - - values = np.asarray(self._values) - weights = np.asarray(self.weights.values, dtype=float) - order = np.argsort(values, kind="mergesort") - sorted_values = values[order] - sorted_weights = weights[order] - cum_w = np.cumsum(sorted_weights) - # Max rank semantics: every tied group gets the cumulative - # weight at the *end* of the group, so ties share one rank. - # searchsorted(side='right') on the sorted values finds the - # index just past each tied block in sort order. - group_end = np.searchsorted(sorted_values, sorted_values, side="right") - 1 - sorted_ranks = cum_w[group_end] - # Invert the sort to put ranks back into the caller's order. - inverse_order = np.argsort(order, kind="mergesort") - ranks = sorted_ranks[inverse_order] - if pct: - ranks = ranks / weights_sum - ranks = np.where(ranks > 1.0, 1.0, ranks) - return MicroSeries(ranks, index=self.index, weights=self.weights) - - @vector_function - def decile_rank(self, negatives_in_zero: Optional[bool] = False): - """Calculate decile ranks (1-10) with optional zero decile for - negatives. - - :param negatives_in_zero: If True, negative values are assigned to - decile 0. If False (default), all values are ranked 1-10. - :type negatives_in_zero: bool - :returns: MicroSeries with decile ranks - :rtype: MicroSeries - """ - if negatives_in_zero: - negative_mask = self < 0 - if negative_mask.any(): - non_negative_values = self[~negative_mask] - if len(non_negative_values) > 0: - non_neg_ranks = non_negative_values.rank(pct=True) - deciles = np.minimum(np.ceil(non_neg_ranks * 10), 10) - else: - deciles = np.array([]) - - result = np.zeros(len(self)) - result[negative_mask] = 0 - if len(deciles) > 0: - result[~negative_mask] = deciles - - return MicroSeries(result, weights=self.weights) - - # Default behavior: rank all values 1-10 - return MicroSeries( - np.minimum(np.ceil(self.rank(pct=True) * 10), 10), - weights=self.weights, - ) - - @vector_function - def quintile_rank(self) -> "MicroSeries": - return MicroSeries( - np.minimum(np.ceil(self.rank(pct=True) * 5), 5), - weights=self.weights, - ) - - @vector_function - def quartile_rank(self) -> "MicroSeries": - return MicroSeries( - np.minimum(np.ceil(self.rank(pct=True) * 4), 4), - weights=self.weights, - ) - - @vector_function - def percentile_rank(self) -> "MicroSeries": - return MicroSeries( - np.minimum(np.ceil(self.rank(pct=True) * 100), 100), - weights=self.weights, - ) - - def groupby(self, *args, **kwargs) -> "MicroSeriesGroupBy": - gb = super().groupby(*args, **kwargs) - gb.__class__ = MicroSeriesGroupBy - gb._init() - gb.weights = pd.Series(self.weights).groupby(*args, **kwargs) - return gb - - def copy(self, deep: Optional[bool] = True): - res = super().copy(deep) - res = MicroSeries(res, weights=self.weights.copy(deep)) - return res - - def clip( - self, - lower: Optional[float] = None, - upper: Optional[float] = None, - axis: Optional[int] = None, - inplace: Optional[bool] = False, - *args, - **kwargs, - ) -> "MicroSeries": - res = super().clip( - lower=lower, - upper=upper, - axis=axis, - inplace=inplace, - *args, - **kwargs, - ) - if not inplace: - return MicroSeries(res, weights=self.weights) - return self - - def round(self, decimals: Optional[int] = 0, *args, **kwargs) -> "MicroSeries": - res = super().round(decimals=decimals, *args, **kwargs) - return MicroSeries(res, weights=self.weights) - - def equals(self, other: "MicroSeries") -> bool: - equal_values = super().equals(other) - equal_weights = self.weights.equals(other.weights) - return equal_values and equal_weights - - def __getitem__( - self, key: Union[str, int, slice, List, np.ndarray] - ) -> Union["MicroSeries", pd.Series]: - result = super().__getitem__(key) - if isinstance(result, pd.Series): - weights = self.weights.__getitem__(key) - return MicroSeries(result, weights=weights) - return result - - def __getattr__(self, name: str) -> "MicroSeries": - return MicroSeries(super().__getattr__(name), weights=self.weights) - - # operators - - def __add__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__add__(other), weights=self.weights) - - def __sub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__sub__(other), weights=self.weights) - - def __mul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__mul__(other), weights=self.weights) - - def __floordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__floordiv__(other), weights=self.weights) - - def __truediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__truediv__(other), weights=self.weights) - - def __mod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__mod__(other), weights=self.weights) - - def __pow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__pow__(other), weights=self.weights) - - def __xor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__xor__(other), weights=self.weights) - - def __and__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__and__(other), weights=self.weights) - - def __or__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__or__(other), weights=self.weights) - - def __invert__(self) -> "MicroSeries": - return MicroSeries(super().__invert__(), weights=self.weights) - - def __radd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__radd__(other), weights=self.weights) - - def __rsub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__rsub__(other), weights=self.weights) - - def __rmul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__rmul__(other), weights=self.weights) - - def __rfloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__rfloordiv__(other), weights=self.weights) - - def __rtruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__rtruediv__(other), weights=self.weights) - - def __rmod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__rmod__(other), weights=self.weights) - - def __rpow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__rpow__(other), weights=self.weights) - - def __rand__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__rand__(other), weights=self.weights) - - def __ror__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__ror__(other), weights=self.weights) - - def __rxor__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__rxor__(other), weights=self.weights) - - def sqrt(self) -> "MicroSeries": - sqrt_values = np.sqrt(self._values) - return MicroSeries(sqrt_values, index=self.index, weights=self.weights) - - # comparators - - def __lt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__lt__(other), weights=self.weights) - - def __le__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__le__(other), weights=self.weights) - - def __eq__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__eq__(other), weights=self.weights) - - def __ne__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__ne__(other), weights=self.weights) - - def __ge__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__ge__(other), weights=self.weights) - - def __gt__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__gt__(other), weights=self.weights) - - # assignment operators - - def __iadd__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__iadd__(other), weights=self.weights) - - def __isub__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__isub__(other), weights=self.weights) - - def __imul__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__imul__(other), weights=self.weights) - - def __ifloordiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__ifloordiv__(other), weights=self.weights) - - def __idiv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__idiv__(other), weights=self.weights) - - def __itruediv__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__itruediv__(other), weights=self.weights) - - def __imod__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__imod__(other), weights=self.weights) - - def __ipow__(self, other: Union[int, float, pd.Series]) -> "MicroSeries": - return MicroSeries(super().__ipow__(other), weights=self.weights) - - # other - - def __neg__(self) -> "MicroSeries": - return MicroSeries(super().__neg__(), weights=self.weights) - - def __pos__(self) -> "MicroSeries": - return MicroSeries(super().__pos__(), weights=self.weights) - - def astype( - self, - dtype, - copy: Optional[bool] = True, - errors: Optional[str] = "raise", - ) -> "MicroSeries": - """Convert MicroSeries to specified data type while preserving weights. - - :param dtype: Data type to convert to. Can be numpy dtype or Python - type. - :param copy: Whether to make a copy of the data (default True). - :param errors: How to handle conversion errors (default "raise"). - :return: New MicroSeries with converted data type and preserved - weights. - """ - converted_series = super().astype(dtype, copy=copy, errors=errors) - return MicroSeries( - converted_series, - weights=self.weights.copy() if copy else self.weights, - ) - - def __repr__(self) -> str: - return pd.DataFrame( - dict(value=self._values, weight=self.weights.values) - ).__repr__() - - -MicroSeries.SCALAR_FUNCTIONS = [ - fn - for fn in dir(MicroSeries) - if "_rtype" in dir(getattr(MicroSeries, fn)) - and getattr(getattr(MicroSeries, fn), "_rtype") == float -] -MicroSeries.VECTOR_FUNCTIONS = [ - fn - for fn in dir(MicroSeries) - if "_rtype" in dir(getattr(MicroSeries, fn)) - and getattr(getattr(MicroSeries, fn), "_rtype") == pd.Series -] -MicroSeries.AGNOSTIC_FUNCTIONS = ["quantile"] -MicroSeries.FUNCTIONS = sum( - [ - MicroSeries.SCALAR_FUNCTIONS, - MicroSeries.VECTOR_FUNCTIONS, - MicroSeries.AGNOSTIC_FUNCTIONS, - ], - [], -) - - -class MicroSeriesGroupBy(pd.core.groupby.generic.SeriesGroupBy): - def _init(self): - def _weighted_agg(name) -> Callable: - def via_micro_series(row, *args, **kwargs): - return getattr(MicroSeries(row.a, weights=row.w), name)(*args, **kwargs) - - fn = getattr(MicroSeries, name) - - @wraps(fn) - def _weighted_agg_fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: - arrays = self.apply(np.array) - weights = self.weights.apply(np.array) - df = pd.DataFrame(dict(a=arrays, w=weights)) - is_array = len(args) > 0 and hasattr(args[0], "__len__") - if ( - name in MicroSeries.SCALAR_FUNCTIONS - or name in MicroSeries.AGNOSTIC_FUNCTIONS - and not is_array - ): - result = df.agg( - lambda row: via_micro_series(row, *args, **kwargs), - axis=1, - ) - elif ( - name in MicroSeries.VECTOR_FUNCTIONS - or name in MicroSeries.AGNOSTIC_FUNCTIONS - and is_array - ): - if name in MicroSeries.AGNOSTIC_FUNCTIONS and not df.empty: - # Concatenate values without keys: concat rejects missing - # MultiIndex keys even when groupby(dropna=False) retains - # them. Reuse the grouping levels and codes so missing - # labels keep the same representation as scalar results. - results = [ - via_micro_series(row, *args, **kwargs) - for _, row in df.iterrows() - ] - result = pd.concat(results) - group_index = ( - df.index - if isinstance(df.index, pd.MultiIndex) - else pd.MultiIndex.from_arrays([df.index]) - ) - quantile_codes, quantile_levels = result.index.factorize( - sort=False - ) - result.index = pd.MultiIndex( - levels=[*group_index.levels, quantile_levels], - codes=[ - codes.repeat(len(results[0])) - for codes in group_index.codes - ] - + [quantile_codes], - names=[*df.index.names, result.index.name], - # Existing group codes are valid; checking would - # rewrite their retained missing labels to -1. - verify_integrity=False, - ) - return result - result = df.apply( - lambda row: via_micro_series(row, *args, **kwargs), - axis=1, - ) - return result.stack() - return result - - return _weighted_agg_fn - - for fn_name in MicroSeries.FUNCTIONS: - setattr(self, fn_name, _weighted_agg(fn_name)) diff --git a/build/lib/microdf/replication.py b/build/lib/microdf/replication.py deleted file mode 100644 index 0bb128e0..00000000 --- a/build/lib/microdf/replication.py +++ /dev/null @@ -1,123 +0,0 @@ -"""Variance estimation from replicate weights. - -Many survey products publish a set of replicate weight vectors alongside the -main weight. Recomputing a statistic once per replicate and measuring the -spread gives a variance estimate that requires no analytic formula, which is -what makes it usable for statistics such as the Gini coefficient or a -quantile where the analytic variance is awkward. - -The scale factor depends on how the replicates were constructed, so the -method must be named rather than guessed. - -Note that this is only valid for replicate weights as published with a -survey. Weights that have been calibrated or reweighted to external targets -no longer correspond to the original replication scheme, and applying these -estimators to them does not describe the variance of the resulting estimator. -""" - -from typing import Callable, Optional, Union - -import numpy as np -import pandas as pd - -# Scale applied to the sum of squared deviations from the full-sample -# estimate. R is the number of replicates. -METHOD_FACTORS = { - # Delete-a-group jackknife: (R - 1) / R. - "jackknife": lambda r: (r - 1) / r, - # Balanced repeated replication: 1 / R. - "brr": lambda r: 1 / r, - # Bootstrap replicates: 1 / R. - "bootstrap": lambda r: 1 / r, - # Successive difference replication, as used for the ACS and CPS: 4 / R. - "successive-difference": lambda r: 4 / r, -} - - -def _fay_factor(r: int, fay_k: float) -> float: - """Scale for Fay's variant of BRR, which perturbs rather than deletes.""" - if not 0 <= fay_k < 1: - raise ValueError(f"fay_k must be in [0, 1), got {fay_k}") - return 1 / (r * (1 - fay_k) ** 2) - - -def replicate_variance( - series, - statistic: Callable, - replicate_weights: Union[np.ndarray, pd.DataFrame], - method: str = "jackknife", - fay_k: Optional[float] = None, -) -> float: - """Variance of ``statistic`` estimated from replicate weights. - - :param series: A MicroSeries. Its own weights give the point estimate. - :param statistic: Callable taking a MicroSeries and returning a float, - for example ``lambda s: s.gini()``. - :param replicate_weights: Array or frame of shape ``(len(series), R)``. - :param method: One of ``jackknife``, ``brr``, ``bootstrap``, - ``successive-difference``, or ``fay`` (which requires ``fay_k``). - :param fay_k: Fay's perturbation constant, required when - ``method="fay"``. - :returns: The estimated variance of the statistic. - """ - from microdf.microseries import MicroSeries - - weights = np.asarray(replicate_weights, dtype=float) - if weights.ndim != 2: - raise ValueError( - f"replicate_weights must be 2-dimensional, got shape {weights.shape}" - ) - if weights.shape[0] != len(series): - raise ValueError( - f"replicate_weights has {weights.shape[0]} rows but the series has " - f"{len(series)}" - ) - - n_replicates = weights.shape[1] - if n_replicates < 2: - raise ValueError("At least two replicate weights are required") - - if method == "fay": - if fay_k is None: - raise ValueError("method='fay' requires fay_k") - factor = _fay_factor(n_replicates, fay_k) - elif method in METHOD_FACTORS: - if fay_k is not None: - raise ValueError("fay_k applies only to method='fay'") - factor = METHOD_FACTORS[method](n_replicates) - else: - known = ", ".join(sorted([*METHOD_FACTORS, "fay"])) - raise ValueError(f"Unknown method {method!r}; expected one of {known}") - - point = float(statistic(series)) - values = np.asarray(series, dtype=float) - index = series.index - - deviations = [] - for column in range(n_replicates): - replicate = MicroSeries( - values, weights=weights[:, column], index=index - ) - deviations.append(float(statistic(replicate)) - point) - - return factor * float(np.sum(np.square(deviations))) - - -def replicate_standard_error( - series, - statistic: Callable, - replicate_weights: Union[np.ndarray, pd.DataFrame], - method: str = "jackknife", - fay_k: Optional[float] = None, -) -> float: - """Standard error of ``statistic``, the square root of its variance. - - Takes the same arguments as :func:`replicate_variance`. - """ - return float( - np.sqrt( - replicate_variance( - series, statistic, replicate_weights, method, fay_k - ) - ) - ) diff --git a/build/lib/microdf/tests/conftest.py b/build/lib/microdf/tests/conftest.py deleted file mode 100644 index 44b0369e..00000000 --- a/build/lib/microdf/tests/conftest.py +++ /dev/null @@ -1,9 +0,0 @@ -import os - -import pytest - - -@pytest.fixture(scope="session") -def tests_path() -> str: - """""" - return os.path.abspath(os.path.dirname(__file__)) diff --git a/build/lib/microdf/tests/test_aggregation_errors.py b/build/lib/microdf/tests/test_aggregation_errors.py deleted file mode 100644 index bc8dd98e..00000000 --- a/build/lib/microdf/tests/test_aggregation_errors.py +++ /dev/null @@ -1,57 +0,0 @@ -import microdf as mdf -import numpy as np -import pandas as pd -import pytest - - -def test_aggregation_surfaces_real_errors(): - """A genuine argument error must raise, not be swallowed per column.""" - df = mdf.MicroDataFrame(pd.DataFrame({"x": [1, 2, 3]}), weights=[1, 1, 1]) - with pytest.raises(ValueError, match="Unknown negatives option"): - df.gini(negatives="bogus") - - -def test_aggregation_still_skips_non_numeric_columns(): - """Narrowing the guard must not change which columns aggregate.""" - df = mdf.MicroDataFrame( - pd.DataFrame( - { - "x": [1, 2, 3], - "s": ["a", "b", "c"], - "dt": pd.to_datetime(["2020-01-01"] * 3), - } - ), - weights=[1, 2, 3], - ) - assert list(df.sum().index) == ["x"] - assert df.sum()["x"] == 14 - assert list(df.mean().index) == ["x"] - - -def test_gini_shift_accepts_nonnegative_and_negative_columns(): - frame = mdf.MicroDataFrame( - {"positive": [1, 2, 3], "negative": [-1, 1, 3]}, weights=[1, 1, 1] - ) - # Shift leaves [1, 2, 3] unchanged and changes [-1, 1, 3] to [0, 2, 4]. - # The pairwise-difference Ginis are 2/9 and 4/9, respectively. - expected = pd.Series({"positive": 2 / 9, "negative": 4 / 9}) - pd.testing.assert_series_equal(frame.gini(negatives="shift"), expected) - - -@pytest.mark.parametrize("selected", [False, True]) -def test_grouped_gini_shift_accepts_positive_groups(selected): - frame = mdf.MicroDataFrame( - {"group": ["a", "a", "a", "b", "b", "b"], "value": [1, 2, 3, -1, 1, 3]}, - weights=[1, 1, 1, 1, 1, 1], - ) - grouped = frame.groupby("group") - if selected: - grouped = grouped[["value"]] - expected = pd.DataFrame( - {"value": [2 / 9, 4 / 9]}, index=pd.Index(["a", "b"], name="group") - ) - pd.testing.assert_frame_equal(grouped.gini(negatives="shift"), expected) - - -def test_gini_shift_accepts_empty_series(): - assert np.isnan(mdf.MicroSeries([], weights=[]).gini(negatives="shift")) diff --git a/build/lib/microdf/tests/test_dataframe_weight_storage.py b/build/lib/microdf/tests/test_dataframe_weight_storage.py deleted file mode 100644 index fc766096..00000000 --- a/build/lib/microdf/tests/test_dataframe_weight_storage.py +++ /dev/null @@ -1,61 +0,0 @@ -import warnings - -import numpy as np -import pandas as pd -import pytest - -import microdf as mdf - - -def test_weights_stay_a_series_after_nullify(): - """nullify_weights must leave weights as an index-aligned Series.""" - df = mdf.MicroDataFrame(pd.DataFrame({"x": [1, 2, 3]}), weights=[4, 5, 6]) - df.nullify_weights() - assert isinstance(df.weights, pd.Series) - assert list(df.weights.index) == list(df.index) - assert df.equals(df) - assert df.sum()["x"] == 6 - - -def test_weights_stay_a_series_after_set_weight_col(): - """The deprecated set_weight_col must also produce a Series.""" - df = mdf.MicroDataFrame(pd.DataFrame({"x": [1, 2, 3], "w": [1.0, 2.0, 3.0]})) - with warnings.catch_warnings(): - warnings.simplefilter("ignore", DeprecationWarning) - df.set_weight_col("w") - assert isinstance(df.weights, pd.Series) - assert df.weights_col == "w" - assert df.equals(df) - assert df.sum()["x"] == 14 - - -@pytest.mark.parametrize("dtype", ["int64", "float64"]) -def test_weight_column_edits_do_not_change_stored_weights(dtype): - """Selecting a weight column takes a copy of its current values.""" - df = mdf.MicroDataFrame( - {"x": [10, 20], "w": np.array([1, 2], dtype=dtype)}, - index=[10, 20], - ) - with pytest.warns(DeprecationWarning): - df.set_weight_col("w") - - df.loc[10, "w"] = 100 - - np.testing.assert_array_equal(df.weights, [1, 2]) - assert df.sum()["x"] == 10 * 1 + 20 * 2 - - -@pytest.mark.parametrize("dtype", ["int64", "float64"]) -def test_stored_weight_edits_do_not_change_weight_column(dtype): - """Changing stored weights leaves the source column values intact.""" - df = mdf.MicroDataFrame( - {"x": [10, 20], "w": np.array([1, 2], dtype=dtype)}, - index=[10, 20], - ) - with pytest.warns(DeprecationWarning): - df.set_weight_col("w") - - df.weights.iloc[0] = 100 - - np.testing.assert_array_equal(df["w"], [1, 2]) - assert df.sum()["x"] == 10 * 100 + 20 * 2 diff --git a/build/lib/microdf/tests/test_microseries_dataframe.py b/build/lib/microdf/tests/test_microseries_dataframe.py deleted file mode 100644 index a73e8bb5..00000000 --- a/build/lib/microdf/tests/test_microseries_dataframe.py +++ /dev/null @@ -1,817 +0,0 @@ -import warnings - -import numpy as np -import pandas as pd -import pytest - -import microdf as mdf -from microdf.microdataframe import MicroDataFrame -from microdf.microseries import MicroSeries - - -def test_df_init() -> None: - arr = np.array([0, 1, 1]) - w = np.array([3, 0, 9]) - df = mdf.MicroDataFrame({"a": arr}, weights=w) - assert df.a.mean() == np.average(arr, weights=w) - - df = mdf.MicroDataFrame() - df["a"] = arr - df.set_weights(w) - assert df.a.mean() == np.average(arr, weights=w) - - df = mdf.MicroDataFrame() - df["a"] = arr - df["w"] = w - df.set_weight_col("w") - assert df.a.mean() == np.average(arr, weights=w) - - # Test set_weights with string (column name) - df2 = mdf.MicroDataFrame() - df2["a"] = arr - df2["w"] = w - df2.set_weights("w") # Using string column name instead of set_weight_col - assert df2.a.mean() == np.average(arr, weights=w) - assert np.array_equal(df2.weights.values, w) - - -def test_handles_empty_index() -> None: - arr = np.array([0, 1, 1]) - w = np.array([3, 0, 9]) - df = mdf.MicroDataFrame({"a": arr}, weights=w) - - empty_index = pd.Index([]) - df[empty_index] # Implicit assert; checking for ValueError - - -def test_series_getitem() -> None: - arr = np.array([0, 1, 1]) - w = np.array([3, 0, 9]) - s = mdf.MicroSeries(arr, weights=w) - assert s[[1, 2]].sum() == np.sum(arr[[1, 2]] * w[[1, 2]]) - - assert s[1:3].sum() == np.sum(arr[1:3] * w[1:3]) - - -def test_sum() -> None: - arr = np.array([0, 1, 1]) - w = np.array([3, 0, 9]) - series = mdf.MicroSeries(arr, weights=w) - assert series.sum() == (arr * w).sum() - - arr = np.linspace(-20, 100, 100) - w = np.linspace(1, 3, 100) - series = mdf.MicroSeries(arr) - series.set_weights(w) - assert series.sum() == (arr * w).sum() - - # Verify that an error is thrown when passing weights of different size - # from the values. - w = np.linspace(1, 3, 101) - series = mdf.MicroSeries(arr) - try: - series.set_weights(w) - assert False - except Exception: - pass - - -def test_mean() -> None: - arr = np.array([3, 0, 2]) - w = np.array([4, 1, 1]) - series = mdf.MicroSeries(arr, weights=w) - assert series.mean() == np.average(arr, weights=w) - - arr = np.linspace(-20, 100, 100) - w = np.linspace(1, 3, 100) - series = mdf.MicroSeries(arr) - series.set_weights(w) - assert series.mean() == np.average(arr, weights=w) - - w = np.linspace(1, 3, 101) - series = mdf.MicroSeries(arr) - try: - series.set_weights(w) - assert False - except Exception: - pass - - -def test_mean_skipna() -> None: - # Test skipna=True (default) - should skip NaN values - arr = np.array([3.0, np.nan, 2.0]) - w = np.array([4.0, 1.0, 1.0]) - series = mdf.MicroSeries(arr, weights=w) - - # skipna=True should exclude NaN and its weight - expected = np.average([3.0, 2.0], weights=[4.0, 1.0]) - assert series.mean(skipna=True) == expected - assert series.mean() == expected # Default should be skipna=True - - # Test skipna=False - should return NaN if any value is NaN - assert np.isnan(series.mean(skipna=False)) - - # Test with all NaN values - arr_all_nan = np.array([np.nan, np.nan, np.nan]) - w_all_nan = np.array([1.0, 2.0, 3.0]) - series_all_nan = mdf.MicroSeries(arr_all_nan, weights=w_all_nan) - assert np.isnan(series_all_nan.mean(skipna=True)) - assert np.isnan(series_all_nan.mean(skipna=False)) - - # Test with no NaN values - skipna should not affect result - arr_no_nan = np.array([3.0, 5.0, 2.0]) - w_no_nan = np.array([4.0, 1.0, 1.0]) - series_no_nan = mdf.MicroSeries(arr_no_nan, weights=w_no_nan) - expected_no_nan = np.average(arr_no_nan, weights=w_no_nan) - assert series_no_nan.mean(skipna=True) == expected_no_nan - assert series_no_nan.mean(skipna=False) == expected_no_nan - - -def test_poverty_count() -> None: - arr = np.array([10000, 20000, 50000]) - w = np.array([1123, 1144, 2211]) - df = pd.DataFrame() - df["income"] = arr - df["threshold"] = 16000 - df = MicroDataFrame(df, weights=w) - assert df.poverty_count("income", "threshold") == w[0] - - -def test_median() -> None: - # 1, 2, 3, 4, *4*, 4, 5, 5, 5 - arr = np.array([1, 2, 3, 4, 5]) - w = np.array([1, 1, 1, 3, 3]) - series = mdf.MicroSeries(arr, weights=w) - assert series.median() == 4 - - -def test_weighted_quantile_skewed() -> None: - # 99% of the population has 0 income, 1% has 1M - # The median should be 0, not an interpolated value - series = mdf.MicroSeries([0, 1_000_000], weights=[99, 1]) - assert series.median() == 0 - assert series.quantile(0.5) == 0 - # 99th percentile is still 0 since exactly 99% have 0 - assert series.quantile(0.99) == 0 - # Only quantile > 0.99 gives 1M - assert series.quantile(1.0) == 1_000_000 - # Test multiple quantiles - result = series.quantile([0.1, 0.5, 0.99, 1.0]) - assert result[0.1] == 0 - assert result[0.5] == 0 - assert result[0.99] == 0 - assert result[1.0] == 1_000_000 - - -def test_weighted_quantile_boundaries() -> None: - # Test q=0 returns minimum, q=1 returns maximum - series = mdf.MicroSeries([10, 20, 30], weights=[1, 1, 1]) - assert series.quantile(0.0) == 10 - assert series.quantile(1.0) == 30 - - -def test_weighted_quantile_equal_weights() -> None: - # With equal weights, should match "replicated" interpretation - # Values: 1, 2, 3 each with weight 2 -> like [1,1,2,2,3,3] - series = mdf.MicroSeries([1, 2, 3], weights=[2, 2, 2]) - # cumsum_normalized = [2/6, 4/6, 6/6] = [0.333, 0.667, 1.0] - # median (0.5): smallest where cumsum >= 0.5 -> index 1 -> value 2 - assert series.median() == 2 - # 0.25 quantile: smallest where cumsum >= 0.25 -> index 0 -> value 1 - assert series.quantile(0.25) == 1 - # 0.75 quantile: smallest where cumsum >= 0.75 -> index 2 -> value 3 - assert series.quantile(0.75) == 3 - - -def test_weighted_quantile_unsorted_input() -> None: - # Ensure sorting works correctly - series = mdf.MicroSeries([30, 10, 20], weights=[1, 2, 1]) - # Sorted: values [10, 20, 30], weights [2, 1, 1] - # cumsum_normalized = [0.5, 0.75, 1.0] - assert series.quantile(0.0) == 10 - assert series.quantile(0.5) == 10 # cumsum[0]=0.5 >= 0.5 - assert series.quantile(0.6) == 20 # cumsum[1]=0.75 >= 0.6 - assert series.quantile(1.0) == 30 - - -def test_unweighted_groupby() -> None: - df = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}) - assert (df.groupby("x").z.sum().values == np.array([5.0, 6.0])).all() - - -def test_multiple_groupby() -> None: - df = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}) - assert (df.groupby(["x", "y"]).z.sum() == np.array([5, 6])).all() - - -def test_set_index() -> None: - d = mdf.MicroDataFrame(dict(x=[1, 2, 3]), weights=[4, 5, 6]) - assert d.x.__class__ == MicroSeries - d.index = [1, 2, 3] - assert d.x.__class__ == MicroSeries - - -def test_reset_index() -> None: - d = mdf.MicroDataFrame(dict(x=[1, 2, 3]), weights=[4, 5, 6]) - assert d.reset_index().__class__ == MicroDataFrame - - -def test_cumsum() -> None: - s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) - assert np.array_equal(s.cumsum().values, [4, 14, 32]) - - s = mdf.MicroSeries([2, 1, 3], weights=[5, 4, 6]) - assert np.array_equal(s.cumsum().values, [10, 14, 32]) - - s = mdf.MicroSeries([3, 1, 2], weights=[6, 4, 5]) - assert np.array_equal(s.cumsum().values, [18, 22, 32]) - - -def test_rank() -> None: - s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) - assert np.array_equal(s.rank().values, [4, 9, 15]) - - s = mdf.MicroSeries([3, 1, 2], weights=[6, 4, 5]) - assert np.array_equal(s.rank().values, [15, 4, 9]) - - s = mdf.MicroSeries([2, 1, 3], weights=[5, 4, 6]) - assert np.array_equal(s.rank().values, [9, 4, 15]) - - -def test_percentile_rank() -> None: - s = mdf.MicroSeries([4, 2, 3, 1], weights=[20, 40, 20, 20]) - assert np.array_equal(s.percentile_rank().values, [100, 60, 80, 20]) - - -def test_quartile_rank() -> None: - s = mdf.MicroSeries([4, 2, 3], weights=[25, 50, 25]) - assert np.array_equal(s.quartile_rank().values, [4, 2, 3]) - - -def test_quintile_rank() -> None: - s = mdf.MicroSeries([4, 2, 3], weights=[20, 60, 20]) - assert np.array_equal(s.quintile_rank().values, [5, 3, 4]) - - -def test_decile_rank() -> None: - s = mdf.MicroSeries( - [5, 4, 3, 2, 1, 6, 7, 8, 9], - weights=[10, 20, 10, 10, 10, 10, 10, 10, 10], - ) - assert np.array_equal(s.decile_rank().values, [6, 5, 3, 2, 1, 7, 8, 9, 10]) - - -def test_copy_equals() -> None: - d = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8]) - d_copy = d.copy() - d_copy_diff_weights = d_copy.copy() - d_copy_diff_weights.weights *= 2 - assert d.equals(d_copy) - assert not d.equals(d_copy_diff_weights) - # Same for a MicroSeries. - assert d.x.equals(d_copy.x) - assert not d.x.equals(d_copy_diff_weights.x) - - -def test_subset() -> None: - df = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4], "z": [5, 6]}, weights=[7, 8]) - df_no_z = mdf.MicroDataFrame({"x": [1, 2], "y": [3, 4]}, weights=[7, 8]) - assert df[["x", "y"]].equals(df_no_z) - df_no_z_diff_weights = df_no_z.copy() - df_no_z_diff_weights.weights += 1 - assert not df[["x", "y"]].equals(df_no_z_diff_weights) - - -def test_value_subset() -> None: - d = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [1, 2, 2]}, weights=[4, 5, 6]) - d2 = d[d.y > 1] - assert d2.y.shape == d2.weights.shape - - -def test_bitwise_ops_return_microseries() -> None: - s1 = mdf.MicroSeries([True, False, True], weights=[1, 2, 3]) - s2 = mdf.MicroSeries([False, False, True], weights=[1, 2, 3]) - and_result = s1 & s2 - or_result = s1 | s2 - assert isinstance(and_result, mdf.MicroSeries) - assert isinstance(or_result, mdf.MicroSeries) - expected_and = mdf.MicroSeries([False, False, True], weights=[1, 2, 3]) - expected_or = mdf.MicroSeries([True, False, True], weights=[1, 2, 3]) - assert and_result.equals(expected_and) - assert or_result.equals(expected_or) - - -def test_additional_ops_return_microseries() -> None: - s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) - radd = 1 + s - xor = s ^ mdf.MicroSeries([0, 1, 0], weights=[4, 5, 6]) - inv = ~mdf.MicroSeries([True, False], weights=[1, 1]) - assert isinstance(radd, mdf.MicroSeries) - assert isinstance(xor, mdf.MicroSeries) - assert isinstance(inv, mdf.MicroSeries) - - -def test_reset_index_inplace() -> None: - df = pd.DataFrame( - {"A": [1, 2, 3, 4], "B": [5, 6, 7, 8]}, index=["a", "b", "c", "d"] - ) - weights = np.array([0.1, 0.2, 0.3, 0.4]) - mdf = MicroDataFrame(df, weights=weights) - - # Test 1: reset_index with inplace=False (default) - mdf_copy = mdf.copy() - result = mdf_copy.reset_index() - assert list(mdf_copy.index) == ["a", "b", "c", "d"] - assert list(result.index) == [0, 1, 2, 3] - assert "index" in result.columns - assert list(result["index"]) == ["a", "b", "c", "d"] - np.testing.assert_array_equal(result.weights.values, weights) - - # Test 2: reset_index with inplace=True - mdf_copy = mdf.copy() - result = mdf_copy.reset_index(inplace=True) - assert result is None - assert list(mdf_copy.index) == [0, 1, 2, 3] - assert "index" in mdf_copy.columns - assert list(mdf_copy["index"]) == ["a", "b", "c", "d"] - np.testing.assert_array_equal(mdf_copy.weights.values, weights) - assert isinstance(mdf_copy["A"], MicroSeries) - assert isinstance(mdf_copy["B"], MicroSeries) - assert isinstance(mdf_copy["index"], MicroSeries) - - # Test 3: reset_index with drop=True - mdf_copy = mdf.copy() - mdf_copy.reset_index(drop=True, inplace=True) - assert list(mdf_copy.index) == [0, 1, 2, 3] - assert "index" not in mdf_copy.columns - assert list(mdf_copy.columns) == ["A", "B"] - np.testing.assert_array_equal(mdf_copy.weights.values, weights) - - # Test 4: Multi-level index - arrays = [["bar", "bar", "baz", "baz"], ["one", "two", "one", "two"]] - multi_index = pd.MultiIndex.from_arrays(arrays, names=["first", "second"]) - df_multi = pd.DataFrame({"A": [1, 2, 3, 4], "B": [5, 6, 7, 8]}, index=multi_index) - mdf_multi = MicroDataFrame(df_multi, weights=weights) - result = mdf_multi.reset_index(level="first") - assert "first" in result.columns - assert result.index.name == "second" - np.testing.assert_array_equal(result.weights.values, weights) - - # Reset all levels in place - mdf_multi.reset_index(inplace=True) - assert "first" in mdf_multi.columns - assert "second" in mdf_multi.columns - assert list(mdf_multi.index) == [0, 1, 2, 3] - np.testing.assert_array_equal(mdf_multi.weights.values, weights) - - -def test_loc_preserves_weights() -> None: - """Test that .loc[] returns MicroDataFrame with proper weights (issue - #265).""" - df = mdf.MicroDataFrame({"one": [1, 1, 1, 1, 1]}, weights=[10, 20, 30, 40, 50]) - - # Filter all rows (should get same weights) - filtered = df.loc[df.one == 1] - assert isinstance(filtered, MicroDataFrame) - assert filtered.one.sum() == 150.0 # Weighted sum - - # Partial filter - df2 = mdf.MicroDataFrame({"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50]) - subset = df2.loc[df2.x > 2] - assert isinstance(subset, MicroDataFrame) - assert subset.x.sum() == 500.0 # 3*30 + 4*40 + 5*50 = 500 - np.testing.assert_array_equal(subset.weights.values, [30.0, 40.0, 50.0]) - - -def test_iloc_preserves_weights() -> None: - """Test that .iloc[] returns MicroDataFrame with proper weights.""" - df = mdf.MicroDataFrame({"x": [1, 2, 3, 4, 5]}, weights=[10, 20, 30, 40, 50]) - - # Select rows by position - subset = df.iloc[2:5] - assert isinstance(subset, MicroDataFrame) - assert subset.x.sum() == 500.0 # 3*30 + 4*40 + 5*50 = 500 - np.testing.assert_array_equal(subset.weights.values, [30.0, 40.0, 50.0]) - - -def test_groupby_column_selection() -> None: - """Test that groupby column selection preserves weights (issue #193).""" - d = mdf.MicroDataFrame(dict(g=["a", "a", "b"], y=[1, 2, 3]), weights=[4, 5, 6]) - - # Test single column string selection - result_str = d.groupby("g")["y"].sum() - assert result_str["a"] == 14.0 # 1*4 + 2*5 = 14 - assert result_str["b"] == 18.0 # 3*6 = 18 - - # Test list column selection - result_list = d.groupby("g")[["y"]].sum() - assert result_list.loc["a", "y"] == 14.0 - assert result_list.loc["b", "y"] == 18.0 - - # Aggregated results should be plain DataFrame (no spurious weight column) - result_all = d.groupby("g").sum() - assert "weight" not in result_all.columns - assert list(result_all.columns) == ["y"] - - -def test_values_warns() -> None: - """Accessing .values on a MicroSeries should emit a UserWarning.""" - ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) - with warnings.catch_warnings(record=True) as w: - warnings.simplefilter("always") - _ = ms.values - assert len(w) == 1 - assert issubclass(w[0].category, UserWarning) - assert "weights" in str(w[0].message).lower() - - -def test_to_numpy_warns() -> None: - """Calling .to_numpy() on a MicroSeries should emit a UserWarning.""" - ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) - with warnings.catch_warnings(record=True) as w: - warnings.simplefilter("always") - _ = ms.to_numpy() - assert len(w) == 1 - assert issubclass(w[0].category, UserWarning) - assert "weights" in str(w[0].message).lower() - - -def test_mean_no_warning() -> None: - """Internal .values usage in .mean() should NOT emit a warning.""" - ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) - with warnings.catch_warnings(record=True) as w: - warnings.simplefilter("always") - _ = ms.mean() - user_warnings = [x for x in w if issubclass(x.category, UserWarning)] - assert len(user_warnings) == 0 - - -def test_sum_with_non_default_index() -> None: - """Weighted sum must not silently return 0 with a non-default index. - - Regression test for the bug where ``set_weights`` stored the weights Series - with a default ``RangeIndex`` regardless of ``self.index``. Element-wise - ops like ``self.multiply(self.weights)`` then aligned on label, producing - all-NaN and a silent ``0.0`` from ``.sum()`` while ``.mean()`` (which uses - a positional ndarray) stayed correct. - """ - # MicroSeries with custom integer index. - s = mdf.MicroSeries([1, 2, 3], index=[100, 200, 300], weights=[10, 20, 30]) - assert s.sum() == 140.0 - assert s.weights.index.tolist() == [100, 200, 300] - - # MicroDataFrame with custom integer index. - df = mdf.MicroDataFrame( - {"x": [1, 2, 3]}, index=[100, 200, 300], weights=[10, 20, 30] - ) - assert df.x.sum() == 140.0 - assert df.weights.index.tolist() == [100, 200, 300] - - # MicroDataFrame with string index + set_weights via column name. - df2 = mdf.MicroDataFrame( - {"x": [1, 2, 3], "w": [10, 20, 30]}, - index=["a", "b", "c"], - ) - df2.set_weights("w") - assert df2.x.sum() == 140.0 - assert df2.weights.index.tolist() == ["a", "b", "c"] - - # Passing a Series with its own index should position-align, not - # label-align, so sum does not depend on accidental index alignment. - df3 = mdf.MicroDataFrame({"x": [1, 2, 3]}, index=[100, 200, 300]) - df3.set_weights(pd.Series([10, 20, 30], index=[0, 1, 2])) - assert df3.x.sum() == 140.0 - - -def test_repr_no_warning() -> None: - """Internal .values usage in __repr__ should NOT emit a warning.""" - ms = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) - with warnings.catch_warnings(record=True) as w: - warnings.simplefilter("always") - _ = repr(ms) - user_warnings = [x for x in w if issubclass(x.category, UserWarning)] - assert len(user_warnings) == 0 - - -def test_drop_inplace_aligns_weights() -> None: - """Regression: ``drop(inplace=True)`` must keep weights in sync. - - Previously, ``weights_backup = self.weights.copy()`` was taken *before* - the drop, then reassigned back afterwards — so the weights vector - kept its original length and any subsequent weighted op either - raised a length-mismatch ValueError or silently returned 0. - """ - # Row drop inplace. - df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40]) - df.drop(index=[0, 1], inplace=True) - assert len(df) == len(df.weights) == 2 - assert df.x.sum() == 3 * 30 + 4 * 40 # 250 - - # Row drop non-inplace. - df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40]) - df2 = df.drop(index=[0, 1]) - assert len(df2) == len(df2.weights) == 2 - assert df2.x.sum() == 250 - # Original untouched. - assert len(df) == 4 - assert df.x.sum() == 1 * 10 + 2 * 20 + 3 * 30 + 4 * 40 - - # Column drop (weights length unchanged). - df = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [4, 5, 6]}, weights=[10, 20, 30]) - df.drop(columns=["y"], inplace=True) - assert list(df.columns) == ["x"] - assert df.x.sum() == 1 * 10 + 2 * 20 + 3 * 30 - - # String index row drop. - df = mdf.MicroDataFrame( - {"x": [1, 2, 3, 4]}, - index=["a", "b", "c", "d"], - weights=[10, 20, 30, 40], - ) - df.drop(index=["a", "b"], inplace=True) - assert df.x.sum() == 250 - - # labels= with default axis=0. - df = mdf.MicroDataFrame({"x": [1, 2, 3, 4]}, weights=[10, 20, 30, 40]) - df.drop(labels=[0, 1], inplace=True) - assert df.x.sum() == 250 - - -def test_merge_preserves_weights_per_surviving_row() -> None: - """Regression: merge must propagate weights onto the merged rows. - - Previously the implementation passed ``self.weights`` straight to the - MicroDataFrame constructor, so any merge that changed row count (inner - filtering, left-with-missing, many-to-many, outer) raised ``ValueError: - Length of weights (N) does not match length of DataFrame (M)``. - """ - # Inner join filters rows. - left = mdf.MicroDataFrame( - {"k": [1, 2, 3, 4], "v": [10, 20, 30, 40]}, weights=[1, 2, 3, 4] - ) - right = pd.DataFrame({"k": [2, 4], "w": [20, 40]}) - res = left.merge(right, on="k") - assert len(res) == 2 - # k=2 carries weight 2; k=4 carries weight 4. - np.testing.assert_array_equal(sorted(res.weights.values), [2.0, 4.0]) - assert res.v.sum() == 2 * 20 + 4 * 40 - - # Left join with missing from right. - left = mdf.MicroDataFrame( - {"k": [1, 2, 3, 4], "v": [10, 20, 30, 40]}, weights=[1, 2, 3, 4] - ) - right = pd.DataFrame({"k": [2, 4], "w": [100, 200]}) - res = left.merge(right, on="k", how="left") - assert len(res) == 4 - assert res.v.sum() == 300 # 1*10 + 2*20 + 3*30 + 4*40 - - # Many-to-many duplicates left rows; the same weight should ride - # along on each duplicate. - left = mdf.MicroDataFrame({"k": [1, 2], "v": [10, 20]}, weights=[5, 7]) - right = pd.DataFrame({"k": [1, 1, 2], "w": [100, 200, 300]}) - res = left.merge(right, on="k") - assert len(res) == 3 - # v=10 weighted 5 appears twice, v=20 weighted 7 appears once. - assert res.v.sum() == 10 * 5 + 10 * 5 + 20 * 7 - - # Outer join: right-only rows have no left weight. We default to 0 - # so they don't silently poison downstream aggregations. - left = mdf.MicroDataFrame({"k": [1, 2], "v": [10, 20]}, weights=[1, 2]) - right = pd.DataFrame({"k": [2, 3], "w": [20, 30]}) - res = left.merge(right, on="k", how="outer") - # k=1 -> weight 1, k=2 -> weight 2, k=3 -> weight 0 (right-only). - assert sorted(res.weights.values) == [0.0, 1.0, 2.0] - - -def test_groupby_does_not_leak_tmp_weights_column() -> None: - """Regression: groupby used to mutate self by adding __tmp_weights. - - Previously, ``MicroDataFrame.groupby`` set ``self["__tmp_weights"]`` and - never cleaned it up, so ``df.columns`` afterwards included the weight - column and any later ``df.sum()`` or iteration over columns picked it up as - data. - """ - df = mdf.MicroDataFrame({"g": ["a", "a", "b"], "v": [1, 2, 3]}, weights=[1, 2, 3]) - original_cols = list(df.columns) - _ = df.groupby("g").sum() - assert list(df.columns) == original_cols - assert "__tmp_weights" not in df.columns - - # Groupby by a list of columns should also not leak. - df2 = mdf.MicroDataFrame( - {"g1": ["a", "a", "b"], "g2": [1, 1, 2], "v": [1, 2, 3]}, - weights=[1, 2, 3], - ) - orig2 = list(df2.columns) - _ = df2.groupby(["g1", "g2"]).v.sum() - assert list(df2.columns) == orig2 - - # Weighted aggregation is still correct after the fix. - result = df.groupby("g").v.sum() - assert result["a"] == 1 * 1 + 2 * 2 - assert result["b"] == 3 * 3 - - -def test_quantile_skips_zero_weight_rows() -> None: - """Regression: quantile(0) shouldn't pick a zero-weight element. - - Previously, ``np.searchsorted(cumsum_norm, 0, side='left')`` returned 0 - even when that first sorted element had zero weight, so ``MicroSeries([10, - 20, 30], weights=[0, 1, 1]).quantile(0)`` returned 10 instead of 20. The - fix drops zero-weight rows before computing the CDF. - """ - s = mdf.MicroSeries([10, 20, 30], weights=[0, 1, 1]) - assert s.quantile(0.0) == 20 - assert s.quantile(0.5) == 20 - assert s.quantile(1.0) == 30 - - # Internal plateau of zero weight. - s = mdf.MicroSeries([10, 20, 30, 40], weights=[1, 0, 1, 1]) - assert s.quantile(0.0) == 10 - # Post-filter values [10, 30, 40] with equal weights -> cum=[.33,.67,1]. - # 0.4 -> smallest cum >= 0.4 is index 1 -> value 30. - assert s.quantile(0.4) == 30 - # The zero-weight value (20) should never be selected. - for q in np.linspace(0, 1, 21): - assert s.quantile(q) != 20 - - # All zero weights -> NaN (defined behaviour). - s = mdf.MicroSeries([10, 20, 30], weights=[0, 0, 0]) - assert np.isnan(s.quantile(0.5)) - - -def test_top_x_pct_share_handles_ties_and_edges() -> None: - """Regression: top_x_pct_share double-counted threshold ties. - - Old implementation: ``self[self >= threshold].sum() / self.sum()``. - With constant values every call returned 1.0 regardless of the - requested top percent; ``top_x_pct_share(0)`` returned the share of - the max bucket instead of 0. - """ - # Constant values: the top p% should hold exactly p% of the total. - for p in [0.0, 0.01, 0.1, 0.5, 1.0]: - got = mdf.MicroSeries([5] * 10, weights=[1] * 10).top_x_pct_share(p) - assert np.isclose(got, p), f"top={p}, got {got}" - - # Non-constant, equal weights. - s = mdf.MicroSeries(list(range(1, 11)), weights=[1] * 10) - # Sum 1..10 = 55. Top 10% = top 1 row = 10 -> 10/55. - assert np.isclose(s.top_x_pct_share(0.1), 10 / 55) - # Top 50% = rows 6..10 -> 40/55. - assert np.isclose(s.top_x_pct_share(0.5), 40 / 55) - # Top 0% = 0, top 100% = 1. - assert s.top_x_pct_share(0.0) == 0.0 - assert s.top_x_pct_share(1.0) == 1.0 - - # Bottom share complements the top share. - assert np.isclose(s.bottom_x_pct_share(0.1), 1 - s.top_x_pct_share(0.9)) - - # Ties with unequal totals. - s_ties = mdf.MicroSeries([1, 1, 10, 10], weights=[1, 1, 1, 1]) - # Top 50% = the two 10s -> 20/22. - assert np.isclose(s_ties.top_x_pct_share(0.5), 20 / 22) - - # Downstream helpers still work. - assert np.isclose(s_ties.top_10_pct_share(), s_ties.top_x_pct_share(0.1)) - assert np.isclose(s_ties.top_50_pct_share(), s_ties.top_x_pct_share(0.5)) - - -def test_gini_negatives_option_applied() -> None: - """Regression: gini(negatives=...) was silently ignored. - - Both branches of the old implementation sorted ``self`` directly rather - than the local ``x`` that was mutated by the ``negatives`` option, so - ``negatives='zero'`` and ``negatives='shift'`` did nothing. - """ - s = mdf.MicroSeries([-5, 0, 10], weights=[1, 1, 1]) - - # Leaving negatives in place now warns. - with warnings.catch_warnings(record=True) as w: - warnings.simplefilter("always") - _ = s.gini() - user_warnings = [x for x in w if issubclass(x.category, UserWarning)] - assert len(user_warnings) == 1 - assert "negative" in str(user_warnings[0].message).lower() - - # 'zero' clamps negatives. Values become [0, 0, 10] with equal - # weights; closed-form gini = 2/3. - assert np.isclose(s.gini(negatives="zero"), 2 / 3) - - # 'shift' adds |min|. Values become [0, 5, 15]; Gini in [0, 1]. - shifted = s.gini(negatives="shift") - assert 0 <= shifted <= 1 - - # All-zero short-circuits to 0 instead of nan/RuntimeWarning. - assert mdf.MicroSeries([0, 0, 0], weights=[1, 2, 3]).gini() == 0.0 - - # Invalid negatives arg raises. - with pytest.raises(ValueError): - mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1]).gini(negatives="bogus") - - -def test_std_var_are_weighted() -> None: - """Regression: std/var used to silently fall through to pandas. - - The old implementation had no override, so a MicroSeries with very uneven - weights returned the unweighted 1.0. Now std and var treat the weights as - frequency counts, matching numpy on the replicated sample. - """ - s = mdf.MicroSeries([1, 2, 3], weights=[100, 1, 1]) - # Unweighted would be 1.0. Weighted std pulls toward the heavy row. - assert s.std() < 1.0 - assert s.var() < 1.0 - - # Integer-replication equivalence. - s = mdf.MicroSeries([1, 2, 3], weights=[2, 3, 1]) - rep = np.array([1, 1, 2, 2, 2, 3]) - assert np.isclose(s.std(), np.std(rep, ddof=1)) - assert np.isclose(s.var(), np.var(rep, ddof=1)) - assert np.isclose(s.var(ddof=0), np.var(rep, ddof=0)) - - # NaN handling. - s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[2, 3, 1]) - assert not np.isnan(s.std()) - assert np.isnan(s.std(skipna=False)) - - # DataFrame dispatch: df.std() / df.var() now return weighted stats. - df = mdf.MicroDataFrame({"x": [1, 2, 3], "y": [10, 20, 30]}, weights=[2, 3, 1]) - np.testing.assert_allclose( - df.std().values, - [ - np.std(rep, ddof=1), - np.std(np.array([10, 10, 20, 20, 20, 30]), ddof=1), - ], - ) - - -def test_cov_corr_warn_when_fallthrough() -> None: - """Regression: cov/corr silently returned unweighted pandas values. - - They still fall through to pandas (a weighted impl is a separate issue) but - now emit a UserWarning so callers aren't misled. - """ - s1 = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1]) - s2 = mdf.MicroSeries([2, 4, 6], weights=[1, 1, 1]) - - with warnings.catch_warnings(record=True) as w: - warnings.simplefilter("always") - _ = s1.cov(s2) - msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)] - assert any("unweighted" in m.lower() for m in msgs) - - with warnings.catch_warnings(record=True) as w: - warnings.simplefilter("always") - _ = s1.corr(s2) - msgs = [str(x.message) for x in w if issubclass(x.category, UserWarning)] - assert any("unweighted" in m.lower() for m in msgs) - - -def test_count_skips_nan_by_default() -> None: - """Regression: ``count()`` included NaN-row weight, contrary to pandas. - - Pandas ``Series.count`` skips NaN; MicroSeries returned the full weight sum - regardless. The fix matches pandas semantics and adds a ``skipna`` kwarg so - callers can opt out. - """ - s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[10, 20, 30]) - assert s.count() == 40.0 - assert s.count(skipna=True) == 40.0 - assert s.count(skipna=False) == 60.0 - - # No NaN: skipna is a no-op. - assert mdf.MicroSeries([1, 2, 3], weights=[2, 3, 4]).count() == 9.0 - - # All NaN: count skips everything. - all_nan = mdf.MicroSeries([np.nan] * 3, weights=[1, 2, 3]) - assert all_nan.count() == 0.0 - assert all_nan.count(skipna=False) == 6.0 - - -def test_rank_ties_share_bucket() -> None: - """Regression: rank used to assign ties to different ranks/buckets. - - Previously ``rank`` returned the running cumulative weight in sort order, - so every row — tied or not — got a distinct value. As a result - ``MicroSeries([5]*5, weights=[1]*5).decile_rank()`` returned ``[2, 4, 6, 8, - 10]`` rather than all 10. With max-rank semantics, tied values share the - cumulative weight at the end of their tie group, so bucketing is stable - under ties. - """ - # All tied: every element lands in the top decile. - s = mdf.MicroSeries([5] * 5, weights=[1] * 5) - np.testing.assert_array_equal(s.rank().values, [5, 5, 5, 5, 5]) - np.testing.assert_array_equal(s.decile_rank().values, [10] * 5) - np.testing.assert_array_equal(s.quintile_rank().values, [5] * 5) - - # Partial ties. - s = mdf.MicroSeries([1, 2, 2, 3], weights=[1, 1, 1, 1]) - np.testing.assert_array_equal(s.rank().values, [1, 3, 3, 4]) - - # pct=True normalizes to (0, 1] and still shares ranks on ties. - s = mdf.MicroSeries([5] * 4, weights=[1] * 4) - np.testing.assert_allclose(s.rank(pct=True).values, [1.0, 1.0, 1.0, 1.0]) - - # Non-ties still match the old cumulative-weight behaviour, so the - # existing ``test_rank`` expectations hold. - s = mdf.MicroSeries([1, 2, 3], weights=[4, 5, 6]) - np.testing.assert_array_equal(s.rank().values, [4, 9, 15]) diff --git a/build/lib/microdf/tests/test_nullify_weights_index.py b/build/lib/microdf/tests/test_nullify_weights_index.py deleted file mode 100644 index 8d4f9c70..00000000 --- a/build/lib/microdf/tests/test_nullify_weights_index.py +++ /dev/null @@ -1,10 +0,0 @@ -import microdf as mdf - - -def test_nullify_weights_non_default_index(): - """nullify_weights must align to the index, not a fresh RangeIndex.""" - s = mdf.MicroSeries([1, 2, 3], index=[10, 11, 12], weights=[1, 2, 3]) - s.nullify_weights() - assert s.sum() == 6 - assert s.mean() == 2 - assert list(s.weights.index) == [10, 11, 12] diff --git a/build/lib/microdf/tests/test_pandas3_compatibility.py b/build/lib/microdf/tests/test_pandas3_compatibility.py deleted file mode 100644 index 1413780d..00000000 --- a/build/lib/microdf/tests/test_pandas3_compatibility.py +++ /dev/null @@ -1,242 +0,0 @@ -"""Tests for pandas 3.0.0 compatibility in microdf. - -These tests verify that microdf works correctly with pandas 3.0.0, -which introduces: -1. PyArrow-backed strings as default (StringDtype) -2. Copy-on-Write by default -3. Changes to how Series subclasses are handled -""" - -import numpy as np -import pandas as pd - -from microdf.microdataframe import MicroDataFrame -from microdf.microseries import MicroSeries - - -class TestMicroSeriesSubclassPreservation: - """Test that MicroSeries subclass is preserved across operations.""" - - def test_microseries_set_weights_after_creation(self): - """Ensure set_weights works on MicroSeries. - - This is the error reported in pandas 3: - AttributeError: 'Series' object has no attribute 'set_weights' - """ - ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 1.0, 1.0])) - assert hasattr(ms, "set_weights") - assert hasattr(ms, "weights") - - # Should be able to call set_weights - ms.set_weights(np.array([2.0, 2.0, 2.0])) - assert np.allclose(ms.weights, [2.0, 2.0, 2.0]) - - def test_microseries_preserved_after_arithmetic(self): - """Arithmetic operations should return MicroSeries, not plain - Series.""" - ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) - - # Addition - result = ms + 1 - assert isinstance(result, MicroSeries), ( - f"Got {type(result)} instead of MicroSeries" - ) - assert hasattr(result, "weights") - assert hasattr(result, "set_weights") - - # Multiplication - result = ms * 2 - assert isinstance(result, MicroSeries), ( - f"Got {type(result)} instead of MicroSeries" - ) - - # Division - result = ms / 2 - assert isinstance(result, MicroSeries), ( - f"Got {type(result)} instead of MicroSeries" - ) - - def test_microseries_preserved_after_comparison(self): - """Comparison operations should return MicroSeries, not plain - Series.""" - ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) - - # Greater than - result = ms > 1 - assert isinstance(result, MicroSeries), ( - f"Got {type(result)} instead of MicroSeries" - ) - assert hasattr(result, "weights") - - # Less than - result = ms < 3 - assert isinstance(result, MicroSeries), ( - f"Got {type(result)} instead of MicroSeries" - ) - - def test_microseries_preserved_after_indexing(self): - """Indexing operations should return MicroSeries, not plain Series.""" - ms = MicroSeries([1, 2, 3, 4, 5], weights=np.array([1.0, 2.0, 3.0, 4.0, 5.0])) - - # Boolean indexing - result = ms[ms > 2] - assert isinstance(result, MicroSeries), ( - f"Got {type(result)} instead of MicroSeries" - ) - assert hasattr(result, "weights") - - # Slice indexing - result = ms[1:3] - assert isinstance(result, MicroSeries), ( - f"Got {type(result)} instead of MicroSeries" - ) - - -class TestMicroDataFrameSubclassPreservation: - """Test that MicroDataFrame column access returns MicroSeries.""" - - def test_microdataframe_column_returns_microseries(self): - """Accessing a column from MicroDataFrame should return MicroSeries.""" - mdf = MicroDataFrame( - {"a": [1, 2, 3], "b": [4, 5, 6]}, weights=np.array([1.0, 2.0, 3.0]) - ) - - # Column access - col = mdf["a"] - assert isinstance(col, MicroSeries), f"Got {type(col)} instead of MicroSeries" - assert hasattr(col, "weights") - assert hasattr(col, "set_weights") - - def test_microdataframe_operations_preserve_type(self): - """Operations on MicroDataFrame columns should preserve MicroSeries - type.""" - mdf = MicroDataFrame( - {"a": [1, 2, 3], "b": [4, 5, 6]}, weights=np.array([1.0, 2.0, 3.0]) - ) - - # Column operations - result = mdf["a"] + mdf["b"] - assert isinstance(result, MicroSeries), ( - f"Got {type(result)} instead of MicroSeries" - ) - assert hasattr(result, "weights") - - -class TestStringDtypeHandling: - """Test that MicroSeries/MicroDataFrame handle pandas 3 string dtypes.""" - - def test_microseries_with_string_data(self): - """MicroSeries should work with string data in pandas 3.""" - # Create with string data - ms = MicroSeries(["a", "b", "c"], weights=np.array([1.0, 2.0, 3.0])) - assert len(ms) == 3 - assert hasattr(ms, "weights") - - def test_microdataframe_with_string_columns(self): - """MicroDataFrame should work with string columns in pandas 3.""" - mdf = MicroDataFrame( - {"names": ["alice", "bob", "charlie"], "values": [1, 2, 3]}, - weights=np.array([1.0, 2.0, 3.0]), - ) - assert len(mdf) == 3 - - # String column access should still work - names = mdf["names"] - assert len(names) == 3 - - -class TestWeightedOperationsWithPandas3: - """Test that weighted operations work correctly with pandas 3.""" - - def test_weighted_sum(self): - """Weighted sum should work correctly.""" - ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) - # Weighted sum: 1*1 + 2*2 + 3*3 = 1 + 4 + 9 = 14 - assert ms.sum() == 14 - - def test_weighted_mean(self): - """Weighted mean should work correctly.""" - ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) - # Weighted mean: (1*1 + 2*2 + 3*3) / (1 + 2 + 3) = 14 / 6 ≈ 2.333 - assert np.isclose(ms.mean(), 14 / 6) - - def test_weighted_count(self): - """Weighted count should return sum of weights.""" - ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) - assert ms.count() == 6.0 - - -class TestCopyOnWriteCompatibility: - """Test compatibility with pandas 3 Copy-on-Write.""" - - def test_microseries_copy_independent(self): - """Copying a MicroSeries should create an independent copy.""" - ms = MicroSeries([1, 2, 3], weights=np.array([1.0, 2.0, 3.0])) - ms_copy = ms.copy() - - # Modify original - ms.set_weights(np.array([4.0, 5.0, 6.0])) - - # Copy should be unchanged - assert np.allclose(ms_copy.weights, [1.0, 2.0, 3.0]) - - def test_microdataframe_copy_independent(self): - """Copying a MicroDataFrame should create an independent copy.""" - mdf = MicroDataFrame({"a": [1, 2, 3]}, weights=np.array([1.0, 2.0, 3.0])) - mdf_copy = mdf.copy() - - # Modify original - mdf.set_weights(np.array([4.0, 5.0, 6.0])) - - # Copy should be unchanged - assert np.allclose(mdf_copy.weights, [1.0, 2.0, 3.0]) - - def test_column_set_weights_after_access_regression(self): - """Regression test for pandas 3.0 CoW compatibility. - - In pandas 3.0 with Copy-on-Write, modifying column.__class__ doesn't - persist because each access returns a copy. This test verifies the fix - that wraps columns as MicroSeries on access in __getitem__. - """ - mdf = MicroDataFrame( - {"income": [10000, 20000, 30000]}, - weights=np.array([1.0, 2.0, 3.0]), - ) - - # This was the exact error that occurred: - # AttributeError: 'Series' object has no attribute 'set_weights' - col = mdf["income"] - col.set_weights(np.array([4.0, 5.0, 6.0])) # Would fail before fix - - # Verify the new weights took effect - assert np.allclose(col.weights, [4.0, 5.0, 6.0]) - - -class TestGroupByWithPandas3: - """Test groupby operations with pandas 3.""" - - def test_microseries_groupby_preserves_weights(self): - """GroupBy operations should preserve weights.""" - ms = MicroSeries([1, 2, 3, 4], weights=np.array([1.0, 2.0, 3.0, 4.0])) - groups = pd.Series(["a", "a", "b", "b"]) - - gb = ms.groupby(groups) - # Should be able to call weighted operations - result = gb.sum() - # Group a: 1*1 + 2*2 = 5 - # Group b: 3*3 + 4*4 = 25 - assert result["a"] == 5 - assert result["b"] == 25 - - def test_microdataframe_groupby_preserves_weights(self): - """MicroDataFrame groupby should preserve weights on columns.""" - mdf = MicroDataFrame( - {"group": ["a", "a", "b", "b"], "value": [1, 2, 3, 4]}, - weights=np.array([1.0, 2.0, 3.0, 4.0]), - ) - - gb = mdf.groupby("group") - result = gb.sum() - - # Check that weighted sum was computed - assert "value" in result.columns diff --git a/build/lib/microdf/tests/test_quantile_missing_values.py b/build/lib/microdf/tests/test_quantile_missing_values.py deleted file mode 100644 index 94e5677e..00000000 --- a/build/lib/microdf/tests/test_quantile_missing_values.py +++ /dev/null @@ -1,134 +0,0 @@ -import microdf as mdf -import numpy as np -import pandas as pd -import pytest - - -def test_quantile_skips_nan(): - """NaN weight must not inflate the cumulative distribution. - - Dropping a NaN row should give the same answer as never having had - it: the inverse-CDF quantile of [1, nan, 3] equals that of [1, 3]. - """ - with_nan = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[1, 1, 1]) - without_nan = mdf.MicroSeries([1.0, 3.0], weights=[1, 1]) - assert with_nan.median() == without_nan.median() - assert with_nan.quantile(0.5) == without_nan.quantile(0.5) - - q = [0.25, 0.5, 0.75] - np.testing.assert_array_equal( - mdf.MicroSeries([1.0, np.nan, 3.0, 5.0], weights=[1, 1, 1, 1]).quantile(q), - mdf.MicroSeries([1.0, 3.0, 5.0], weights=[1, 1, 1]).quantile(q), - ) - - -def test_quantile_skipna_false_propagates_nan(): - """Skipna=False returns NaN when any value is NaN, like mean/var.""" - s = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[1, 1, 1]) - assert np.isnan(s.quantile(0.5, skipna=False)) - assert np.isnan(s.median(skipna=False)) - assert s.quantile([0.25, 0.75], skipna=False).isna().all() - - -def test_quantile_all_nan_returns_nan(): - s = mdf.MicroSeries([np.nan, np.nan], weights=[1, 1]) - assert np.isnan(s.median()) - - -@pytest.mark.parametrize("skipna", [True, False]) -@pytest.mark.parametrize("q", [-0.1, 1.1, [0.5, 1.1]]) -def test_quantile_validates_bounds_with_missing_values(q, skipna): - series = mdf.MicroSeries([1.0, np.nan, 3.0], weights=[1, 7, 1]) - with pytest.raises(AssertionError, match="quantiles should be in"): - series.quantile(q, skipna=skipna) - - -@pytest.mark.parametrize("skipna", [True, False]) -@pytest.mark.parametrize("multiple_keys", [False, True]) -def test_grouped_quantiles_preserve_missing_groups(skipna, multiple_keys): - series = mdf.MicroSeries( - [1.0, np.nan, 3.0, 5.0, np.nan, np.nan], weights=[1, 1, 1, 1, 1, 1] - ) - groups = ["a", "a", "b", "b", "c", "c"] - keys = [groups, [1, 1, 2, 2, 3, 3]] if multiple_keys else groups - grouped = series.groupby(keys) - quantiles = [0.25, 0.75] - result = grouped.quantile(quantiles, skipna=skipna) - # Scalar calls retain every group. Vector calls must retain the same - # groups, including the partial-NaN and all-NaN groups. - for quantile in quantiles: - pd.testing.assert_series_equal( - result.xs(quantile, level=-1), - grouped.quantile(quantile, skipna=skipna), - ) - first_group = 1.0 if skipna else np.nan - np.testing.assert_allclose( - result.to_numpy(), - [first_group, first_group, 3.0, 5.0, np.nan, np.nan], - equal_nan=True, - ) - - -def test_grouped_quantiles_preserve_repeated_requests(): - series = mdf.MicroSeries([1.0, np.nan, 3.0, 5.0], weights=[1, 1, 1, 1]) - result = series.groupby(["a", "a", "b", "b"]).quantile([0.5, 0.5], skipna=False) - assert result.index.tolist() == [("a", 0.5), ("a", 0.5), ("b", 0.5), ("b", 0.5)] - np.testing.assert_allclose( - result.to_numpy(), [np.nan, np.nan, 3.0, 3.0], equal_nan=True - ) - - -@pytest.mark.parametrize("quantiles", [[0.75, 0.25], [0.5, 0.5], []]) -@pytest.mark.parametrize("skipna", [True, False]) -@pytest.mark.parametrize("sort", [True, False]) -def test_grouped_quantiles_preserve_missing_multiple_keys(quantiles, skipna, sort): - """Missing group keys survive alongside missing values and repeated q.""" - frame = mdf.MicroDataFrame( - { - "region": ["north", "north", None, "south", "south"], - "year": [2024, 2024, 2024, np.nan, 2025], - "income": [10.0, np.nan, 20.0, 30.0, 40.0], - }, - weights=[1, 4, 2, 3, 1], - ) - grouped = frame.groupby(["region", "year"], dropna=False, sort=sort)["income"] - result = grouped.quantile(quantiles, skipna=skipna) - - # Each retained group has one nonmissing value. With skipna=False, - # the north group is NaN because it also contains a missing value. - north = 10.0 if skipna else np.nan - groups = [("north", 2024.0, north)] - if sort: - groups += [ - ("south", 2025.0, 40.0), - ("south", np.nan, 30.0), - (np.nan, 2024.0, 20.0), - ] - else: - groups += [ - (np.nan, 2024.0, 20.0), - ("south", np.nan, 30.0), - ("south", 2025.0, 40.0), - ] - expected_index = pd.MultiIndex.from_tuples( - [(region, year, q) for region, year, _ in groups for q in quantiles], - names=["region", "year", None], - ) - expected_values = [value for _, _, value in groups for _ in quantiles] - assert result.index.names == expected_index.names - if quantiles: - for level in range(3): - pd.testing.assert_index_equal( - result.index.get_level_values(level), - expected_index.get_level_values(level), - ) - assert result.index.nlevels == 3 - np.testing.assert_allclose(result.to_numpy(), expected_values, equal_nan=True) - for q in set(quantiles): - if quantiles.count(q) == 1: - selected = result.xs(q, level=-1) - scalar = grouped.quantile(q, skipna=skipna) - assert selected.index.equals(scalar.index) - np.testing.assert_allclose( - selected.to_numpy(), scalar.to_numpy(), equal_nan=True - ) diff --git a/build/lib/microdf/tests/test_replication.py b/build/lib/microdf/tests/test_replication.py deleted file mode 100644 index 28b5bbb0..00000000 --- a/build/lib/microdf/tests/test_replication.py +++ /dev/null @@ -1,110 +0,0 @@ -"""Variance estimation from replicate weights. - -Recomputing a statistic once per replicate weight vector and measuring the -spread gives a variance estimate for statistics whose analytic variance is -awkward, such as the Gini coefficient or a quantile. -""" - -import numpy as np -import pytest - -from microdf import MicroSeries, replicate_standard_error, replicate_variance - - -@pytest.fixture -def series_and_replicates(): - rng = np.random.default_rng(0) - values = rng.lognormal(mean=10, sigma=1.0, size=500) - weights = np.full(500, 40.0) - replicates = weights[:, None] * rng.poisson(1.0, size=(500, 200)) - return MicroSeries(values, weights=weights), replicates - - -def test_matches_the_analytic_standard_error_of_a_weighted_mean( - series_and_replicates, -): - """The mean has a closed form, so it is the case we can check exactly.""" - series, replicates = series_and_replicates - - replicate_se = replicate_standard_error( - series, lambda s: s.mean(), replicates, method="bootstrap" - ) - - values = np.asarray(series) - analytic_se = np.sqrt(np.var(values, ddof=1) / len(values)) - - assert replicate_se == pytest.approx(analytic_se, rel=0.15) - - -def test_works_for_statistics_with_no_analytic_variance( - series_and_replicates, -): - """The point of the method: Gini and quantiles come out like anything else.""" - series, replicates = series_and_replicates - - for statistic in (lambda s: s.gini(), lambda s: s.median()): - se = replicate_standard_error( - series, statistic, replicates, method="bootstrap" - ) - assert se > 0 - assert np.isfinite(se) - - -@pytest.mark.parametrize( - "method,expected_factor", - [ - ("jackknife", 199 / 200), - ("brr", 1 / 200), - ("bootstrap", 1 / 200), - ("successive-difference", 4 / 200), - ], -) -def test_each_method_applies_its_own_scale( - series_and_replicates, method, expected_factor -): - """The scale factor is what distinguishes the replication schemes.""" - series, replicates = series_and_replicates - - variance = replicate_variance( - series, lambda s: s.mean(), replicates, method=method - ) - reference = replicate_variance( - series, lambda s: s.mean(), replicates, method="brr" - ) - - assert variance == pytest.approx(reference * expected_factor * 200, rel=1e-9) - - -def test_fay_requires_and_uses_its_constant(series_and_replicates): - series, replicates = series_and_replicates - - with pytest.raises(ValueError, match="requires fay_k"): - replicate_variance(series, lambda s: s.mean(), replicates, method="fay") - - fay = replicate_variance( - series, lambda s: s.mean(), replicates, method="fay", fay_k=0.5 - ) - brr = replicate_variance( - series, lambda s: s.mean(), replicates, method="brr" - ) - - # 1 / (R (1 - k)^2) against 1 / R, so a factor of four at k = 0.5. - assert fay == pytest.approx(brr * 4, rel=1e-9) - - -def test_rejects_input_that_cannot_be_right(series_and_replicates): - series, replicates = series_and_replicates - - with pytest.raises(ValueError, match="2-dimensional"): - replicate_variance(series, lambda s: s.mean(), np.ones(500)) - - with pytest.raises(ValueError, match="rows but the series has"): - replicate_variance(series, lambda s: s.mean(), np.ones((499, 10))) - - with pytest.raises(ValueError, match="At least two"): - replicate_variance(series, lambda s: s.mean(), np.ones((500, 1))) - - with pytest.raises(ValueError, match="Unknown method"): - replicate_variance( - series, lambda s: s.mean(), replicates, method="nonsense" - ) diff --git a/build/lib/microdf/tests/test_serialization.py b/build/lib/microdf/tests/test_serialization.py deleted file mode 100644 index 59d4bfdc..00000000 --- a/build/lib/microdf/tests/test_serialization.py +++ /dev/null @@ -1,143 +0,0 @@ -import copy -import io -import pickle - -import pandas as pd -import pytest - -import microdf as mdf - - -def test_microseries_survives_pickling(): - """Weights must survive a pickle round-trip.""" - import pickle - - s = mdf.MicroSeries([1, 2, 3], index=[7, 8, 9], weights=[1, 2, 3]) - restored = pickle.loads(pickle.dumps(s)) - assert isinstance(restored, mdf.MicroSeries) - assert restored.sum() == 14 - assert list(restored.weights) == [1.0, 2.0, 3.0] - - -def test_microdataframe_survives_pickling(): - """Weights and the weighted aggregations must survive a round-trip.""" - import pickle - - df = mdf.MicroDataFrame( - pd.DataFrame({"x": [1, 2, 3]}, index=[7, 8, 9]), weights=[1, 2, 3] - ) - restored = pickle.loads(pickle.dumps(df)) - assert isinstance(restored, mdf.MicroDataFrame) - assert isinstance(restored.weights, pd.Series) - # Would be 6 (unweighted) if the aggregation overrides were not - # reinstalled after unpickling. - assert restored.sum()["x"] == 14 - - -def test_deepcopy_preserves_weights(): - df = mdf.MicroDataFrame(pd.DataFrame({"x": [1, 2, 3]}), weights=[1, 2, 3]) - assert copy.deepcopy(df).sum()["x"] == 14 - s = mdf.MicroSeries([1, 2, 3], weights=[1, 2, 3]) - assert copy.deepcopy(s).sum() == 14 - - -@pytest.mark.parametrize("use_weight_column", [False, True]) -@pytest.mark.parametrize("use_pandas_pickle", [False, True]) -def test_serialization_preserves_weight_column_state( - use_weight_column, use_pandas_pickle -): - """Replacing restored weights can preserve the original weight column.""" - frame = mdf.MicroDataFrame( - pd.DataFrame({"x": [1, 2, 3], "w": [1, 2, 3]}, index=[7, 8, 9]), - weights="w" if use_weight_column else [1, 2, 3], - ) - if use_pandas_pickle: - buffer = io.BytesIO() - frame.to_pickle(buffer) - buffer.seek(0) - restored = pd.read_pickle(buffer) - else: - restored = pickle.loads(pickle.dumps(frame)) - - assert restored.weights_col == ("w" if use_weight_column else None) - restored.set_weights([3, 2, 1], preserve_old=True) - assert restored.sum()["x"] == 10 - assert restored.index.equals(frame.index) - if use_weight_column: - assert restored["old_w"].tolist() == [1, 2, 3] - else: - assert "old_w" not in restored.columns - - -@pytest.mark.parametrize("operation", ["pickle", "pandas_pickle", "deepcopy"]) -def test_named_microseries_preserves_name_and_weights(operation): - """Serialization retains pandas metadata as well as survey weights.""" - series = mdf.MicroSeries( - [1, 2, 3], index=[7, 8, 9], name="group", weights=[1, 2, 3] - ) - if operation == "deepcopy": - restored = copy.deepcopy(series) - elif operation == "pandas_pickle": - buffer = io.BytesIO() - series.to_pickle(buffer) - buffer.seek(0) - restored = pd.read_pickle(buffer) - else: - restored = pickle.loads(pickle.dumps(series)) - - assert restored.name == "group" - assert restored.index.equals(series.index) - pd.testing.assert_series_equal(restored.weights, series.weights) - assert restored.sum() == 14 - - -@pytest.mark.parametrize("selected", [False, True]) -def test_grouped_aggregation_retains_index_name(selected): - """Copying internal grouped weights must retain the grouping label.""" - frame = mdf.MicroDataFrame( - {"group": ["a", "a", "b", "b"], "value": [1, 2, 3, 4]}, - weights=[1, 2, 3, 4], - ) - grouped = frame.groupby("group") - if selected: - grouped = grouped[["value"]] - expected = pd.DataFrame( - {"value": [5.0, 25.0]}, index=pd.Index(["a", "b"], name="group") - ) - pd.testing.assert_frame_equal(grouped.sum(), expected) - - -@pytest.mark.parametrize("kind", ["frame_columns", "frame_index", "series_index"]) -def test_renamed_weights_are_independent(kind): - """Pandas finalization must retain the renamed result's copied weights.""" - if kind == "series_index": - original = mdf.MicroSeries( - [10, 20], index=[7, 8], name="income", weights=[1, 2] - ) - renamed = original.rename(index={7: 70}) - assert renamed.name == "income" - else: - original = mdf.MicroDataFrame( - {"x": [10, 20], "w": [1, 2]}, index=[7, 8], weights="w" - ) - renamed = ( - original.rename(columns={"x": "income"}) - if kind == "frame_columns" - else original.rename(index={7: 70}) - ) - assert renamed.weights_col == "w" - - renamed.weights.iloc[0] = 100 - pd.testing.assert_series_equal( - original.weights, pd.Series([1.0, 2.0], index=[7, 8]) - ) - original_total = original.sum() if kind == "series_index" else original.sum()["x"] - assert original_total == 10 * 1 + 20 * 2 - - original.weights.iloc[1] = 9 - assert renamed.weights.iloc[1] == 2 - if kind == "frame_columns": - # The renamed frame must still use weighted aggregation after pickle. - restored = pickle.loads(pickle.dumps(renamed)) - assert restored.weights_col == "w" - assert restored.sum()["income"] == 10 * 100 + 20 * 2 diff --git a/build/lib/microdf/tests/test_version_metadata.py b/build/lib/microdf/tests/test_version_metadata.py deleted file mode 100644 index 5bd1c874..00000000 --- a/build/lib/microdf/tests/test_version_metadata.py +++ /dev/null @@ -1,8 +0,0 @@ -import microdf as mdf - - -def test_version_matches_package_metadata(): - """__version__ must not drift from pyproject.toml.""" - from importlib.metadata import version - - assert mdf.__version__ == version("microdf-python") From 7a645f490452191e46af06e72cfbf4b999b28f8f Mon Sep 17 00:00:00 2001 From: Vahid Ahmadi Date: Wed, 16 Sep 2026 11:05:20 +0100 Subject: [PATCH 4/7] Leave the lockfile alone Co-Authored-By: Claude Opus 5 (1M context) --- uv.lock | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/uv.lock b/uv.lock index a5288ea3..0911adbd 100644 --- a/uv.lock +++ b/uv.lock @@ -99,7 +99,7 @@ resolution-markers = [ "python_full_version < '3.10'", ] dependencies = [ - { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "colorama", marker = "python_full_version < '3.10' and sys_platform == 'win32'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/b9/2e/0090cbf739cee7d23781ad4b89a9894a41538e4fcf4c31dcdd705b78eb8b/click-8.1.8.tar.gz", hash = "sha256:ed53c9d8990d83c2a27deae68e4ee337473f6330c040a31d4225c9574d16096a", size = 226593, upload-time = "2024-12-21T18:38:44.339Z" } wheels = [ @@ -116,7 +116,7 @@ resolution-markers = [ "python_full_version == '3.10.*'", ] dependencies = [ - { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "colorama", marker = "python_full_version >= '3.10' and sys_platform == 'win32'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/60/6c/8ca2efa64cf75a977a0d7fac081354553ebe483345c734fb6b6515d96bbc/click-8.2.1.tar.gz", hash = "sha256:27c491cc05d968d271d5a1db13e3b5a184636d9d930f148c50b038f0d0646202", size = 286342, upload-time = "2025-05-20T23:19:49.832Z" } wheels = [ @@ -243,7 +243,7 @@ name = "exceptiongroup" version = "1.3.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "typing-extensions" }, + { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/0b/9f/a65090624ecf468cdca03533906e7c69ed7588582240cfe7cc9e770b50eb/exceptiongroup-1.3.0.tar.gz", hash = "sha256:b241f5885f560bc56a59ee63ca4c6a8bfa46ae4ad651af316d4e81817bb9fd88", size = 29749, upload-time = "2025-05-10T17:42:51.123Z" } wheels = [ @@ -264,7 +264,7 @@ name = "importlib-metadata" version = "8.7.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "zipp" }, + { name = "zipp", marker = "python_full_version < '3.10'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/76/66/650a33bd90f786193e4de4b3ad86ea60b53c89b669a5c7be931fac31cdb0/importlib_metadata-8.7.0.tar.gz", hash = "sha256:d13b81ad223b890aa16c5471f2ac3056cf76c5f10f82d6f9292f0b415f389000", size = 56641, upload-time = "2025-04-27T15:29:01.736Z" } wheels = [ @@ -276,7 +276,7 @@ name = "importlib-resources" version = "6.5.2" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "zipp" }, + { name = "zipp", marker = "python_full_version < '3.10'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/cf/8c/f834fbf984f691b4f7ff60f50b514cc3de5cc08abfc3295564dd89c5e2e7/importlib_resources-6.5.2.tar.gz", hash = "sha256:185f87adef5bcc288449d98fb4fba07cea78bc036455dd44c5fc4a2fe78fed2c", size = 44693, upload-time = "2025-01-03T18:51:56.698Z" } wheels = [ @@ -402,7 +402,7 @@ wheels = [ [[package]] name = "microdf-python" -version = "1.3.8" +version = "1.3.1" source = { editable = "." } dependencies = [ { name = "numpy", version = "2.0.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.10'" }, @@ -777,10 +777,10 @@ resolution-markers = [ "python_full_version == '3.10.*'", ] dependencies = [ - { name = "certifi" }, - { name = "charset-normalizer" }, - { name = "idna" }, - { name = "urllib3" }, + { name = "certifi", marker = "python_full_version >= '3.10'" }, + { name = "charset-normalizer", marker = "python_full_version >= '3.10'" }, + { name = "idna", marker = "python_full_version >= '3.10'" }, + { name = "urllib3", marker = "python_full_version >= '3.10'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/e1/0a/929373653770d8a0d7ea76c37de6e41f11eb07559b103b1c02cafb3f7cf8/requests-2.32.4.tar.gz", hash = "sha256:27d0316682c8a29834d3264820024b62a36942083d52caf2f14c0591336d3422", size = 135258, upload-time = "2025-06-09T16:43:07.34Z" } wheels = [ From 058f8db282183b6fee3fb8ae5b5d59230861d4a3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Wed, 16 Sep 2026 14:43:34 +0200 Subject: [PATCH 5/7] Fix issues from review: preserve replicate data and expose centering --- .../replicate-weight-variance.added.md | 2 +- microdf/microseries.py | 24 ++- microdf/replication.py | 80 ++++++--- microdf/tests/test_replication.py | 166 ++++++++++++++++-- 4 files changed, 230 insertions(+), 42 deletions(-) diff --git a/changelog.d/replicate-weight-variance.added.md b/changelog.d/replicate-weight-variance.added.md index 425cb938..261c6d55 100644 --- a/changelog.d/replicate-weight-variance.added.md +++ b/changelog.d/replicate-weight-variance.added.md @@ -1 +1 @@ -- Added variance and standard error estimation from replicate weights, supporting jackknife, BRR, Fay's BRR, bootstrap and successive-difference schemes, for any statistic in the package. +Added variance and standard error estimation from replicate weights for scalar statistics, preserving input dtypes and supporting common-factor jackknife, BRR, Fay's BRR, bootstrap and successive-difference schemes with explicit full-sample or replicate-mean centering. diff --git a/microdf/microseries.py b/microdf/microseries.py index 1185d2e6..d5c6243e 100644 --- a/microdf/microseries.py +++ b/microdf/microseries.py @@ -585,6 +585,8 @@ def replicate_standard_error( replicate_weights, method: str = "jackknife", fay_k: Optional[float] = None, + *, + centering: str = "full-sample", ) -> float: """Standard error of ``statistic`` from a set of replicate weights. @@ -592,22 +594,32 @@ def replicate_standard_error( the factor appropriate to how the replicates were built, so it works for any statistic including the Gini coefficient and quantiles. - Only valid for replicate weights as published with a survey. Weights - calibrated to external targets no longer correspond to the original - replication scheme. + The factor and centering convention must match the survey design. + Supported schemes use a common factor: jackknife covers unstratified + JK1 or common-factor delete-group replication, not arbitrary stratified + jackknife. Averaged bootstrap requiring additional factors is not + supported. See :func:`microdf.replication.replicate_variance` for factors. + + Changing main weights without corresponding design-consistent replicate + adjustments invalidates the original replicates. Calibration can be + valid when repeated appropriately for every replicate. - :param statistic: Callable taking a MicroSeries, e.g. + :param statistic: Callable taking a MicroSeries and returning a float, e.g. ``lambda s: s.median()``. - :param replicate_weights: Array or frame of shape ``(len(self), R)``. + :param replicate_weights: Array or frame of shape ``(len(self), R)`` + in the same row order as this series. DataFrame labels are ignored. :param method: ``jackknife``, ``brr``, ``bootstrap``, ``successive-difference`` or ``fay``. :param fay_k: Fay's perturbation constant, for ``method="fay"``. + :param centering: ``full-sample`` (default) centers on ``statistic(self)``; + ``replicate-mean`` centers on the mean of the replicate estimates. + The method's scale factor is unchanged. :returns: The estimated standard error. """ from microdf.replication import replicate_standard_error return replicate_standard_error( - self, statistic, replicate_weights, method, fay_k + self, statistic, replicate_weights, method, fay_k, centering=centering ) @scalar_function diff --git a/microdf/replication.py b/microdf/replication.py index 56f72abb..c326318d 100644 --- a/microdf/replication.py +++ b/microdf/replication.py @@ -3,16 +3,19 @@ Many survey products publish a set of replicate weight vectors alongside the main weight. Recomputing a statistic once per replicate and measuring the spread gives a variance estimate that requires no analytic formula, which is -what makes it usable for statistics such as the Gini coefficient or a -quantile where the analytic variance is awkward. - -The scale factor depends on how the replicates were constructed, so the -method must be named rather than guessed. - -Note that this is only valid for replicate weights as published with a -survey. Weights that have been calibrated or reweighted to external targets -no longer correspond to the original replication scheme, and applying these -estimators to them does not describe the variance of the resulting estimator. +what makes it usable for statistics such as the Gini coefficient or a quantile +where the analytic variance is awkward. + +The scale factor and centering convention must match the survey's replication +design. The default centers on the full-sample estimate; ``replicate-mean`` +centering is also available. These estimators support common-factor replication +schemes, not arbitrary stratified jackknife or averaged-bootstrap designs that +require additional or replicate-specific factors. + +Changing the main weights without corresponding design-consistent adjustments +to the replicate weights invalidates the original replicates. Calibration can +be valid when the required calibration is repeated appropriately for every +replicate. """ from __future__ import annotations @@ -22,10 +25,10 @@ import numpy as np import pandas as pd -# Scale applied to the sum of squared deviations from the full-sample -# estimate. R is the number of replicates. +# Scale applied to the sum of squared deviations from the selected center. +# R is the number of replicates. METHOD_FACTORS = { - # Delete-a-group jackknife: (R - 1) / R. + # Unstratified JK1 / common-factor delete-group jackknife: (R - 1) / R. "jackknife": lambda r: (r - 1) / r, # Balanced repeated replication: 1 / R. "brr": lambda r: 1 / r, @@ -49,21 +52,41 @@ def replicate_variance( replicate_weights: np.ndarray | pd.DataFrame, method: str = "jackknife", fay_k: float | None = None, + *, + centering: str = "full-sample", ) -> float: """Variance of ``statistic`` estimated from replicate weights. + Variance is the method's scale factor times the sum of squared deviations + of replicate estimates from the selected center. The supported factors + are ``(R - 1) / R`` for unstratified JK1 or common-factor delete-group + jackknife, ``1 / R`` for BRR and bootstrap, ``4 / R`` for successive + difference, and ``1 / (R * (1 - fay_k)**2)`` for Fay's BRR. Select the + factor and center specified by the survey; arbitrary stratified jackknife + and averaged-bootstrap schemes requiring other factors are unsupported. + :param series: A MicroSeries. Its own weights give the point estimate. :param statistic: Callable taking a MicroSeries and returning a float, for example ``lambda s: s.gini()``. - :param replicate_weights: Array or frame of shape ``(len(series), R)``. + :param replicate_weights: Array or frame of shape ``(len(series), R)`` + in the same row order as ``series``. Rows are matched by position; + DataFrame index labels are ignored. :param method: One of ``jackknife``, ``brr``, ``bootstrap``, ``successive-difference``, or ``fay`` (which requires ``fay_k``). :param fay_k: Fay's perturbation constant, required when ``method="fay"``. + :param centering: ``full-sample`` (default) centers on ``statistic(series)``; + ``replicate-mean`` centers on the mean of the replicate estimates. + This choice does not change the method's scale factor. :returns: The estimated variance of the statistic. """ from microdf.microseries import MicroSeries + if centering not in ("full-sample", "replicate-mean"): + raise ValueError( + f"centering must be 'full-sample' or 'replicate-mean', got {centering!r}" + ) + weights = np.asarray(replicate_weights, dtype=float) if weights.ndim != 2: raise ValueError( @@ -91,16 +114,25 @@ def replicate_variance( known = ", ".join(sorted([*METHOD_FACTORS, "fay"])) raise ValueError(f"Unknown method {method!r}; expected one of {known}") - point = float(statistic(series)) - values = np.asarray(series, dtype=float) + center = float(statistic(series)) if centering == "full-sample" else None + values = series.array index = series.index - deviations = [] + estimates = [] for column in range(n_replicates): - replicate = MicroSeries(values, weights=weights[:, column], index=index) - deviations.append(float(statistic(replicate)) - point) + replicate = MicroSeries( + values.copy(), + weights=weights[:, column].copy(), + index=index.copy(), + name=series.name, + dtype=series.dtype, + ) + estimates.append(float(statistic(replicate))) - return factor * float(np.sum(np.square(deviations))) + estimates = np.asarray(estimates) + if centering == "replicate-mean": + center = float(np.mean(estimates)) + return factor * float(np.sum(np.square(estimates - center))) def replicate_standard_error( @@ -109,11 +141,17 @@ def replicate_standard_error( replicate_weights: np.ndarray | pd.DataFrame, method: str = "jackknife", fay_k: float | None = None, + *, + centering: str = "full-sample", ) -> float: """Standard error of ``statistic``, the square root of its variance. Takes the same arguments as :func:`replicate_variance`. """ return float( - np.sqrt(replicate_variance(series, statistic, replicate_weights, method, fay_k)) + np.sqrt( + replicate_variance( + series, statistic, replicate_weights, method, fay_k, centering=centering + ) + ) ) diff --git a/microdf/tests/test_replication.py b/microdf/tests/test_replication.py index 28b5bbb0..d19e7fcc 100644 --- a/microdf/tests/test_replication.py +++ b/microdf/tests/test_replication.py @@ -6,6 +6,7 @@ """ import numpy as np +import pandas as pd import pytest from microdf import MicroSeries, replicate_standard_error, replicate_variance @@ -39,13 +40,12 @@ def test_matches_the_analytic_standard_error_of_a_weighted_mean( def test_works_for_statistics_with_no_analytic_variance( series_and_replicates, ): - """The point of the method: Gini and quantiles come out like anything else.""" + """The point of the method: Gini and quantiles come out like anything + else.""" series, replicates = series_and_replicates for statistic in (lambda s: s.gini(), lambda s: s.median()): - se = replicate_standard_error( - series, statistic, replicates, method="bootstrap" - ) + se = replicate_standard_error(series, statistic, replicates, method="bootstrap") assert se > 0 assert np.isfinite(se) @@ -65,12 +65,8 @@ def test_each_method_applies_its_own_scale( """The scale factor is what distinguishes the replication schemes.""" series, replicates = series_and_replicates - variance = replicate_variance( - series, lambda s: s.mean(), replicates, method=method - ) - reference = replicate_variance( - series, lambda s: s.mean(), replicates, method="brr" - ) + variance = replicate_variance(series, lambda s: s.mean(), replicates, method=method) + reference = replicate_variance(series, lambda s: s.mean(), replicates, method="brr") assert variance == pytest.approx(reference * expected_factor * 200, rel=1e-9) @@ -84,9 +80,7 @@ def test_fay_requires_and_uses_its_constant(series_and_replicates): fay = replicate_variance( series, lambda s: s.mean(), replicates, method="fay", fay_k=0.5 ) - brr = replicate_variance( - series, lambda s: s.mean(), replicates, method="brr" - ) + brr = replicate_variance(series, lambda s: s.mean(), replicates, method="brr") # 1 / (R (1 - k)^2) against 1 / R, so a factor of four at k = 0.5. assert fay == pytest.approx(brr * 4, rel=1e-9) @@ -105,6 +99,150 @@ def test_rejects_input_that_cannot_be_right(series_and_replicates): replicate_variance(series, lambda s: s.mean(), np.ones((500, 1))) with pytest.raises(ValueError, match="Unknown method"): + replicate_variance(series, lambda s: s.mean(), replicates, method="nonsense") + + +@pytest.mark.parametrize("dtype", ["object", "category", "string"]) +def test_replicates_preserve_categorical_values_and_metadata(dtype): + series = MicroSeries( + ["employed", "unemployed", "employed", None], + weights=[1, 1, 1, 1], + index=["a", "b", "c", "d"], + name="employment", + dtype=dtype, + ) + replicates = np.array([[2, 2, 0, 0], [0, 0, 2, 2], [2, 0, 2, 0], [0, 2, 0, 2]]) + original = pd.Series(series).copy(deep=True) + original_weights = series.weights.copy(deep=True) + original_replicates = replicates.copy() + + def count(sample): + assert sample.dtype == series.dtype + assert sample.name == "employment" + pd.testing.assert_index_equal(sample.index, series.index) + return sample.count() + + # Full count is 3; replicate counts are 4, 2, 4, 2. + assert replicate_variance(series, count, replicates, method="brr") == 1 + pd.testing.assert_series_equal(pd.Series(series), original) + pd.testing.assert_series_equal(series.weights, original_weights) + np.testing.assert_array_equal(replicates, original_replicates) + + +@pytest.mark.parametrize("dtype", ["bool", "boolean"]) +def test_replicates_preserve_boolean_domain_counts(dtype): + series = MicroSeries([True, False, True, False], weights=[1, 1, 1, 1], dtype=dtype) + replicates = np.array([[2, 2, 0, 0], [0, 0, 2, 2], [2, 0, 2, 0], [0, 2, 0, 2]]) + # Complement counts are 0, 2, 2, 4, around the full-sample count of 2. + assert ( + replicate_variance( + series, lambda sample: (~sample).sum(), replicates, method="brr" + ) + == 2 + ) + + +@pytest.mark.parametrize("dtype", ["int64", "Int64"]) +def test_replicates_preserve_exact_large_integer_categories(dtype): + category = 2**53 + series = MicroSeries([category, category + 1], weights=[1, 2], dtype=dtype) + replicates = np.array([[1, 1], [2, 2]]) + # Identical weights must keep the weighted category count exactly 1. + assert ( + replicate_variance( + series, lambda sample: (sample == category).sum(), replicates, method="brr" + ) + == 0 + ) + + +@pytest.mark.parametrize( + "method,fay_k,full_sample_variance,replicate_mean_variance", + [ + ("jackknife", None, 1088 / 3, 3136 / 9), + ("brr", None, 544 / 3, 1568 / 9), + ("bootstrap", None, 544 / 3, 1568 / 9), + ("successive-difference", None, 2176 / 3, 6272 / 9), + ("fay", 0.5, 2176 / 3, 6272 / 9), + ], +) +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_centering_uses_exact_nonlinear_replicate_estimates( + method, fay_k, full_sample_variance, replicate_mean_variance, centering, api +): + series = MicroSeries([1, 3], weights=[1, 1]) + replicates = np.array([[2, 1, 0], [0, 1, 2]]) + + # Squared totals: full sample 16; replicates 4, 16, 36; replicate mean 56/3. + # Squared deviations sum to 544 from 16 and 1568/3 from 56/3. + def squared_total(sample): + return sample.sum() ** 2 + + expected = ( + full_sample_variance if centering == "full-sample" else replicate_mean_variance + ) + kwargs = {"method": method, "fay_k": fay_k, "centering": centering} + if api == "variance": + observed = replicate_variance(series, squared_total, replicates, **kwargs) + elif api == "standard_error": + observed = replicate_standard_error(series, squared_total, replicates, **kwargs) + expected = np.sqrt(expected) + else: + observed = series.replicate_standard_error(squared_total, replicates, **kwargs) + expected = np.sqrt(expected) + assert observed == pytest.approx(expected) + + +def test_default_centering_retains_full_sample_estimate(): + series = MicroSeries([1, 3], weights=[1, 1]) + replicates = np.array([[2, 1, 0], [0, 1, 2]]) + expected_variance = 544 / 3 + + def squared_total(sample): + return sample.sum() ** 2 + + assert replicate_variance( + series, squared_total, replicates, method="bootstrap" + ) == pytest.approx(expected_variance) + assert replicate_standard_error( + series, squared_total, replicates, method="bootstrap" + ) == pytest.approx(np.sqrt(expected_variance)) + assert series.replicate_standard_error( + squared_total, replicates, method="bootstrap" + ) == pytest.approx(np.sqrt(expected_variance)) + + +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_rejects_unknown_centering_before_calling_statistic(api): + series = MicroSeries([1, 3], weights=[1, 1]) + replicates = np.array([[2, 0], [0, 2]]) + + def unexpected_statistic(sample): + raise AssertionError("Invalid centering must be rejected first") + + with pytest.raises(ValueError, match="centering"): + if api == "variance": + replicate_variance( + series, unexpected_statistic, replicates, centering="unknown" + ) + elif api == "standard_error": + replicate_standard_error( + series, unexpected_statistic, replicates, centering="unknown" + ) + else: + series.replicate_standard_error( + unexpected_statistic, replicates, centering="unknown" + ) + + +def test_replicate_weight_frames_use_positional_rows(): + series = MicroSeries([1, 3], weights=[1, 1], index=["a", "b"]) + replicates = pd.DataFrame([[2, 0], [0, 2]], index=["b", "a"]) + # Position-defined means are 1 and 3 around the full-sample mean of 2. + assert ( replicate_variance( - series, lambda s: s.mean(), replicates, method="nonsense" + series, lambda sample: sample.mean(), replicates, method="brr" ) + == 1 + ) From d38c2a3ff4e16c51daa1f4de30951e61c78a925d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Wed, 16 Sep 2026 16:42:47 +0200 Subject: [PATCH 6/7] Fix issues from review: stabilize replicate variance and qualify quantiles --- .../replicate-weight-variance.added.md | 2 +- microdf/microseries.py | 6 +- microdf/replication.py | 44 ++++- microdf/tests/test_replication.py | 175 ++++++++++++++++++ 4 files changed, 218 insertions(+), 9 deletions(-) diff --git a/changelog.d/replicate-weight-variance.added.md b/changelog.d/replicate-weight-variance.added.md index 261c6d55..000c242b 100644 --- a/changelog.d/replicate-weight-variance.added.md +++ b/changelog.d/replicate-weight-variance.added.md @@ -1 +1 @@ -Added variance and standard error estimation from replicate weights for scalar statistics, preserving input dtypes and supporting common-factor jackknife, BRR, Fay's BRR, bootstrap and successive-difference schemes with explicit full-sample or replicate-mean centering. +Added variance and standard error estimation from replicate weights for scalar statistics, preserving input dtypes and supporting common-factor jackknife, BRR, Fay's BRR, bootstrap and successive-difference schemes with explicit full-sample or replicate-mean centering. Reference-relative centering and scaled accumulation preserve representable variances at extreme magnitudes. Statistical validity depends on the statistic and survey design; nonsmooth quantiles can require an appropriate replication method or smoothing. diff --git a/microdf/microseries.py b/microdf/microseries.py index d5c6243e..47569055 100644 --- a/microdf/microseries.py +++ b/microdf/microseries.py @@ -591,8 +591,10 @@ def replicate_standard_error( """Standard error of ``statistic`` from a set of replicate weights. Recomputes the statistic once per replicate and scales the spread by - the factor appropriate to how the replicates were built, so it works - for any statistic including the Gini coefficient and quantiles. + the factor appropriate to how the replicates were built. Statistical + validity depends on both the statistic and the survey design. + Nonsmooth statistics such as quantiles can require an appropriate + replication method or smoothing of replicate estimates. The factor and centering convention must match the survey design. Supported schemes use a common factor: jackknife covers unstratified diff --git a/microdf/replication.py b/microdf/replication.py index c326318d..2d3ca8c2 100644 --- a/microdf/replication.py +++ b/microdf/replication.py @@ -2,9 +2,10 @@ Many survey products publish a set of replicate weight vectors alongside the main weight. Recomputing a statistic once per replicate and measuring the -spread gives a variance estimate that requires no analytic formula, which is -what makes it usable for statistics such as the Gini coefficient or a quantile -where the analytic variance is awkward. +spread gives a variance estimate without an analytic variance formula. Its +statistical validity depends on both the statistic and the replication design. +Nonsmooth statistics such as quantiles can require an appropriate replication +method or smoothing; these functions apply the supplied statistic directly. The scale factor and centering convention must match the survey's replication design. The default centers on the full-sample estimate; ``replicate-mean`` @@ -64,6 +65,8 @@ def replicate_variance( difference, and ``1 / (R * (1 - fay_k)**2)`` for Fay's BRR. Select the factor and center specified by the survey; arbitrary stratified jackknife and averaged-bootstrap schemes requiring other factors are unsupported. + Validity also depends on the statistic: nonsmooth quantiles can require + an appropriate replication method or smoothing of replicate estimates. :param series: A MicroSeries. Its own weights give the point estimate. :param statistic: Callable taking a MicroSeries and returning a float, @@ -130,9 +133,38 @@ def replicate_variance( estimates.append(float(statistic(replicate))) estimates = np.asarray(estimates) - if centering == "replicate-mean": - center = float(np.mean(estimates)) - return factor * float(np.sum(np.square(estimates - center))) + if not np.all(np.isfinite(estimates)) or ( + center is not None and not np.isfinite(center) + ): + # Retain the original infinity/NaN propagation for nonfinite callbacks. + if centering == "replicate-mean": + center = float(np.mean(estimates)) + return factor * float(np.sum(np.square(estimates - center))) + + with np.errstate(over="ignore"): + if centering == "replicate-mean": + # Keep the common offset out of the mean so small differences are + # preserved even when the absolute mean is not representable. + deviations = estimates - estimates[0] + if not np.all(np.isfinite(deviations)): + return float("inf") + deviations -= np.mean(deviations) + else: + deviations = estimates - center + + scale = np.max(np.abs(deviations)) + if scale == 0: + return 0.0 + if not np.isfinite(scale): + return float("inf") + + scaled_squares = np.sum(np.square(deviations / scale)) + # Restore the scale by its binary exponent after applying the factor. + # Squaring scale first could overflow, or underflow before Fay's factor + # brings a tiny squared deviation back into the representable range. + significand, exponent = np.frexp(scale) + with np.errstate(over="ignore"): + return float(np.ldexp(significand**2 * factor * scaled_squares, 2 * exponent)) def replicate_standard_error( diff --git a/microdf/tests/test_replication.py b/microdf/tests/test_replication.py index d19e7fcc..835176fb 100644 --- a/microdf/tests/test_replication.py +++ b/microdf/tests/test_replication.py @@ -246,3 +246,178 @@ def test_replicate_weight_frames_use_positional_rows(): ) == 1 ) + + +def _replication_result(series, statistic, replicates, api, **kwargs): + if api == "variance": + return replicate_variance(series, statistic, replicates, **kwargs) + if api == "standard_error": + return replicate_standard_error(series, statistic, replicates, **kwargs) + return series.replicate_standard_error(statistic, replicates, **kwargs) + + +@pytest.mark.parametrize( + "method,fay_k,amplitude,expected_variance", + [ + ("jackknife", None, 5e153, 1.75e308), + ("brr", None, 1e154, 1e308), + ("bootstrap", None, 1e154, 1e308), + ("successive-difference", None, 5e153, 1e308), + ("fay", 0.5, 5e153, 1e308), + ], +) +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_large_finite_replicate_variance( + method, fay_k, amplitude, expected_variance, centering, api +): + series = MicroSeries([0.0, 2 * amplitude], weights=[1, 1]) + replicates = np.tile([[2, 0], [0, 2]], (1, 4)) + # Eight deviations are +/- amplitude. The factors are 7/8, 1/8 or 1/2. + # Every method has a finite variance despite an overflowing raw sum. + expected = expected_variance if api == "variance" else np.sqrt(expected_variance) + observed = _replication_result( + series, + lambda sample: sample.mean(), + replicates, + api, + method=method, + fay_k=fay_k, + centering=centering, + ) + assert observed == pytest.approx(expected, rel=1e-14, abs=0) + + +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_identical_large_replicates_have_zero_variance(centering, api): + series = MicroSeries([1e308, 1e308], weights=[1, 1]) + replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]]) + # The median remains finite even though summing the four estimates overflows. + assert ( + _replication_result( + series, + lambda sample: sample.median(), + replicates, + api, + method="brr", + centering=centering, + ) + == 0 + ) + + +@pytest.mark.parametrize( + "method,fay_k,variance_multiplier", + [ + ("jackknife", None, 3), + ("brr", None, 1), + ("bootstrap", None, 1), + ("successive-difference", None, 4), + ("fay", 0.5, 4), + ], +) +@pytest.mark.parametrize("offset", [0.0, float(2**53)]) +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_replicate_centering_preserves_small_differences( + method, fay_k, variance_multiplier, offset, centering, api +): + series = MicroSeries([offset, offset + 2], weights=[1, 1]) + replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]]) + # The exact replicate mean has deviations +/-1, including at 2**53. + # Full-sample centering must retain the callback's rounded mean at 2**53, + # so its deviations are 0 and 2, giving twice the centered variance. + expected = variance_multiplier + if offset and centering == "full-sample": + expected *= 2 + if api != "variance": + expected = np.sqrt(expected) + observed = _replication_result( + series, + lambda sample: sample.mean(), + replicates, + api, + method=method, + fay_k=fay_k, + centering=centering, + ) + assert observed == pytest.approx(expected, rel=1e-14, abs=0) + + +@pytest.mark.parametrize( + "amplitude,method,fay_k,expected_variance", + [ + (2.0**-537, "brr", None, 2.0**-1074), + (2.0**-550, "fay", 1 - 2.0**-53, 2.0**-994), + ], +) +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_small_deviations_keep_representable_variance( + amplitude, method, fay_k, expected_variance, centering, api +): + series = MicroSeries([-amplitude, amplitude], weights=[1, 1]) + replicates = np.array([[2, 0, 2, 0], [0, 2, 0, 2]]) + # BRR variance is amplitude**2. Fay's factor multiplies that by 2**106; + # it must be applied before rounding the initially unrepresentable square. + expected = expected_variance if api == "variance" else np.sqrt(expected_variance) + observed = _replication_result( + series, + lambda sample: sample.mean(), + replicates, + api, + method=method, + fay_k=fay_k, + centering=centering, + ) + assert observed == expected + + +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +def test_finite_replicates_with_unrepresentable_variance(centering): + series = MicroSeries([-1e308, 1e308], weights=[1, 1]) + replicates = np.array([[1, 0], [0, 1]]) + # A weighted total gives finite estimates +/-1e308 and a zero point estimate. + with np.errstate(over="ignore", invalid="ignore"): + observed = replicate_variance( + series, + lambda sample: sample.sum(), + replicates, + method="brr", + centering=centering, + ) + assert observed == np.inf + + +@pytest.mark.parametrize( + "point,estimates,centering,expected", + [ + (0.0, [0.0, np.inf], "full-sample", np.inf), + (np.inf, [0.0, 1.0], "full-sample", np.inf), + (np.inf, [0.0, np.inf], "full-sample", np.nan), + (np.nan, [0.0, 1.0], "full-sample", np.nan), + (0.0, [0.0, np.nan], "full-sample", np.nan), + (0.0, [0.0, np.inf], "replicate-mean", np.nan), + (0.0, [np.inf, -np.inf], "replicate-mean", np.nan), + (0.0, [0.0, np.nan], "replicate-mean", np.nan), + ], +) +def test_nonfinite_callback_results_keep_existing_behavior( + point, estimates, centering, expected +): + series = MicroSeries([1, 2], weights=[1, 1]) + replicates = np.array([[2, 0], [0, 2]]) + results = iter([point, *estimates] if centering == "full-sample" else estimates) + with np.errstate(over="ignore", invalid="ignore"): + observed = replicate_variance( + series, + lambda sample: next(results), + replicates, + method="brr", + centering=centering, + ) + if np.isnan(expected): + assert np.isnan(observed) + else: + assert observed == expected From f8b75baa1fb14741f600dedc7b2d97e947105268 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Thu, 17 Sep 2026 10:32:39 +0100 Subject: [PATCH 7/7] Fix issues from review: isolate full-sample callbacks --- .../replicate-weight-variance.added.md | 2 +- microdf/replication.py | 5 +- microdf/tests/test_replication.py | 121 ++++++++++++++++++ 3 files changed, 126 insertions(+), 2 deletions(-) diff --git a/changelog.d/replicate-weight-variance.added.md b/changelog.d/replicate-weight-variance.added.md index 000c242b..d8f076e7 100644 --- a/changelog.d/replicate-weight-variance.added.md +++ b/changelog.d/replicate-weight-variance.added.md @@ -1 +1 @@ -Added variance and standard error estimation from replicate weights for scalar statistics, preserving input dtypes and supporting common-factor jackknife, BRR, Fay's BRR, bootstrap and successive-difference schemes with explicit full-sample or replicate-mean centering. Reference-relative centering and scaled accumulation preserve representable variances at extreme magnitudes. Statistical validity depends on the statistic and survey design; nonsmooth quantiles can require an appropriate replication method or smoothing. +Added variance and standard error estimation from replicate weights for scalar statistics, preserving input dtypes and supporting common-factor jackknife, BRR, Fay's BRR, bootstrap and successive-difference schemes with explicit full-sample or replicate-mean centering. Reference-relative centering and scaled accumulation preserve representable variances at extreme magnitudes. Full-sample callbacks receive independent copies so in-place transformations preserve caller data and subsequent replicate inputs. Statistical validity depends on the statistic and survey design; nonsmooth quantiles can require an appropriate replication method or smoothing. diff --git a/microdf/replication.py b/microdf/replication.py index 2d3ca8c2..a8923dc8 100644 --- a/microdf/replication.py +++ b/microdf/replication.py @@ -117,7 +117,10 @@ def replicate_variance( known = ", ".join(sorted([*METHOD_FACTORS, "fay"])) raise ValueError(f"Unknown method {method!r}; expected one of {known}") - center = float(statistic(series)) if centering == "full-sample" else None + # Keep in-place callback transformations out of the caller and replicates. + center = ( + float(statistic(series.copy(deep=True))) if centering == "full-sample" else None + ) values = series.array index = series.index diff --git a/microdf/tests/test_replication.py b/microdf/tests/test_replication.py index 835176fb..dea219e0 100644 --- a/microdf/tests/test_replication.py +++ b/microdf/tests/test_replication.py @@ -421,3 +421,124 @@ def test_nonfinite_callback_results_keep_existing_behavior( assert np.isnan(observed) else: assert observed == expected + + +@pytest.mark.parametrize("inplace", [False, True]) +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_callback_transforms_each_sample_once(inplace, centering, api): + series = MicroSeries([120.0, 180.0], weights=[1, 1]) + replicates = np.array([[2.0, 0.0], [0.0, 2.0]]) + original_values = pd.Series(series).copy(deep=True) + original_weights = series.weights.copy(deep=True) + original_replicates = replicates.copy() + calls = [] + + def total_after_allowance(sample): + calls.append(sample.weights.tolist()) + if inplace: + sample -= 100 + return sample.sum() + return (sample - 100).sum() + + # Transformed values are 20 and 80; totals are 100, 40 and 160. + # BRR variance is ((40 - 100)**2 + (160 - 100)**2) / 2 = 3600. + expected = 3600 if api == "variance" else 60 + assert ( + _replication_result( + series, + total_after_allowance, + replicates, + api, + method="brr", + centering=centering, + ) + == expected + ) + expected_calls = [[2.0, 0.0], [0.0, 2.0]] + if centering == "full-sample": + expected_calls.insert(0, [1.0, 1.0]) + assert calls == expected_calls + pd.testing.assert_series_equal(pd.Series(series), original_values) + pd.testing.assert_series_equal(series.weights, original_weights) + np.testing.assert_array_equal(replicates, original_replicates) + + +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_callbacks_isolate_values_weights_and_metadata(centering, api): + series = MicroSeries( + [120, 180], + weights=[1, 1], + index=pd.Index(["a", "b"], name="person"), + name="income", + dtype="Int64", + ) + replicates = np.array([[2.0, 0.0], [0.0, 2.0]]) + original = series.copy(deep=True) + original_replicates = replicates.copy() + calls = [] + + def mutating_total(sample): + assert sample.tolist() == [120, 180] + assert sample.dtype == original.dtype + assert sample.name == "income" + pd.testing.assert_index_equal(sample.index, original.index) + calls.append(sample.weights.tolist()) + sample -= 100 + estimate = sample.sum() + sample.weights.iloc[:] = 7 + sample.index = pd.Index(["x", "y"], name="changed") + sample.name = "changed" + return estimate + + expected = 3600 if api == "variance" else 60 + assert ( + _replication_result( + series, mutating_total, replicates, api, method="brr", centering=centering + ) + == expected + ) + expected_calls = [[2.0, 0.0], [0.0, 2.0]] + if centering == "full-sample": + expected_calls.insert(0, [1.0, 1.0]) + assert calls == expected_calls + pd.testing.assert_series_equal(pd.Series(series), pd.Series(original)) + pd.testing.assert_series_equal(series.weights, original.weights) + np.testing.assert_array_equal(replicates, original_replicates) + + +@pytest.mark.parametrize("centering", ["full-sample", "replicate-mean"]) +@pytest.mark.parametrize("api", ["variance", "standard_error", "series_method"]) +def test_callback_exception_preserves_inputs(centering, api): + series = MicroSeries( + [120.0, 180.0], weights=[1, 1], index=["a", "b"], name="income" + ) + replicates = np.array([[2.0, 0.0], [0.0, 2.0]]) + original = series.copy(deep=True) + original_replicates = replicates.copy() + callback_error = ValueError("statistic failed after mutation") + calls = [] + + def failing_statistic(sample): + calls.append(sample.weights.tolist()) + sample -= 100 + sample.weights.iloc[:] = 7 + sample.index = ["x", "y"] + sample.name = "changed" + raise callback_error + + with pytest.raises(ValueError, match="statistic failed") as raised: + _replication_result( + series, + failing_statistic, + replicates, + api, + method="brr", + centering=centering, + ) + assert raised.value is callback_error + assert calls == ([[1.0, 1.0]] if centering == "full-sample" else [[2.0, 0.0]]) + pd.testing.assert_series_equal(pd.Series(series), pd.Series(original)) + pd.testing.assert_series_equal(series.weights, original.weights) + np.testing.assert_array_equal(replicates, original_replicates)