Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/weighted-dataframe-cov-corr.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
`MicroDataFrame.cov()` and `.corr()` now return frequency-weighted pairwise matrices using the same estimator as `MicroSeries.cov` and `.corr`, instead of unweighted pandas results. As on `MicroSeries`, `corr` accepts only `method="pearson"`.
111 changes: 101 additions & 10 deletions microdf/microdataframe.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,15 @@
import numpy as np
import pandas as pd

from microdf.microseries import MicroSeries, MicroSeriesGroupBy
from microdf.microseries import (
MicroSeries,
MicroSeriesGroupBy,
_pair_options,
_usable_pair,
_validate_frequency_weights,
_weighted_correlation,
_weighted_covariance,
)
from microdf._weights import (
WeightPropagationMixin,
aligned_weights,
Expand Down Expand Up @@ -74,16 +82,99 @@ def __finalize__(self, other, method=None, **kwargs):
super().__finalize__(other, method=method, **kwargs)
return finalize_weights(self, other, method, previous)

@wraps(pd.DataFrame.cov)
def cov(self, *args, **kwargs) -> pd.DataFrame:
# Column summaries have no observation weights, even if labels match.
result = pd.DataFrame(self, copy=False).cov(*args, **kwargs)
return result.__finalize__(self, method="cov")
def cov(
self,
min_periods: Optional[int] = None,
ddof: int = 1,
numeric_only: bool = False,
) -> pd.DataFrame:
"""Pairwise frequency-weighted covariance of the columns.

Every cell uses the estimator of :meth:`MicroSeries.cov` with this
frame's weights: ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) -
ddof)`` over the rows where both columns are present and the weight
is positive. Integer weights therefore match ``pandas.DataFrame.cov``
on the replicated sample. Missing values are removed pairwise, so
each cell can use a different set of rows, as in pandas.

The result is a plain ``pandas.DataFrame``: it summarises columns,
so it carries no row weights.

:param min_periods: Minimum usable row pairs per cell, not the sum
of frequency weights. Cells with fewer are NaN. Defaults to 1.
:param ddof: Degrees of freedom subtracted from the weight total.
:param numeric_only: Use only numeric columns. Otherwise every column
is converted to float, raising the error pandas raises.
:returns: Covariance matrix indexed by column in both directions.
"""
return self._weighted_pairwise(
"cov", min_periods=min_periods, ddof=ddof, numeric_only=numeric_only
)

@wraps(pd.DataFrame.corr)
def corr(self, *args, **kwargs) -> pd.DataFrame:
result = pd.DataFrame(self, copy=False).corr(*args, **kwargs)
return result.__finalize__(self, method="corr")
def corr(
self,
method: str = "pearson",
min_periods: int = 1,
numeric_only: bool = False,
) -> pd.DataFrame:
"""Pairwise frequency-weighted Pearson correlation of the columns.

Every cell uses the estimator of :meth:`MicroSeries.corr` with this
frame's weights over the rows where both columns are present and the
weight is positive. Constant columns give NaN. Only ``"pearson"`` is
supported, as on :meth:`MicroSeries.corr`; for an unweighted rank
correlation convert to ``pandas.DataFrame`` first.

The result is a plain ``pandas.DataFrame``: it summarises columns, so
it carries no row weights.

:param method: Only "pearson" is supported.
:param min_periods: Minimum usable row pairs per cell, not the sum of
frequency weights. Cells with fewer are NaN.
:param numeric_only: Use only numeric columns. Otherwise every column
is converted to float, raising the error pandas raises.
:returns: Correlation matrix indexed by column in both directions.
"""
if method != "pearson":
raise ValueError("weighted correlation only supports method='pearson'")
return self._weighted_pairwise(
"corr", min_periods=min_periods, ddof=1, numeric_only=numeric_only
)

def _weighted_pairwise(
self,
statistic: str,
*,
min_periods: Optional[int],
ddof: int,
numeric_only: bool,
) -> pd.DataFrame:
"""Fill a symmetric column matrix one usable pair at a time."""
min_periods, ddof = _pair_options(min_periods, ddof)
data = self._get_numeric_data() if numeric_only else self
frame = pd.DataFrame(data, copy=False)
# The conversion pandas uses, so non-numeric columns raise its error.
values = frame.to_numpy(dtype=float, na_value=np.nan)
weights = np.asarray(self.weights, dtype=float)
_validate_frequency_weights(weights)
columns = frame.columns
matrix = np.full((len(columns), len(columns)), np.nan)
for i in range(len(columns)):
for j in range(i, len(columns)):
pair = _usable_pair(
values[:, i], values[:, j], weights, min_periods, ddof, True
)
if pair is None:
continue
x, y, pair_weights, denominator = pair
if statistic == "cov":
cell = _weighted_covariance(x, y, pair_weights, denominator)
else:
cell = _weighted_correlation(x, y, pair_weights)
matrix[i, j] = matrix[j, i] = cell
result = pd.DataFrame(matrix, index=columns, columns=columns)
# Column summaries have no observation weights, even if labels match.
return pd.DataFrame.__finalize__(result, self, method=statistic)

def __setstate__(self, state) -> None:
"""Restore a pickled MicroDataFrame.
Expand Down
137 changes: 83 additions & 54 deletions microdf/microseries.py
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,84 @@ def _weighted_centered_vector(
return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift)


def _pair_options(min_periods: Optional[int], ddof: int) -> tuple[int, int]:
"""Validate the row minimum and degrees of freedom for paired moments."""
if not isinstance(ddof, (int, np.integer)):
raise TypeError("ddof must be an integer")
if min_periods is None:
min_periods = 1
if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
raise ValueError("min_periods must be a nonnegative integer")
return int(min_periods), int(ddof)


def _validate_frequency_weights(weights: np.ndarray) -> None:
"""Reject weights that cannot be frequencies."""
if not np.isfinite(weights).all() or (weights < 0).any():
raise ValueError("frequency weights must be finite and nonnegative")


def _usable_pair(
x: np.ndarray,
y: np.ndarray,
weights: np.ndarray,
min_periods: int,
ddof: int,
skipna: bool,
) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]:
"""Keep the present, positively weighted rows of an aligned pair."""
# Zero frequency means the row is absent, including for skipna=False.
positive = weights > 0
x, y, weights = x[positive], y[positive], weights[positive]
missing = np.isnan(x) | np.isnan(y)
if not skipna and missing.any():
return None
x, y, weights = x[~missing], y[~missing], weights[~missing]
total_weight = weights.sum()
if not np.isfinite(total_weight):
raise ValueError("the sum of frequency weights must be finite")
if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
return None
return x, y, weights, float(total_weight - ddof)


def _weighted_covariance(
x: np.ndarray, y: np.ndarray, weights: np.ndarray, denominator: float
) -> float:
"""Frequency-weighted covariance of a usable pair."""
if not np.isfinite(x).all() or not np.isfinite(y).all():
return np.nan
x, x_exponent = _weighted_centered_vector(x, weights)
y, y_exponent = _weighted_centered_vector(y, weights)
# Combine exponents only after dividing out sum(weights) - ddof.
# Neither the original squared scale nor raw weighted sum need fit.
denominator, denominator_exponent = np.frexp(denominator)
return float(
np.ldexp(
np.sum(x * y) / denominator,
x_exponent + y_exponent - int(denominator_exponent),
)
)


def _weighted_correlation(x: np.ndarray, y: np.ndarray, weights: np.ndarray) -> float:
"""Frequency-weighted Pearson correlation of a usable pair."""
if not np.isfinite(x).all() or not np.isfinite(y).all():
return np.nan
# A weighted mean can round away from identical decimal inputs.
# Check the retained observations exactly before subtracting it.
if (x == x[0]).all() or (y == y[0]).all():
return np.nan
x, _ = _weighted_centered_vector(x, weights)
y, _ = _weighted_centered_vector(y, weights)
x_ss = np.sum(x * x)
y_ss = np.sum(y * y)
if x_ss == 0 or y_ss == 0:
return np.nan
result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
return float(np.clip(result, -1.0, 1.0))


def _weighted_top_share(
values: np.ndarray, weights: np.ndarray, top_x_pct: float
) -> float:
Expand Down Expand Up @@ -456,12 +534,7 @@ def _weighted_pair(
"""Align usable paired observations with their left weights."""
if not isinstance(other, pd.Series):
raise TypeError("other must be a pandas Series or MicroSeries")
if not isinstance(ddof, (int, np.integer)):
raise TypeError("ddof must be an integer")
if min_periods is None:
min_periods = 1
if not isinstance(min_periods, (int, np.integer)) or min_periods < 0:
raise ValueError("min_periods must be a nonnegative integer")
min_periods, ddof = _pair_options(min_periods, ddof)
if len(self) == 0 or len(other) == 0:
return None

Expand All @@ -477,27 +550,8 @@ def _weighted_pair(
)
y = right.to_numpy(dtype=float, na_value=np.nan)
weights = np.asarray(self.weights, dtype=float)[positions]
if not np.isfinite(weights).all() or (weights < 0).any():
raise ValueError("frequency weights must be finite and nonnegative")

# Zero frequency means the row is absent, including for skipna=False.
positive = weights > 0
x, y, weights = x[positive], y[positive], weights[positive]
missing = np.isnan(x) | np.isnan(y)
if not skipna and missing.any():
return None
x, y, weights = x[~missing], y[~missing], weights[~missing]
total_weight = weights.sum()
if not np.isfinite(total_weight):
raise ValueError("the sum of frequency weights must be finite")
if len(x) < min_periods or total_weight == 0 or total_weight <= ddof:
return None
return (
x,
y,
weights,
float(total_weight - ddof),
)
_validate_frequency_weights(weights)
return _usable_pair(x, y, weights, min_periods, ddof, skipna)

def cov(
self,
Expand Down Expand Up @@ -531,19 +585,7 @@ def cov(
if pair is None:
return np.nan
x, y, weights, denominator = pair
if not np.isfinite(x).all() or not np.isfinite(y).all():
return np.nan
x, x_exponent = _weighted_centered_vector(x, weights)
y, y_exponent = _weighted_centered_vector(y, weights)
# Combine exponents only after dividing out sum(weights) - ddof.
# Neither the original squared scale nor raw weighted sum need fit.
denominator, denominator_exponent = np.frexp(denominator)
return float(
np.ldexp(
np.sum(x * y) / denominator,
x_exponent + y_exponent - int(denominator_exponent),
)
)
return _weighted_covariance(x, y, weights, denominator)

def corr(
self,
Expand Down Expand Up @@ -578,20 +620,7 @@ def corr(
if pair is None:
return np.nan
x, y, weights, _ = pair
if not np.isfinite(x).all() or not np.isfinite(y).all():
return np.nan
# A weighted mean can round away from identical decimal inputs.
# Check the retained observations exactly before subtracting it.
if (x == x[0]).all() or (y == y[0]).all():
return np.nan
x, _ = _weighted_centered_vector(x, weights)
y, _ = _weighted_centered_vector(y, weights)
x_ss = np.sum(x * x)
y_ss = np.sum(y * y)
if x_ss == 0 or y_ss == 0:
return np.nan
result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss))
return float(np.clip(result, -1.0, 1.0))
return _weighted_correlation(x, y, weights)

def quantile(self, q: np.array, skipna: bool = True) -> pd.Series:
"""Calculates weighted quantiles of the MicroSeries.
Expand Down
Loading
Loading