From 85d0b2f1e49d6615fcc392a31ab1336997fb8d84 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Fri, 18 Sep 2026 12:31:59 +0100 Subject: [PATCH] Weight MicroDataFrame.cov() and .corr() MicroDataFrame.cov() and .corr() delegated to pandas on the unweighted frame, so a weighted frame disagreed with its own columns. Each cell now uses the MicroSeries estimator with the frame's weights over the rows where both columns are present. The pair numerics move to module-level helpers shared by both classes. corr() accepts only method="pearson", as on MicroSeries. Fixes #327 Co-Authored-By: Claude Fable 5.1 --- .../weighted-dataframe-cov-corr.fixed.md | 1 + microdf/microdataframe.py | 111 ++++++++++++-- microdf/microseries.py | 137 ++++++++++------- microdf/tests/test_weight_propagation.py | 143 +++++++++++------- microdf/tests/test_weighted_cov_corr.py | 105 +++++++++++++ 5 files changed, 379 insertions(+), 118 deletions(-) create mode 100644 changelog.d/weighted-dataframe-cov-corr.fixed.md diff --git a/changelog.d/weighted-dataframe-cov-corr.fixed.md b/changelog.d/weighted-dataframe-cov-corr.fixed.md new file mode 100644 index 00000000..fd7a40f6 --- /dev/null +++ b/changelog.d/weighted-dataframe-cov-corr.fixed.md @@ -0,0 +1 @@ +`MicroDataFrame.cov()` and `.corr()` now return frequency-weighted pairwise matrices using the same estimator as `MicroSeries.cov` and `.corr`, instead of unweighted pandas results. As on `MicroSeries`, `corr` accepts only `method="pearson"`. diff --git a/microdf/microdataframe.py b/microdf/microdataframe.py index e756d44e..76858a08 100644 --- a/microdf/microdataframe.py +++ b/microdf/microdataframe.py @@ -7,7 +7,15 @@ import numpy as np import pandas as pd -from microdf.microseries import MicroSeries, MicroSeriesGroupBy +from microdf.microseries import ( + MicroSeries, + MicroSeriesGroupBy, + _pair_options, + _usable_pair, + _validate_frequency_weights, + _weighted_correlation, + _weighted_covariance, +) from microdf._weights import ( WeightPropagationMixin, aligned_weights, @@ -74,16 +82,99 @@ def __finalize__(self, other, method=None, **kwargs): super().__finalize__(other, method=method, **kwargs) return finalize_weights(self, other, method, previous) - @wraps(pd.DataFrame.cov) - def cov(self, *args, **kwargs) -> pd.DataFrame: - # Column summaries have no observation weights, even if labels match. - result = pd.DataFrame(self, copy=False).cov(*args, **kwargs) - return result.__finalize__(self, method="cov") + def cov( + self, + min_periods: Optional[int] = None, + ddof: int = 1, + numeric_only: bool = False, + ) -> pd.DataFrame: + """Pairwise frequency-weighted covariance of the columns. + + Every cell uses the estimator of :meth:`MicroSeries.cov` with this + frame's weights: ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) - + ddof)`` over the rows where both columns are present and the weight + is positive. Integer weights therefore match ``pandas.DataFrame.cov`` + on the replicated sample. Missing values are removed pairwise, so + each cell can use a different set of rows, as in pandas. + + The result is a plain ``pandas.DataFrame``: it summarises columns, + so it carries no row weights. + + :param min_periods: Minimum usable row pairs per cell, not the sum + of frequency weights. Cells with fewer are NaN. Defaults to 1. + :param ddof: Degrees of freedom subtracted from the weight total. + :param numeric_only: Use only numeric columns. Otherwise every column + is converted to float, raising the error pandas raises. + :returns: Covariance matrix indexed by column in both directions. + """ + return self._weighted_pairwise( + "cov", min_periods=min_periods, ddof=ddof, numeric_only=numeric_only + ) - @wraps(pd.DataFrame.corr) - def corr(self, *args, **kwargs) -> pd.DataFrame: - result = pd.DataFrame(self, copy=False).corr(*args, **kwargs) - return result.__finalize__(self, method="corr") + def corr( + self, + method: str = "pearson", + min_periods: int = 1, + numeric_only: bool = False, + ) -> pd.DataFrame: + """Pairwise frequency-weighted Pearson correlation of the columns. + + Every cell uses the estimator of :meth:`MicroSeries.corr` with this + frame's weights over the rows where both columns are present and the + weight is positive. Constant columns give NaN. Only ``"pearson"`` is + supported, as on :meth:`MicroSeries.corr`; for an unweighted rank + correlation convert to ``pandas.DataFrame`` first. + + The result is a plain ``pandas.DataFrame``: it summarises columns, so + it carries no row weights. + + :param method: Only "pearson" is supported. + :param min_periods: Minimum usable row pairs per cell, not the sum of + frequency weights. Cells with fewer are NaN. + :param numeric_only: Use only numeric columns. Otherwise every column + is converted to float, raising the error pandas raises. + :returns: Correlation matrix indexed by column in both directions. + """ + if method != "pearson": + raise ValueError("weighted correlation only supports method='pearson'") + return self._weighted_pairwise( + "corr", min_periods=min_periods, ddof=1, numeric_only=numeric_only + ) + + def _weighted_pairwise( + self, + statistic: str, + *, + min_periods: Optional[int], + ddof: int, + numeric_only: bool, + ) -> pd.DataFrame: + """Fill a symmetric column matrix one usable pair at a time.""" + min_periods, ddof = _pair_options(min_periods, ddof) + data = self._get_numeric_data() if numeric_only else self + frame = pd.DataFrame(data, copy=False) + # The conversion pandas uses, so non-numeric columns raise its error. + values = frame.to_numpy(dtype=float, na_value=np.nan) + weights = np.asarray(self.weights, dtype=float) + _validate_frequency_weights(weights) + columns = frame.columns + matrix = np.full((len(columns), len(columns)), np.nan) + for i in range(len(columns)): + for j in range(i, len(columns)): + pair = _usable_pair( + values[:, i], values[:, j], weights, min_periods, ddof, True + ) + if pair is None: + continue + x, y, pair_weights, denominator = pair + if statistic == "cov": + cell = _weighted_covariance(x, y, pair_weights, denominator) + else: + cell = _weighted_correlation(x, y, pair_weights) + matrix[i, j] = matrix[j, i] = cell + result = pd.DataFrame(matrix, index=columns, columns=columns) + # Column summaries have no observation weights, even if labels match. + return pd.DataFrame.__finalize__(result, self, method=statistic) def __setstate__(self, state) -> None: """Restore a pickled MicroDataFrame. diff --git a/microdf/microseries.py b/microdf/microseries.py index 1ddd458a..df89376b 100644 --- a/microdf/microseries.py +++ b/microdf/microseries.py @@ -51,6 +51,84 @@ def _weighted_centered_vector( return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift) +def _pair_options(min_periods: Optional[int], ddof: int) -> tuple[int, int]: + """Validate the row minimum and degrees of freedom for paired moments.""" + if not isinstance(ddof, (int, np.integer)): + raise TypeError("ddof must be an integer") + if min_periods is None: + min_periods = 1 + if not isinstance(min_periods, (int, np.integer)) or min_periods < 0: + raise ValueError("min_periods must be a nonnegative integer") + return int(min_periods), int(ddof) + + +def _validate_frequency_weights(weights: np.ndarray) -> None: + """Reject weights that cannot be frequencies.""" + if not np.isfinite(weights).all() or (weights < 0).any(): + raise ValueError("frequency weights must be finite and nonnegative") + + +def _usable_pair( + x: np.ndarray, + y: np.ndarray, + weights: np.ndarray, + min_periods: int, + ddof: int, + skipna: bool, +) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]: + """Keep the present, positively weighted rows of an aligned pair.""" + # Zero frequency means the row is absent, including for skipna=False. + positive = weights > 0 + x, y, weights = x[positive], y[positive], weights[positive] + missing = np.isnan(x) | np.isnan(y) + if not skipna and missing.any(): + return None + x, y, weights = x[~missing], y[~missing], weights[~missing] + total_weight = weights.sum() + if not np.isfinite(total_weight): + raise ValueError("the sum of frequency weights must be finite") + if len(x) < min_periods or total_weight == 0 or total_weight <= ddof: + return None + return x, y, weights, float(total_weight - ddof) + + +def _weighted_covariance( + x: np.ndarray, y: np.ndarray, weights: np.ndarray, denominator: float +) -> float: + """Frequency-weighted covariance of a usable pair.""" + if not np.isfinite(x).all() or not np.isfinite(y).all(): + return np.nan + x, x_exponent = _weighted_centered_vector(x, weights) + y, y_exponent = _weighted_centered_vector(y, weights) + # Combine exponents only after dividing out sum(weights) - ddof. + # Neither the original squared scale nor raw weighted sum need fit. + denominator, denominator_exponent = np.frexp(denominator) + return float( + np.ldexp( + np.sum(x * y) / denominator, + x_exponent + y_exponent - int(denominator_exponent), + ) + ) + + +def _weighted_correlation(x: np.ndarray, y: np.ndarray, weights: np.ndarray) -> float: + """Frequency-weighted Pearson correlation of a usable pair.""" + if not np.isfinite(x).all() or not np.isfinite(y).all(): + return np.nan + # A weighted mean can round away from identical decimal inputs. + # Check the retained observations exactly before subtracting it. + if (x == x[0]).all() or (y == y[0]).all(): + return np.nan + x, _ = _weighted_centered_vector(x, weights) + y, _ = _weighted_centered_vector(y, weights) + x_ss = np.sum(x * x) + y_ss = np.sum(y * y) + if x_ss == 0 or y_ss == 0: + return np.nan + result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss)) + return float(np.clip(result, -1.0, 1.0)) + + def _weighted_top_share( values: np.ndarray, weights: np.ndarray, top_x_pct: float ) -> float: @@ -456,12 +534,7 @@ def _weighted_pair( """Align usable paired observations with their left weights.""" if not isinstance(other, pd.Series): raise TypeError("other must be a pandas Series or MicroSeries") - if not isinstance(ddof, (int, np.integer)): - raise TypeError("ddof must be an integer") - if min_periods is None: - min_periods = 1 - if not isinstance(min_periods, (int, np.integer)) or min_periods < 0: - raise ValueError("min_periods must be a nonnegative integer") + min_periods, ddof = _pair_options(min_periods, ddof) if len(self) == 0 or len(other) == 0: return None @@ -477,27 +550,8 @@ def _weighted_pair( ) y = right.to_numpy(dtype=float, na_value=np.nan) weights = np.asarray(self.weights, dtype=float)[positions] - if not np.isfinite(weights).all() or (weights < 0).any(): - raise ValueError("frequency weights must be finite and nonnegative") - - # Zero frequency means the row is absent, including for skipna=False. - positive = weights > 0 - x, y, weights = x[positive], y[positive], weights[positive] - missing = np.isnan(x) | np.isnan(y) - if not skipna and missing.any(): - return None - x, y, weights = x[~missing], y[~missing], weights[~missing] - total_weight = weights.sum() - if not np.isfinite(total_weight): - raise ValueError("the sum of frequency weights must be finite") - if len(x) < min_periods or total_weight == 0 or total_weight <= ddof: - return None - return ( - x, - y, - weights, - float(total_weight - ddof), - ) + _validate_frequency_weights(weights) + return _usable_pair(x, y, weights, min_periods, ddof, skipna) def cov( self, @@ -531,19 +585,7 @@ def cov( if pair is None: return np.nan x, y, weights, denominator = pair - if not np.isfinite(x).all() or not np.isfinite(y).all(): - return np.nan - x, x_exponent = _weighted_centered_vector(x, weights) - y, y_exponent = _weighted_centered_vector(y, weights) - # Combine exponents only after dividing out sum(weights) - ddof. - # Neither the original squared scale nor raw weighted sum need fit. - denominator, denominator_exponent = np.frexp(denominator) - return float( - np.ldexp( - np.sum(x * y) / denominator, - x_exponent + y_exponent - int(denominator_exponent), - ) - ) + return _weighted_covariance(x, y, weights, denominator) def corr( self, @@ -578,20 +620,7 @@ def corr( if pair is None: return np.nan x, y, weights, _ = pair - if not np.isfinite(x).all() or not np.isfinite(y).all(): - return np.nan - # A weighted mean can round away from identical decimal inputs. - # Check the retained observations exactly before subtracting it. - if (x == x[0]).all() or (y == y[0]).all(): - return np.nan - x, _ = _weighted_centered_vector(x, weights) - y, _ = _weighted_centered_vector(y, weights) - x_ss = np.sum(x * x) - y_ss = np.sum(y * y) - if x_ss == 0 or y_ss == 0: - return np.nan - result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss)) - return float(np.clip(result, -1.0, 1.0)) + return _weighted_correlation(x, y, weights) def quantile(self, q: np.array, skipna: bool = True) -> pd.Series: """Calculates weighted quantiles of the MicroSeries. diff --git a/microdf/tests/test_weight_propagation.py b/microdf/tests/test_weight_propagation.py index 51ee5acb..1df4fdbd 100644 --- a/microdf/tests/test_weight_propagation.py +++ b/microdf/tests/test_weight_propagation.py @@ -355,31 +355,51 @@ def test_series_to_dataframe_weights_follow_aligned_rows_and_explicit_override(m mdf.MicroDataFrame(data, index=[3, 99]) +def replicated(data, weights, index=None): + """Frequency-weight oracle: the plain frame with each row repeated.""" + plain = pd.DataFrame(data, index=index) + return plain.iloc[np.repeat(np.arange(len(plain)), weights)] + + +def pairwise(oracle, method, *args, **kwargs): + """Apply a pandas matrix summary one complete pair at a time. + + pandas honours ``ddof`` only through ``np.cov`` on complete data; once any + value is missing it falls back to a kernel that ignores it. Dropping the + missing rows per pair keeps the oracle on the ``np.cov`` path. + """ + columns = oracle.columns + out = pd.DataFrame(np.nan, index=columns, columns=columns, dtype=float) + for i, left in enumerate(columns): + for right in columns[i:]: + pair = oracle[[left, right]] if left != right else oracle[[left]] + cell = getattr(pair.dropna(), method)(*args, **kwargs).iloc[0, -1] + out.loc[left, right] = out.loc[right, left] = cell + return out + + @pytest.mark.parametrize("method", ["cov", "corr"]) @pytest.mark.parametrize("coincident_labels", [False, True]) -def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_labels): +def test_dataframe_matrix_summaries_are_plain_and_frequency_weighted( + method, coincident_labels +): if coincident_labels: - frame = mdf.MicroDataFrame( - {"x": [10.0, 20.0], "y": [4.0, 8.0]}, - index=["x", "y"], - weights=[2, 3], - ) - # Sample covariance divides centered cross-products by n - 1. - covariance = [[50.0, 20.0], [20.0, 8.0]] + data = {"x": [10.0, 20.0], "y": [4.0, 8.0]} + weights = [2, 3] + frame = mdf.MicroDataFrame(data, index=["x", "y"], weights=weights) + # Weighted means x = 16, y = 6.4; sum(w) - 1 = 4 in the denominator. + covariance = [[30.0, 12.0], [12.0, 4.8]] correlation = [[1.0, 1.0], [1.0, 1.0]] - else: - frame = mdf.MicroDataFrame( - {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}, - weights=[2, 3, 5], + expected = pd.DataFrame( + covariance if method == "cov" else correlation, + index=frame.columns, + columns=frame.columns, ) - # Centered x = [-10, 0, 10], y = [-2, 2, 0]; n - 1 = 2. - covariance = [[100.0, 10.0], [10.0, 4.0]] - correlation = [[1.0, 0.5], [0.5, 1.0]] - expected = pd.DataFrame( - covariance if method == "cov" else correlation, - index=frame.columns, - columns=frame.columns, - ) + else: + data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]} + weights = [2, 3, 5] + frame = mdf.MicroDataFrame(data, weights=weights) + expected = getattr(replicated(data, weights), method)() result = getattr(frame, method)() @@ -395,17 +415,15 @@ def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_ [ ("cov", (), {}), ("cov", (2, 0), {}), - ("cov", (), {"min_periods": 4, "ddof": 2}), + ("cov", (), {"min_periods": 2, "ddof": 2}), ("cov", (), {"min_periods": 2, "ddof": 0, "numeric_only": True}), ("corr", (), {}), ("corr", ("pearson", 2, True), {}), - ("corr", (), {"method": "spearman", "min_periods": 2}), - ("corr", (), {"min_periods": 4}), - ("corr", (), {"method": lambda x, y: np.dot(x, y), "min_periods": 2}), + ("corr", (), {"min_periods": 2}), ], ) @pytest.mark.parametrize("missing", [False, True]) -def test_dataframe_matrix_summaries_preserve_pandas_arguments( +def test_dataframe_matrix_summaries_follow_pandas_arguments( method, args, kwargs, missing ): data = { @@ -413,8 +431,11 @@ def test_dataframe_matrix_summaries_preserve_pandas_arguments( "y": [4.0, 8.0, np.nan if missing else 6.0, 9.0], "flag": [True, False, True, True], } - frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=[2, 3, 5, 7]) - expected = getattr(pd.DataFrame(data, index=frame.index), method)(*args, **kwargs) + weights = [2, 3, 5, 7] + # Duplicate labels: weights follow row position, never labels. + frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=weights) + oracle = replicated(data, weights, index=frame.index) + expected = pairwise(oracle, method, *args, **kwargs) result = getattr(frame, method)(*args, **kwargs) @@ -423,12 +444,38 @@ def test_dataframe_matrix_summaries_preserve_pandas_arguments( pd.testing.assert_series_equal(result.sum(), expected.sum()) +@pytest.mark.parametrize("method", ["cov", "corr"]) +def test_dataframe_matrix_summaries_min_periods_counts_usable_rows(method): + data = {"x": [10.0, 20.0, 30.0, 40.0], "y": [4.0, 8.0, np.nan, 9.0]} + frame = mdf.MicroDataFrame(data, weights=[20, 30, 50, 70]) + # x and y share three usable rows, whatever their weight total. + assert np.isfinite(getattr(frame, method)(min_periods=3).loc["x", "y"]) + result = getattr(frame, method)(min_periods=4) + assert np.isnan(result.loc["x", "y"]) and np.isnan(result.loc["y", "x"]) + assert np.isfinite(result.loc["x", "x"]) + + +@pytest.mark.parametrize( + "method", ["spearman", "kendall", lambda x, y: float(np.dot(x, y))] +) +def test_dataframe_correlation_rejects_other_methods(method): + frame = mdf.MicroDataFrame( + {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}, weights=[2, 3, 5] + ) + with pytest.raises(ValueError, match="pearson") as frame_error: + frame.corr(method=method) + with pytest.raises(ValueError) as series_error: + frame.x.corr(frame.y, method=method) + assert str(frame_error.value) == str(series_error.value) + + @pytest.mark.parametrize("method", ["cov", "corr"]) def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method): data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0], "label": ["a", "b", "c"]} - frame = mdf.MicroDataFrame(data, weights=[2, 3, 5]) + weights = [2, 3, 5] + frame = mdf.MicroDataFrame(data, weights=weights) plain = pd.DataFrame(data) - expected = getattr(plain, method)(numeric_only=True) + expected = getattr(replicated(data, weights), method)(numeric_only=True) result = getattr(frame, method)(numeric_only=True) @@ -442,37 +489,25 @@ def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method): assert str(microdf_error.value) == str(pandas_error.value) -def test_dataframe_correlation_preserves_optional_kendall_support(): - data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]} - frame = mdf.MicroDataFrame(data, weights=[2, 3, 5]) - try: - expected = pd.DataFrame(data).corr(method="kendall") - except ImportError as pandas_error: - # Kendall requires scipy; delegation preserves pandas' dependency error. - with pytest.raises(type(pandas_error)) as microdf_error: - frame.corr(method="kendall") - assert str(microdf_error.value) == str(pandas_error) - else: - result = frame.corr(method="kendall") - assert type(result) is pd.DataFrame - pd.testing.assert_frame_equal(result, expected) - - @pytest.mark.parametrize("method", ["cov", "corr"]) def test_dataframe_matrix_summaries_preserve_pandas_metadata(method): data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]} - frame = mdf.MicroDataFrame(data, weights=[2, 3, 5]) - plain = pd.DataFrame(data) - for source in [frame, plain]: - source.attrs = {"survey": {"year": 2026}} - source.flags.allows_duplicate_labels = False - source.columns.name = "measure" - expected = getattr(plain, method)() + weights = [2, 3, 5] + frame = mdf.MicroDataFrame(data, weights=weights) + frame.attrs = {"survey": {"year": 2026}} + frame.flags.allows_duplicate_labels = False + frame.columns.name = "measure" + expected = getattr(replicated(data, weights), method)() result = getattr(frame, method)() assert type(result) is pd.DataFrame - pd.testing.assert_frame_equal(result, expected) - assert result.attrs == expected.attrs + pd.testing.assert_frame_equal( + result, expected, check_flags=False, check_names=False + ) + assert result.attrs == {"survey": {"year": 2026}} + assert result.flags.allows_duplicate_labels is False + assert result.index.name == "measure" and result.columns.name == "measure" + # attrs are copied, not shared, with the source frame. result.attrs["survey"]["year"] = 2025 assert frame.attrs["survey"]["year"] == 2026 diff --git a/microdf/tests/test_weighted_cov_corr.py b/microdf/tests/test_weighted_cov_corr.py index 334ff3ae..3794861f 100644 --- a/microdf/tests/test_weighted_cov_corr.py +++ b/microdf/tests/test_weighted_cov_corr.py @@ -341,3 +341,108 @@ def test_cov_corr_center_weight_scaling_preserves_original_frequency_ddof( else: expected = exact_weighted_moments(x, y, weights, ddof=ddof) np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0) + + +def replicated_frame(data, weights): + """Frequency-weight oracle for a frame: repeat each row by its weight.""" + plain = pd.DataFrame(data) + return plain.iloc[np.repeat(np.arange(len(plain)), weights)] + + +@pytest.mark.parametrize("ddof", [0, 1, 2]) +def test_dataframe_cov_corr_match_replicated_frequency_sample(ddof): + data = {"x": [1.0, 4.0, 8.0, 2.0], "y": [5.0, 2.0, 9.0, 7.0], "z": [3, 3, 1, 6]} + weights = [1, 3, 2, 4] + frame = mdf.MicroDataFrame(data, weights=weights) + oracle = replicated_frame(data, weights) + pd.testing.assert_frame_equal(frame.cov(ddof=ddof), oracle.cov(ddof=ddof)) + pd.testing.assert_frame_equal(frame.corr(), oracle.corr()) + + +def test_dataframe_cov_corr_cells_equal_microseries_results(): + frame = mdf.MicroDataFrame( + { + "a": [1.0, 4.0, 8.0, 3.0, 6.0], + "b": [5.0, 2.0, 9.0, np.nan, 1.0], + "c": [2.0, 2.0, 7.0, 1.0, 1e9], + }, + weights=[1, 3, 2, 0, 0.5], + ) + cov, corr = frame.cov(), frame.corr() + for left in frame.columns: + for right in frame.columns: + assert cov.loc[left, right] == frame[left].cov(frame[right]) + assert corr.loc[left, right] == frame[left].corr(frame[right]) + assert cov.loc[left, right] == cov.loc[right, left] + + +def test_dataframe_cov_corr_use_pairwise_complete_rows(): + frame = mdf.MicroDataFrame( + { + "x": [1.0, np.nan, 5.0, 9.0, 20.0], + "y": [2.0, 4.0, np.nan, 8.0, 30.0], + "z": [1.0, 1.0, 2.0, 3.0, 5.0], + }, + weights=[1, 9, 2, 3, 7], + ) + cov, corr = frame.cov(), frame.corr() + # x-y keeps rows 0, 3, 4; x-z rows 0, 2, 3, 4; y-z rows 0, 1, 3, 4. + cells = { + ("x", "y"): replicated_moments([1, 9, 20], [2, 8, 30], [1, 3, 7]), + ("x", "z"): replicated_moments([1, 5, 9, 20], [1, 2, 3, 5], [1, 2, 3, 7]), + ("y", "z"): replicated_moments([2, 4, 8, 30], [1, 1, 3, 5], [1, 9, 3, 7]), + } + for (left, right), (expected_cov, expected_corr) in cells.items(): + assert cov.loc[left, right] == pytest.approx(expected_cov) + assert corr.loc[left, right] == pytest.approx(expected_corr) + + +def test_dataframe_cov_corr_omit_zero_weight_rows(): + frame = mdf.MicroDataFrame( + {"x": [1.0, 4.0, 8.0, 1e6], "y": [5.0, 2.0, 9.0, -1e6]}, weights=[1, 3, 2, 0] + ) + expected_cov, expected_corr = replicated_moments([1, 4, 8], [5, 2, 9], [1, 3, 2]) + assert frame.cov().loc["x", "y"] == pytest.approx(expected_cov) + assert frame.corr().loc["x", "y"] == pytest.approx(expected_corr) + + +@pytest.mark.parametrize("weights", [[1, -1, 2], [1, np.inf, 2], [1, np.nan, 2]]) +def test_dataframe_cov_corr_reject_invalid_frequency_weights(weights): + frame = mdf.MicroDataFrame( + {"x": [1.0, 2.0, 3.0], "y": [3.0, 1.0, 2.0]}, weights=weights + ) + with pytest.raises(ValueError, match="finite and nonnegative"): + frame.cov() + with pytest.raises(ValueError, match="finite and nonnegative"): + frame.corr() + + +def test_dataframe_cov_corr_diagonal_and_constant_columns(): + frame = mdf.MicroDataFrame( + {"x": [1.0, 4.0, 8.0], "c": [2.0, 2.0, 2.0]}, weights=[1, 3, 2] + ) + cov, corr = frame.cov(), frame.corr() + assert cov.loc["x", "x"] == pytest.approx(frame.x.var()) + assert cov.loc["c", "c"] == 0 and cov.loc["x", "c"] == 0 + assert corr.loc["x", "x"] == 1.0 + assert np.isnan(corr.loc["c", "c"]) and np.isnan(corr.loc["x", "c"]) + + +def test_dataframe_cov_corr_insufficient_weight_total_and_empty_frame(): + frame = mdf.MicroDataFrame({"x": [1.0, 4.0], "y": [5.0, 2.0]}, weights=[1, 1]) + assert np.isnan(frame.cov(ddof=2)).all().all() + assert np.isfinite(frame.cov(ddof=1)).all().all() + empty = mdf.MicroDataFrame({"x": [], "y": []}, weights=[]) + for result in (empty.cov(), empty.corr()): + assert list(result.columns) == ["x", "y"] + assert np.isnan(result).all().all() + + +def test_dataframe_cov_corr_validate_options_like_microseries(): + frame = mdf.MicroDataFrame({"x": [1.0, 2.0], "y": [2.0, 1.0]}, weights=[1, 1]) + with pytest.raises(TypeError, match="ddof"): + frame.cov(ddof=1.5) + with pytest.raises(ValueError, match="min_periods"): + frame.cov(min_periods=-1) + with pytest.raises(ValueError, match="min_periods"): + frame.corr(min_periods=-1)