diff --git a/changelog.d/weighted-dataframe-cov-corr.fixed.md b/changelog.d/weighted-dataframe-cov-corr.fixed.md new file mode 100644 index 00000000..fd7a40f6 --- /dev/null +++ b/changelog.d/weighted-dataframe-cov-corr.fixed.md @@ -0,0 +1 @@ +`MicroDataFrame.cov()` and `.corr()` now return frequency-weighted pairwise matrices using the same estimator as `MicroSeries.cov` and `.corr`, instead of unweighted pandas results. As on `MicroSeries`, `corr` accepts only `method="pearson"`. diff --git a/microdf/microdataframe.py b/microdf/microdataframe.py index e756d44e..76858a08 100644 --- a/microdf/microdataframe.py +++ b/microdf/microdataframe.py @@ -7,7 +7,15 @@ import numpy as np import pandas as pd -from microdf.microseries import MicroSeries, MicroSeriesGroupBy +from microdf.microseries import ( + MicroSeries, + MicroSeriesGroupBy, + _pair_options, + _usable_pair, + _validate_frequency_weights, + _weighted_correlation, + _weighted_covariance, +) from microdf._weights import ( WeightPropagationMixin, aligned_weights, @@ -74,16 +82,99 @@ def __finalize__(self, other, method=None, **kwargs): super().__finalize__(other, method=method, **kwargs) return finalize_weights(self, other, method, previous) - @wraps(pd.DataFrame.cov) - def cov(self, *args, **kwargs) -> pd.DataFrame: - # Column summaries have no observation weights, even if labels match. - result = pd.DataFrame(self, copy=False).cov(*args, **kwargs) - return result.__finalize__(self, method="cov") + def cov( + self, + min_periods: Optional[int] = None, + ddof: int = 1, + numeric_only: bool = False, + ) -> pd.DataFrame: + """Pairwise frequency-weighted covariance of the columns. + + Every cell uses the estimator of :meth:`MicroSeries.cov` with this + frame's weights: ``sum(w * (x - xmean) * (y - ymean)) / (sum(w) - + ddof)`` over the rows where both columns are present and the weight + is positive. Integer weights therefore match ``pandas.DataFrame.cov`` + on the replicated sample. Missing values are removed pairwise, so + each cell can use a different set of rows, as in pandas. + + The result is a plain ``pandas.DataFrame``: it summarises columns, + so it carries no row weights. + + :param min_periods: Minimum usable row pairs per cell, not the sum + of frequency weights. Cells with fewer are NaN. Defaults to 1. + :param ddof: Degrees of freedom subtracted from the weight total. + :param numeric_only: Use only numeric columns. Otherwise every column + is converted to float, raising the error pandas raises. + :returns: Covariance matrix indexed by column in both directions. + """ + return self._weighted_pairwise( + "cov", min_periods=min_periods, ddof=ddof, numeric_only=numeric_only + ) - @wraps(pd.DataFrame.corr) - def corr(self, *args, **kwargs) -> pd.DataFrame: - result = pd.DataFrame(self, copy=False).corr(*args, **kwargs) - return result.__finalize__(self, method="corr") + def corr( + self, + method: str = "pearson", + min_periods: int = 1, + numeric_only: bool = False, + ) -> pd.DataFrame: + """Pairwise frequency-weighted Pearson correlation of the columns. + + Every cell uses the estimator of :meth:`MicroSeries.corr` with this + frame's weights over the rows where both columns are present and the + weight is positive. Constant columns give NaN. Only ``"pearson"`` is + supported, as on :meth:`MicroSeries.corr`; for an unweighted rank + correlation convert to ``pandas.DataFrame`` first. + + The result is a plain ``pandas.DataFrame``: it summarises columns, so + it carries no row weights. + + :param method: Only "pearson" is supported. + :param min_periods: Minimum usable row pairs per cell, not the sum of + frequency weights. Cells with fewer are NaN. + :param numeric_only: Use only numeric columns. Otherwise every column + is converted to float, raising the error pandas raises. + :returns: Correlation matrix indexed by column in both directions. + """ + if method != "pearson": + raise ValueError("weighted correlation only supports method='pearson'") + return self._weighted_pairwise( + "corr", min_periods=min_periods, ddof=1, numeric_only=numeric_only + ) + + def _weighted_pairwise( + self, + statistic: str, + *, + min_periods: Optional[int], + ddof: int, + numeric_only: bool, + ) -> pd.DataFrame: + """Fill a symmetric column matrix one usable pair at a time.""" + min_periods, ddof = _pair_options(min_periods, ddof) + data = self._get_numeric_data() if numeric_only else self + frame = pd.DataFrame(data, copy=False) + # The conversion pandas uses, so non-numeric columns raise its error. + values = frame.to_numpy(dtype=float, na_value=np.nan) + weights = np.asarray(self.weights, dtype=float) + _validate_frequency_weights(weights) + columns = frame.columns + matrix = np.full((len(columns), len(columns)), np.nan) + for i in range(len(columns)): + for j in range(i, len(columns)): + pair = _usable_pair( + values[:, i], values[:, j], weights, min_periods, ddof, True + ) + if pair is None: + continue + x, y, pair_weights, denominator = pair + if statistic == "cov": + cell = _weighted_covariance(x, y, pair_weights, denominator) + else: + cell = _weighted_correlation(x, y, pair_weights) + matrix[i, j] = matrix[j, i] = cell + result = pd.DataFrame(matrix, index=columns, columns=columns) + # Column summaries have no observation weights, even if labels match. + return pd.DataFrame.__finalize__(result, self, method=statistic) def __setstate__(self, state) -> None: """Restore a pickled MicroDataFrame. diff --git a/microdf/microseries.py b/microdf/microseries.py index 1ddd458a..df89376b 100644 --- a/microdf/microseries.py +++ b/microdf/microseries.py @@ -51,6 +51,84 @@ def _weighted_centered_vector( return np.ldexp(shifted, -weight_shift), exponent + int(shift) + int(weight_shift) +def _pair_options(min_periods: Optional[int], ddof: int) -> tuple[int, int]: + """Validate the row minimum and degrees of freedom for paired moments.""" + if not isinstance(ddof, (int, np.integer)): + raise TypeError("ddof must be an integer") + if min_periods is None: + min_periods = 1 + if not isinstance(min_periods, (int, np.integer)) or min_periods < 0: + raise ValueError("min_periods must be a nonnegative integer") + return int(min_periods), int(ddof) + + +def _validate_frequency_weights(weights: np.ndarray) -> None: + """Reject weights that cannot be frequencies.""" + if not np.isfinite(weights).all() or (weights < 0).any(): + raise ValueError("frequency weights must be finite and nonnegative") + + +def _usable_pair( + x: np.ndarray, + y: np.ndarray, + weights: np.ndarray, + min_periods: int, + ddof: int, + skipna: bool, +) -> Optional[tuple[np.ndarray, np.ndarray, np.ndarray, float]]: + """Keep the present, positively weighted rows of an aligned pair.""" + # Zero frequency means the row is absent, including for skipna=False. + positive = weights > 0 + x, y, weights = x[positive], y[positive], weights[positive] + missing = np.isnan(x) | np.isnan(y) + if not skipna and missing.any(): + return None + x, y, weights = x[~missing], y[~missing], weights[~missing] + total_weight = weights.sum() + if not np.isfinite(total_weight): + raise ValueError("the sum of frequency weights must be finite") + if len(x) < min_periods or total_weight == 0 or total_weight <= ddof: + return None + return x, y, weights, float(total_weight - ddof) + + +def _weighted_covariance( + x: np.ndarray, y: np.ndarray, weights: np.ndarray, denominator: float +) -> float: + """Frequency-weighted covariance of a usable pair.""" + if not np.isfinite(x).all() or not np.isfinite(y).all(): + return np.nan + x, x_exponent = _weighted_centered_vector(x, weights) + y, y_exponent = _weighted_centered_vector(y, weights) + # Combine exponents only after dividing out sum(weights) - ddof. + # Neither the original squared scale nor raw weighted sum need fit. + denominator, denominator_exponent = np.frexp(denominator) + return float( + np.ldexp( + np.sum(x * y) / denominator, + x_exponent + y_exponent - int(denominator_exponent), + ) + ) + + +def _weighted_correlation(x: np.ndarray, y: np.ndarray, weights: np.ndarray) -> float: + """Frequency-weighted Pearson correlation of a usable pair.""" + if not np.isfinite(x).all() or not np.isfinite(y).all(): + return np.nan + # A weighted mean can round away from identical decimal inputs. + # Check the retained observations exactly before subtracting it. + if (x == x[0]).all() or (y == y[0]).all(): + return np.nan + x, _ = _weighted_centered_vector(x, weights) + y, _ = _weighted_centered_vector(y, weights) + x_ss = np.sum(x * x) + y_ss = np.sum(y * y) + if x_ss == 0 or y_ss == 0: + return np.nan + result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss)) + return float(np.clip(result, -1.0, 1.0)) + + def _weighted_top_share( values: np.ndarray, weights: np.ndarray, top_x_pct: float ) -> float: @@ -456,12 +534,7 @@ def _weighted_pair( """Align usable paired observations with their left weights.""" if not isinstance(other, pd.Series): raise TypeError("other must be a pandas Series or MicroSeries") - if not isinstance(ddof, (int, np.integer)): - raise TypeError("ddof must be an integer") - if min_periods is None: - min_periods = 1 - if not isinstance(min_periods, (int, np.integer)) or min_periods < 0: - raise ValueError("min_periods must be a nonnegative integer") + min_periods, ddof = _pair_options(min_periods, ddof) if len(self) == 0 or len(other) == 0: return None @@ -477,27 +550,8 @@ def _weighted_pair( ) y = right.to_numpy(dtype=float, na_value=np.nan) weights = np.asarray(self.weights, dtype=float)[positions] - if not np.isfinite(weights).all() or (weights < 0).any(): - raise ValueError("frequency weights must be finite and nonnegative") - - # Zero frequency means the row is absent, including for skipna=False. - positive = weights > 0 - x, y, weights = x[positive], y[positive], weights[positive] - missing = np.isnan(x) | np.isnan(y) - if not skipna and missing.any(): - return None - x, y, weights = x[~missing], y[~missing], weights[~missing] - total_weight = weights.sum() - if not np.isfinite(total_weight): - raise ValueError("the sum of frequency weights must be finite") - if len(x) < min_periods or total_weight == 0 or total_weight <= ddof: - return None - return ( - x, - y, - weights, - float(total_weight - ddof), - ) + _validate_frequency_weights(weights) + return _usable_pair(x, y, weights, min_periods, ddof, skipna) def cov( self, @@ -531,19 +585,7 @@ def cov( if pair is None: return np.nan x, y, weights, denominator = pair - if not np.isfinite(x).all() or not np.isfinite(y).all(): - return np.nan - x, x_exponent = _weighted_centered_vector(x, weights) - y, y_exponent = _weighted_centered_vector(y, weights) - # Combine exponents only after dividing out sum(weights) - ddof. - # Neither the original squared scale nor raw weighted sum need fit. - denominator, denominator_exponent = np.frexp(denominator) - return float( - np.ldexp( - np.sum(x * y) / denominator, - x_exponent + y_exponent - int(denominator_exponent), - ) - ) + return _weighted_covariance(x, y, weights, denominator) def corr( self, @@ -578,20 +620,7 @@ def corr( if pair is None: return np.nan x, y, weights, _ = pair - if not np.isfinite(x).all() or not np.isfinite(y).all(): - return np.nan - # A weighted mean can round away from identical decimal inputs. - # Check the retained observations exactly before subtracting it. - if (x == x[0]).all() or (y == y[0]).all(): - return np.nan - x, _ = _weighted_centered_vector(x, weights) - y, _ = _weighted_centered_vector(y, weights) - x_ss = np.sum(x * x) - y_ss = np.sum(y * y) - if x_ss == 0 or y_ss == 0: - return np.nan - result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss)) - return float(np.clip(result, -1.0, 1.0)) + return _weighted_correlation(x, y, weights) def quantile(self, q: np.array, skipna: bool = True) -> pd.Series: """Calculates weighted quantiles of the MicroSeries. diff --git a/microdf/tests/test_weight_propagation.py b/microdf/tests/test_weight_propagation.py index 51ee5acb..1df4fdbd 100644 --- a/microdf/tests/test_weight_propagation.py +++ b/microdf/tests/test_weight_propagation.py @@ -355,31 +355,51 @@ def test_series_to_dataframe_weights_follow_aligned_rows_and_explicit_override(m mdf.MicroDataFrame(data, index=[3, 99]) +def replicated(data, weights, index=None): + """Frequency-weight oracle: the plain frame with each row repeated.""" + plain = pd.DataFrame(data, index=index) + return plain.iloc[np.repeat(np.arange(len(plain)), weights)] + + +def pairwise(oracle, method, *args, **kwargs): + """Apply a pandas matrix summary one complete pair at a time. + + pandas honours ``ddof`` only through ``np.cov`` on complete data; once any + value is missing it falls back to a kernel that ignores it. Dropping the + missing rows per pair keeps the oracle on the ``np.cov`` path. + """ + columns = oracle.columns + out = pd.DataFrame(np.nan, index=columns, columns=columns, dtype=float) + for i, left in enumerate(columns): + for right in columns[i:]: + pair = oracle[[left, right]] if left != right else oracle[[left]] + cell = getattr(pair.dropna(), method)(*args, **kwargs).iloc[0, -1] + out.loc[left, right] = out.loc[right, left] = cell + return out + + @pytest.mark.parametrize("method", ["cov", "corr"]) @pytest.mark.parametrize("coincident_labels", [False, True]) -def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_labels): +def test_dataframe_matrix_summaries_are_plain_and_frequency_weighted( + method, coincident_labels +): if coincident_labels: - frame = mdf.MicroDataFrame( - {"x": [10.0, 20.0], "y": [4.0, 8.0]}, - index=["x", "y"], - weights=[2, 3], - ) - # Sample covariance divides centered cross-products by n - 1. - covariance = [[50.0, 20.0], [20.0, 8.0]] + data = {"x": [10.0, 20.0], "y": [4.0, 8.0]} + weights = [2, 3] + frame = mdf.MicroDataFrame(data, index=["x", "y"], weights=weights) + # Weighted means x = 16, y = 6.4; sum(w) - 1 = 4 in the denominator. + covariance = [[30.0, 12.0], [12.0, 4.8]] correlation = [[1.0, 1.0], [1.0, 1.0]] - else: - frame = mdf.MicroDataFrame( - {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}, - weights=[2, 3, 5], + expected = pd.DataFrame( + covariance if method == "cov" else correlation, + index=frame.columns, + columns=frame.columns, ) - # Centered x = [-10, 0, 10], y = [-2, 2, 0]; n - 1 = 2. - covariance = [[100.0, 10.0], [10.0, 4.0]] - correlation = [[1.0, 0.5], [0.5, 1.0]] - expected = pd.DataFrame( - covariance if method == "cov" else correlation, - index=frame.columns, - columns=frame.columns, - ) + else: + data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]} + weights = [2, 3, 5] + frame = mdf.MicroDataFrame(data, weights=weights) + expected = getattr(replicated(data, weights), method)() result = getattr(frame, method)() @@ -395,17 +415,15 @@ def test_dataframe_matrix_summaries_are_plain_and_unweighted(method, coincident_ [ ("cov", (), {}), ("cov", (2, 0), {}), - ("cov", (), {"min_periods": 4, "ddof": 2}), + ("cov", (), {"min_periods": 2, "ddof": 2}), ("cov", (), {"min_periods": 2, "ddof": 0, "numeric_only": True}), ("corr", (), {}), ("corr", ("pearson", 2, True), {}), - ("corr", (), {"method": "spearman", "min_periods": 2}), - ("corr", (), {"min_periods": 4}), - ("corr", (), {"method": lambda x, y: np.dot(x, y), "min_periods": 2}), + ("corr", (), {"min_periods": 2}), ], ) @pytest.mark.parametrize("missing", [False, True]) -def test_dataframe_matrix_summaries_preserve_pandas_arguments( +def test_dataframe_matrix_summaries_follow_pandas_arguments( method, args, kwargs, missing ): data = { @@ -413,8 +431,11 @@ def test_dataframe_matrix_summaries_preserve_pandas_arguments( "y": [4.0, 8.0, np.nan if missing else 6.0, 9.0], "flag": [True, False, True, True], } - frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=[2, 3, 5, 7]) - expected = getattr(pd.DataFrame(data, index=frame.index), method)(*args, **kwargs) + weights = [2, 3, 5, 7] + # Duplicate labels: weights follow row position, never labels. + frame = mdf.MicroDataFrame(data, index=[7, 7, 3, 9], weights=weights) + oracle = replicated(data, weights, index=frame.index) + expected = pairwise(oracle, method, *args, **kwargs) result = getattr(frame, method)(*args, **kwargs) @@ -423,12 +444,38 @@ def test_dataframe_matrix_summaries_preserve_pandas_arguments( pd.testing.assert_series_equal(result.sum(), expected.sum()) +@pytest.mark.parametrize("method", ["cov", "corr"]) +def test_dataframe_matrix_summaries_min_periods_counts_usable_rows(method): + data = {"x": [10.0, 20.0, 30.0, 40.0], "y": [4.0, 8.0, np.nan, 9.0]} + frame = mdf.MicroDataFrame(data, weights=[20, 30, 50, 70]) + # x and y share three usable rows, whatever their weight total. + assert np.isfinite(getattr(frame, method)(min_periods=3).loc["x", "y"]) + result = getattr(frame, method)(min_periods=4) + assert np.isnan(result.loc["x", "y"]) and np.isnan(result.loc["y", "x"]) + assert np.isfinite(result.loc["x", "x"]) + + +@pytest.mark.parametrize( + "method", ["spearman", "kendall", lambda x, y: float(np.dot(x, y))] +) +def test_dataframe_correlation_rejects_other_methods(method): + frame = mdf.MicroDataFrame( + {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]}, weights=[2, 3, 5] + ) + with pytest.raises(ValueError, match="pearson") as frame_error: + frame.corr(method=method) + with pytest.raises(ValueError) as series_error: + frame.x.corr(frame.y, method=method) + assert str(frame_error.value) == str(series_error.value) + + @pytest.mark.parametrize("method", ["cov", "corr"]) def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method): data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0], "label": ["a", "b", "c"]} - frame = mdf.MicroDataFrame(data, weights=[2, 3, 5]) + weights = [2, 3, 5] + frame = mdf.MicroDataFrame(data, weights=weights) plain = pd.DataFrame(data) - expected = getattr(plain, method)(numeric_only=True) + expected = getattr(replicated(data, weights), method)(numeric_only=True) result = getattr(frame, method)(numeric_only=True) @@ -442,37 +489,25 @@ def test_dataframe_matrix_summaries_preserve_numeric_only_and_errors(method): assert str(microdf_error.value) == str(pandas_error.value) -def test_dataframe_correlation_preserves_optional_kendall_support(): - data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]} - frame = mdf.MicroDataFrame(data, weights=[2, 3, 5]) - try: - expected = pd.DataFrame(data).corr(method="kendall") - except ImportError as pandas_error: - # Kendall requires scipy; delegation preserves pandas' dependency error. - with pytest.raises(type(pandas_error)) as microdf_error: - frame.corr(method="kendall") - assert str(microdf_error.value) == str(pandas_error) - else: - result = frame.corr(method="kendall") - assert type(result) is pd.DataFrame - pd.testing.assert_frame_equal(result, expected) - - @pytest.mark.parametrize("method", ["cov", "corr"]) def test_dataframe_matrix_summaries_preserve_pandas_metadata(method): data = {"x": [10.0, 20.0, 30.0], "y": [4.0, 8.0, 6.0]} - frame = mdf.MicroDataFrame(data, weights=[2, 3, 5]) - plain = pd.DataFrame(data) - for source in [frame, plain]: - source.attrs = {"survey": {"year": 2026}} - source.flags.allows_duplicate_labels = False - source.columns.name = "measure" - expected = getattr(plain, method)() + weights = [2, 3, 5] + frame = mdf.MicroDataFrame(data, weights=weights) + frame.attrs = {"survey": {"year": 2026}} + frame.flags.allows_duplicate_labels = False + frame.columns.name = "measure" + expected = getattr(replicated(data, weights), method)() result = getattr(frame, method)() assert type(result) is pd.DataFrame - pd.testing.assert_frame_equal(result, expected) - assert result.attrs == expected.attrs + pd.testing.assert_frame_equal( + result, expected, check_flags=False, check_names=False + ) + assert result.attrs == {"survey": {"year": 2026}} + assert result.flags.allows_duplicate_labels is False + assert result.index.name == "measure" and result.columns.name == "measure" + # attrs are copied, not shared, with the source frame. result.attrs["survey"]["year"] = 2025 assert frame.attrs["survey"]["year"] == 2026 diff --git a/microdf/tests/test_weighted_cov_corr.py b/microdf/tests/test_weighted_cov_corr.py index 334ff3ae..3794861f 100644 --- a/microdf/tests/test_weighted_cov_corr.py +++ b/microdf/tests/test_weighted_cov_corr.py @@ -341,3 +341,108 @@ def test_cov_corr_center_weight_scaling_preserves_original_frequency_ddof( else: expected = exact_weighted_moments(x, y, weights, ddof=ddof) np.testing.assert_allclose(actual, expected, rtol=3e-15, atol=0) + + +def replicated_frame(data, weights): + """Frequency-weight oracle for a frame: repeat each row by its weight.""" + plain = pd.DataFrame(data) + return plain.iloc[np.repeat(np.arange(len(plain)), weights)] + + +@pytest.mark.parametrize("ddof", [0, 1, 2]) +def test_dataframe_cov_corr_match_replicated_frequency_sample(ddof): + data = {"x": [1.0, 4.0, 8.0, 2.0], "y": [5.0, 2.0, 9.0, 7.0], "z": [3, 3, 1, 6]} + weights = [1, 3, 2, 4] + frame = mdf.MicroDataFrame(data, weights=weights) + oracle = replicated_frame(data, weights) + pd.testing.assert_frame_equal(frame.cov(ddof=ddof), oracle.cov(ddof=ddof)) + pd.testing.assert_frame_equal(frame.corr(), oracle.corr()) + + +def test_dataframe_cov_corr_cells_equal_microseries_results(): + frame = mdf.MicroDataFrame( + { + "a": [1.0, 4.0, 8.0, 3.0, 6.0], + "b": [5.0, 2.0, 9.0, np.nan, 1.0], + "c": [2.0, 2.0, 7.0, 1.0, 1e9], + }, + weights=[1, 3, 2, 0, 0.5], + ) + cov, corr = frame.cov(), frame.corr() + for left in frame.columns: + for right in frame.columns: + assert cov.loc[left, right] == frame[left].cov(frame[right]) + assert corr.loc[left, right] == frame[left].corr(frame[right]) + assert cov.loc[left, right] == cov.loc[right, left] + + +def test_dataframe_cov_corr_use_pairwise_complete_rows(): + frame = mdf.MicroDataFrame( + { + "x": [1.0, np.nan, 5.0, 9.0, 20.0], + "y": [2.0, 4.0, np.nan, 8.0, 30.0], + "z": [1.0, 1.0, 2.0, 3.0, 5.0], + }, + weights=[1, 9, 2, 3, 7], + ) + cov, corr = frame.cov(), frame.corr() + # x-y keeps rows 0, 3, 4; x-z rows 0, 2, 3, 4; y-z rows 0, 1, 3, 4. + cells = { + ("x", "y"): replicated_moments([1, 9, 20], [2, 8, 30], [1, 3, 7]), + ("x", "z"): replicated_moments([1, 5, 9, 20], [1, 2, 3, 5], [1, 2, 3, 7]), + ("y", "z"): replicated_moments([2, 4, 8, 30], [1, 1, 3, 5], [1, 9, 3, 7]), + } + for (left, right), (expected_cov, expected_corr) in cells.items(): + assert cov.loc[left, right] == pytest.approx(expected_cov) + assert corr.loc[left, right] == pytest.approx(expected_corr) + + +def test_dataframe_cov_corr_omit_zero_weight_rows(): + frame = mdf.MicroDataFrame( + {"x": [1.0, 4.0, 8.0, 1e6], "y": [5.0, 2.0, 9.0, -1e6]}, weights=[1, 3, 2, 0] + ) + expected_cov, expected_corr = replicated_moments([1, 4, 8], [5, 2, 9], [1, 3, 2]) + assert frame.cov().loc["x", "y"] == pytest.approx(expected_cov) + assert frame.corr().loc["x", "y"] == pytest.approx(expected_corr) + + +@pytest.mark.parametrize("weights", [[1, -1, 2], [1, np.inf, 2], [1, np.nan, 2]]) +def test_dataframe_cov_corr_reject_invalid_frequency_weights(weights): + frame = mdf.MicroDataFrame( + {"x": [1.0, 2.0, 3.0], "y": [3.0, 1.0, 2.0]}, weights=weights + ) + with pytest.raises(ValueError, match="finite and nonnegative"): + frame.cov() + with pytest.raises(ValueError, match="finite and nonnegative"): + frame.corr() + + +def test_dataframe_cov_corr_diagonal_and_constant_columns(): + frame = mdf.MicroDataFrame( + {"x": [1.0, 4.0, 8.0], "c": [2.0, 2.0, 2.0]}, weights=[1, 3, 2] + ) + cov, corr = frame.cov(), frame.corr() + assert cov.loc["x", "x"] == pytest.approx(frame.x.var()) + assert cov.loc["c", "c"] == 0 and cov.loc["x", "c"] == 0 + assert corr.loc["x", "x"] == 1.0 + assert np.isnan(corr.loc["c", "c"]) and np.isnan(corr.loc["x", "c"]) + + +def test_dataframe_cov_corr_insufficient_weight_total_and_empty_frame(): + frame = mdf.MicroDataFrame({"x": [1.0, 4.0], "y": [5.0, 2.0]}, weights=[1, 1]) + assert np.isnan(frame.cov(ddof=2)).all().all() + assert np.isfinite(frame.cov(ddof=1)).all().all() + empty = mdf.MicroDataFrame({"x": [], "y": []}, weights=[]) + for result in (empty.cov(), empty.corr()): + assert list(result.columns) == ["x", "y"] + assert np.isnan(result).all().all() + + +def test_dataframe_cov_corr_validate_options_like_microseries(): + frame = mdf.MicroDataFrame({"x": [1.0, 2.0], "y": [2.0, 1.0]}, weights=[1, 1]) + with pytest.raises(TypeError, match="ddof"): + frame.cov(ddof=1.5) + with pytest.raises(ValueError, match="min_periods"): + frame.cov(min_periods=-1) + with pytest.raises(ValueError, match="min_periods"): + frame.corr(min_periods=-1)