From b1f938e25c54da98c05f069295380f2a1f951322 Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Fri, 18 Sep 2026 11:17:50 +0100 Subject: [PATCH 1/7] Add an API reference to the documentation The documentation had no API page, which is the largest gap a JOSS reviewer assessing documentation would find. Lists every weighted estimator and weight-handling method on both classes with its signature and a one-line description, taken from the docstrings, plus the estimator conventions that distinguish these from naive weighted equivalents. Registered in both docs/myst.yml and docs/_toc.yml, which are kept in step until the duplicate configs are reconciled. --- changelog.d/api-reference.added.md | 1 + docs/_toc.yml | 1 + docs/api.md | 94 ++++++++++++++++++++++++++++++ docs/myst.yml | 1 + 4 files changed, 97 insertions(+) create mode 100644 changelog.d/api-reference.added.md create mode 100644 docs/api.md diff --git a/changelog.d/api-reference.added.md b/changelog.d/api-reference.added.md new file mode 100644 index 00000000..be020b44 --- /dev/null +++ b/changelog.d/api-reference.added.md @@ -0,0 +1 @@ +Adds an API reference to the documentation. diff --git a/docs/_toc.yml b/docs/_toc.yml index d7ff41ff..56709778 100644 --- a/docs/_toc.yml +++ b/docs/_toc.yml @@ -4,3 +4,4 @@ sections: - file: examples sections: - file: gini +- file: api diff --git a/docs/api.md b/docs/api.md new file mode 100644 index 00000000..c6d90604 --- /dev/null +++ b/docs/api.md @@ -0,0 +1,94 @@ +# API reference + +`microdf` exposes two classes. `MicroSeries` is a `pandas.Series` carrying a weight +vector; `MicroDataFrame` is a `pandas.DataFrame` carrying a weight column. Both +behave like their pandas counterparts, and the methods below either add a +weighted estimator or preserve weights through an operation that would otherwise +drop them. + +```python +import microdf as mdf + +df = mdf.MicroDataFrame({"income": [10_000, 30_000, 120_000]}, weights=[800, 1_200, 50]) +df.income.gini() +``` + +## MicroSeries + +### Inequality and distribution + +| Method | Signature | Description | +|---|---|---| +| `gini` | `(negatives: Optional[str] = None) -> float` | Calculates Gini index. | +| `top_1_pct_share` | `() -> float` | Calculates top 1% share. | +| `top_10_pct_share` | `() -> float` | Calculates top 10% share. | +| `top_50_pct_share` | `() -> float` | Calculates top 50% share. | +| `bottom_50_pct_share` | `() -> float` | Calculates bottom 50% share. | +| `top_0_1_pct_share` | `() -> float` | Calculates top 0.1% share. | +| `top_x_pct_share` | `(top_x_pct: float) -> float` | Calculates top x% share. | +| `bottom_x_pct_share` | `(bottom_x_pct: float) -> float` | Calculates bottom x% share. | +| `t10_b50` | `() -> float` | Calculates ratio between the top 10% and bottom 50% shares. | + +### Ranking + +| Method | Signature | Description | +|---|---|---| +| `decile_rank` | `(negatives_in_zero: Optional[bool] = False)` | Calculate decile ranks (1-10) with optional zero decile for negatives. | +| `quintile_rank` | `() -> 'MicroSeries'` | | +| `quartile_rank` | `() -> 'MicroSeries'` | | +| `percentile_rank` | `() -> 'MicroSeries'` | | + +### Variance from replicate weights + +| Method | Signature | Description | +|---|---|---| +| `replicate_standard_error` | `(statistic: Callable, replicate_weights, method: str = 'jackknife', fay_k: Optional[float] = None, *, centering: str = 'full-sample') -> float` | Standard error of ``statistic`` from a set of replicate weights. | + +`replicate_standard_error` accepts `method` of `jackknife`, `brr`, `bootstrap`, +`successive-difference`, or `fay` (which also requires `fay_k`). Because it +resamples rather than applying an analytic formula, it works for any statistic +the series can compute, including the Gini coefficient and quantiles. + +### Weights + +| Method | Signature | Description | +|---|---|---| +| `set_weights` | `(weights: , preserve_old: Optional[bool] = False) -> None` | Sets the weight values. | +| `nullify_weights` | `() -> None` | Set all weights to 1, effectively making the Series unweighted. | +| `weight` | `() -> pandas.core.series.Series` | Calculates the weighted value of the MicroSeries. | + +## MicroDataFrame + +### Poverty + +| Method | Signature | Description | +|---|---|---| +| `poverty_rate` | `(income: str, threshold: str) -> float` | Calculate poverty rate, i.e., the population share with income below their poverty threshold. | +| `poverty_gap` | `(income: str, threshold: str) -> float` | Calculate poverty gap, i.e., the total gap between income and poverty thresholds for all people in poverty. | +| `poverty_count` | `(income: Union[microdf.microseries.MicroSeries, str], threshold: Union[microdf.microseries.MicroSeries, str]) -> int` | Calculates the number of entities with income below a poverty threshold. | +| `deep_poverty_rate` | `(income: str, threshold: str) -> float` | Calculate deep poverty rate, i.e., the population share with income below half their poverty threshold. | +| `deep_poverty_gap` | `(income: str, threshold: str) -> float` | Calculate deep poverty gap, i.e., the total gap between income and half of poverty thresholds for all people in deep poverty. | +| `squared_poverty_gap` | `(income: str, threshold: str) -> float` | Calculate squared poverty gap, i.e., the total squared gap between income and poverty thresholds for all people in poverty. Also known as the poverty severity index. | + +### Weights + +| Method | Signature | Description | +|---|---|---| +| `set_weights` | `(weights: Union[numpy.ndarray, str], preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame. | +| `set_weight_col` | `(column: str, preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame by specifying the name of the weight column. | +| `nullify_weights` | `() -> None` | Set all weights to 1, effectively making the DataFrame unweighted. | + +## Module-level functions + +| Function | Description | +|---|---| +| `microdf.replicate_variance` | Variance of a statistic from replicate weights. | +| `microdf.replicate_standard_error` | Square root of the above. | + +## A note on estimator conventions + +Quantiles follow the inverse cumulative distribution function, so results can be +checked against `survey::svyquantile` in R. Weighted variance treats weights as +frequency weights, so integer weights agree with `numpy` computed on the +replicated sample. Top-share cutoffs split a record that straddles the boundary +in proportion, rather than assigning it wholly to one side. diff --git a/docs/myst.yml b/docs/myst.yml index b4a87154..d1b54907 100644 --- a/docs/myst.yml +++ b/docs/myst.yml @@ -20,6 +20,7 @@ project: - file: examples.md children: - file: gini.ipynb + - file: api.md site: options: logo: microdf_logo.png From 4eb1082fedfc2a0fece5d5216dc2812a4577b445 Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Fri, 18 Sep 2026 12:05:22 +0100 Subject: [PATCH 2/7] Document the weighted estimators, and check coverage both ways Per review, the page listed the named estimators and the weight helpers but omitted the weighted aggregation methods and the weight-preserving overrides - sum, mean, median, quantile, var, std, cov, corr, rank, groupby, merge, drop and the rest. Those are the core of both things the page says it covers, and the README's feature list promises them. The coverage check only ran one way, from the page to the package, which is why it passed. It is now a test that runs in both directions, so a new public method that is not documented fails. Also adds docstrings to quintile_rank, quartile_rank and percentile_rank rather than hand-writing their description cells, and corrects the set_weights annotation from np.array, a function, to np.ndarray. The changelog fragment becomes .changed so this releases as a patch. --- ...ence.added.md => api-reference.changed.md} | 0 docs/api.md | 55 +++++++++++++++++ microdf/microseries.py | 19 +++++- microdf/tests/test_docs_api_coverage.py | 60 +++++++++++++++++++ 4 files changed, 132 insertions(+), 2 deletions(-) rename changelog.d/{api-reference.added.md => api-reference.changed.md} (100%) create mode 100644 microdf/tests/test_docs_api_coverage.py diff --git a/changelog.d/api-reference.added.md b/changelog.d/api-reference.changed.md similarity index 100% rename from changelog.d/api-reference.added.md rename to changelog.d/api-reference.changed.md diff --git a/docs/api.md b/docs/api.md index c6d90604..b7f579ba 100644 --- a/docs/api.md +++ b/docs/api.md @@ -15,6 +15,41 @@ df.income.gini() ## MicroSeries +### Weighted aggregation + +These have the same names as their pandas equivalents and return weighted results. + +| Method | Signature | Description | +|---|---|---| +| `sum` | `(axis: Union[int, str, NoneType] = 0, skipna: bool = True, numeric_only: bool = False, min_count: int = 0, **kwargs) -> float` | Calculates the weighted sum of the MicroSeries. | +| `count` | `(skipna: bool = True) -> float` | Calculates the weighted count of the MicroSeries. | +| `mean` | `(skipna: bool = True) -> float` | Calculates the weighted mean of the MicroSeries. | +| `median` | `(skipna: bool = True) -> float` | Calculates the weighted median of the MicroSeries. | +| `quantile` | `(q: , skipna: bool = True) -> pandas.core.series.Series` | Calculates weighted quantiles of the MicroSeries. | +| `var` | `(ddof: int = 1, skipna: bool = True) -> float` | Calculates the weighted variance of the MicroSeries. | +| `std` | `(ddof: int = 1, skipna: bool = True) -> float` | Calculates the weighted standard deviation of the MicroSeries. | +| `cov` | `(other: pandas.core.series.Series, min_periods: Optional[int] = None, ddof: int = 1, *, skipna: bool = True) -> float` | Calculate frequency-weighted covariance with another Series. | +| `corr` | `(other: pandas.core.series.Series, method: str = 'pearson', min_periods: Optional[int] = None, *, ddof: int = 1, skipna: bool = True) -> float` | Calculate frequency-weighted Pearson correlation. | +| `rank` | `(pct: Optional[bool] = False) -> pandas.core.series.Series` | Weighted rank of each element. | + +### Weight-preserving operations + +Operations that change the shape or type of the data, overridden so weights stay aligned with their rows. + +| Method | Signature | Description | +|---|---|---| +| `groupby` | `(*args, **kwargs) -> 'MicroSeriesGroupBy'` | Group into `MicroSeriesGroupBy`, carrying weights into each group. | +| `cumsum` | `() -> pandas.core.series.Series` | Weighted cumulative sum, i.e. the cumulative sum of value times weight. | +| `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> 'MicroSeries'` | Convert MicroSeries to specified data type while preserving weights. | +| `clip` | `(lower: Optional[float] = None, upper: Optional[float] = None, axis: Optional[int] = None, inplace: Optional[bool] = False, *args, **kwargs) -> 'MicroSeries'` | Trim values at the given thresholds, preserving weights. | +| `round` | `(decimals: Optional[int] = 0, *args, **kwargs) -> 'MicroSeries'` | Round each value, preserving weights. | +| `repeat` | `(repeats, axis=None)` | Repeat elements, repeating their weights alongside. | +| `sqrt` | `() -> 'MicroSeries'` | Element-wise square root, preserving weights. | +| `copy` | `(deep: Optional[bool] = True)` | Copy the series and its weights. | +| `equals` | `(other: 'MicroSeries') -> bool` | Compare values; weights are not part of the comparison. | +| `values` | `()` | Access underlying numpy array. | +| `to_numpy` | `(*args, **kwargs)` | Convert to numpy array. | + ### Inequality and distribution | Method | Signature | Description | @@ -59,6 +94,26 @@ the series can compute, including the Gini coefficient and quantiles. ## MicroDataFrame +### Weighted aggregation + +| Method | Signature | Description | +|---|---|---| +| `sum` | `(axis: Union[int, str, NoneType] = 0, skipna: bool = True, numeric_only: bool = False, min_count: int = 0, **kwargs) -> Union[pandas.core.series.Series, microdf.microseries.MicroSeries, float]` | Sum numeric columns, weighting reductions across observations. | +| `cov` | `(min_periods: 'int \| None' = None, ddof: 'int \| None' = 1, numeric_only: 'bool' = False) -> 'DataFrame'` | Pairwise frequency-weighted covariance of the columns. | +| `corr` | `(method: 'CorrelationMethod' = 'pearson', min_periods: 'int' = 1, numeric_only: 'bool' = False) -> 'DataFrame'` | Pairwise frequency-weighted Pearson correlation of the columns. | + +### Weight-preserving operations + +| Method | Signature | Description | +|---|---|---| +| `groupby` | `(by: Union[str, List], *args, **kwargs) -> 'MicroDataFrameGroupBy'` | Returns a GroupBy object with MicroSeriesGroupBy objects for each column. | +| `merge` | `(right, how='inner', on=None, left_on=None, right_on=None, left_index=False, right_index=False, sort=False, suffixes=('_x', '_y'), copy=True, indicator=False, validate=None)` | Database-style join that carries the weight column through. | +| `reset_index` | `(level: Optional[int] = None, drop: Optional[bool] = False, inplace: Optional[bool] = False, col_level: Optional[int] = 0, col_fill: Optional[str] = '', allow_duplicates: Optional[bool] = None, names: Optional[List[str]] = None) -> Optional[ForwardRef('MicroDataFrame')]` | Reset the index, keeping weights aligned to their rows. | +| `drop` | `(labels=None, axis=0, index=None, columns=None, level=None, inplace=False, errors='raise')` | Drop rows or columns, keeping weights aligned to the remaining rows. | +| `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> 'MicroDataFrame'` | Convert MicroDataFrame to specified data type while preserving weights. | +| `copy` | `(deep: Optional[bool] = True) -> 'MicroDataFrame'` | Copy the frame and its weights. | +| `equals` | `(other: 'MicroDataFrame') -> bool` | Compare values; weights are not part of the comparison. | + ### Poverty | Method | Signature | Description | diff --git a/microdf/microseries.py b/microdf/microseries.py index 1ddd458a..83b3cfbe 100644 --- a/microdf/microseries.py +++ b/microdf/microseries.py @@ -269,14 +269,14 @@ def vector_function(fn: Callable) -> Callable: return fn def set_weights( - self, weights: np.array, preserve_old: Optional[bool] = False + self, weights: np.ndarray, preserve_old: Optional[bool] = False ) -> None: """Sets the weight values. :param weights: Array of weights. :param preserve_old: If True, keeps the old weights as a column when new weights are provided. - :type weights: np.array. + :type weights: np.ndarray. """ if weights is None: self.weights = weight_series(np.ones(len(self)), self.index) @@ -945,6 +945,11 @@ def decile_rank(self, negatives_in_zero: Optional[bool] = False): @vector_function def quintile_rank(self) -> "MicroSeries": + """Calculate weighted quintile ranks (1-5). + + :returns: MicroSeries of quintile ranks. + :rtype: MicroSeries + """ return MicroSeries( np.minimum(np.ceil(self.rank(pct=True) * 5), 5), weights=self.weights, @@ -952,6 +957,11 @@ def quintile_rank(self) -> "MicroSeries": @vector_function def quartile_rank(self) -> "MicroSeries": + """Calculate weighted quartile ranks (1-4). + + :returns: MicroSeries of quartile ranks. + :rtype: MicroSeries + """ return MicroSeries( np.minimum(np.ceil(self.rank(pct=True) * 4), 4), weights=self.weights, @@ -959,6 +969,11 @@ def quartile_rank(self) -> "MicroSeries": @vector_function def percentile_rank(self) -> "MicroSeries": + """Calculate weighted percentile ranks (1-100). + + :returns: MicroSeries of percentile ranks. + :rtype: MicroSeries + """ return MicroSeries( np.minimum(np.ceil(self.rank(pct=True) * 100), 100), weights=self.weights, diff --git a/microdf/tests/test_docs_api_coverage.py b/microdf/tests/test_docs_api_coverage.py new file mode 100644 index 00000000..bef55c84 --- /dev/null +++ b/microdf/tests/test_docs_api_coverage.py @@ -0,0 +1,60 @@ +"""The API reference must list every public method, in both directions. + +A one-way check lets the page fall behind the code silently, which is how the +weighted estimators came to be missing from it. +""" + +import inspect +import re +from pathlib import Path + +import microdf as mdf + +DOCS = Path(__file__).resolve().parents[2] / "docs" / "api.md" + +# Public names that read as internals rather than API a user would call. +INTERNAL = { + "scalar_function", + "vector_function", + "override_df_functions", + "catch_series_relapse", + "get_args_as_micro_series", +} + + +def documented_names(): + return set(re.findall(r"^\| `(\w+)` \|", DOCS.read_text(), re.M)) + + +def public_methods(cls, base=None): + """Public names this class defines itself. + + Anything inherited unchanged from pandas is pandas' to document; what + matters here is what microdf adds or overrides. + """ + names = set() + for name, attr in vars(cls).items(): + if name.startswith("_") or name in INTERNAL: + continue + if callable(attr) or isinstance(attr, property): + names.add(name) + return names + + +def test_every_documented_method_exists(): + documented = documented_names() + real = ( + public_methods(mdf.MicroSeries) + | public_methods(mdf.MicroDataFrame) + | {"replicate_variance", "replicate_standard_error"} + ) + assert not (documented - real), f"documented but absent: {documented - real}" + + +def test_every_public_method_is_documented(): + documented = documented_names() + for cls in (mdf.MicroSeries, mdf.MicroDataFrame): + missing = public_methods(cls) - documented + assert not missing, ( + f"{cls.__name__} methods missing from docs/api.md: {sorted(missing)}" + ) From 7f3f9a0f5a8c2ca6f696b1d89ca7a844d1b2f5cc Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Fri, 18 Sep 2026 12:27:48 +0100 Subject: [PATCH 3/7] Correct three wrong descriptions on the API page MicroDataFrame.cov and corr do not use the weights - they call pandas on the plain frame - but the rows described them as frequency-weighted and sat under Weighted aggregation. Verified: with w=[1,1,1,5] the frame gives 3.1667, the unweighted pandas value, while the replicated sample and MicroSeries.cov both give 3.1786. They now carry a warning pointing at #327 and say plainly that they are unweighted. equals compares weights on both classes - both return equal_values and equal_weights - and the rows claimed the opposite. The set_weights row still rendered '' because the page was not regenerated after the annotation fix, and quantile had the same artifact from an unfixed 'q: np.array'. Both annotations are now np.ndarray in source, in microseries.py and microdataframe.py, and the rows regenerated. Also marks values as an attribute rather than showing a call signature, skips the coverage test when docs/ is absent so the packaged tests pass, and drops an unused import and parameter. --- docs/api.md | 26 +++++++++++++++++-------- microdf/microdataframe.py | 4 ++-- microdf/microseries.py | 8 ++++---- microdf/tests/test_docs_api_coverage.py | 11 +++++++++-- 4 files changed, 33 insertions(+), 16 deletions(-) diff --git a/docs/api.md b/docs/api.md index b7f579ba..c93787b2 100644 --- a/docs/api.md +++ b/docs/api.md @@ -25,7 +25,7 @@ These have the same names as their pandas equivalents and return weighted result | `count` | `(skipna: bool = True) -> float` | Calculates the weighted count of the MicroSeries. | | `mean` | `(skipna: bool = True) -> float` | Calculates the weighted mean of the MicroSeries. | | `median` | `(skipna: bool = True) -> float` | Calculates the weighted median of the MicroSeries. | -| `quantile` | `(q: , skipna: bool = True) -> pandas.core.series.Series` | Calculates weighted quantiles of the MicroSeries. | +| `quantile` | `(q: numpy.ndarray, skipna: bool = True) -> pandas.core.series.Series` | Calculates weighted quantiles of the MicroSeries. | | `var` | `(ddof: int = 1, skipna: bool = True) -> float` | Calculates the weighted variance of the MicroSeries. | | `std` | `(ddof: int = 1, skipna: bool = True) -> float` | Calculates the weighted standard deviation of the MicroSeries. | | `cov` | `(other: pandas.core.series.Series, min_periods: Optional[int] = None, ddof: int = 1, *, skipna: bool = True) -> float` | Calculate frequency-weighted covariance with another Series. | @@ -46,8 +46,8 @@ Operations that change the shape or type of the data, overridden so weights stay | `repeat` | `(repeats, axis=None)` | Repeat elements, repeating their weights alongside. | | `sqrt` | `() -> 'MicroSeries'` | Element-wise square root, preserving weights. | | `copy` | `(deep: Optional[bool] = True)` | Copy the series and its weights. | -| `equals` | `(other: 'MicroSeries') -> bool` | Compare values; weights are not part of the comparison. | -| `values` | `()` | Access underlying numpy array. | +| `equals` | `(other: 'MicroSeries') -> bool` | True when both the values and the weights are equal. | +| `values` | *attribute* | Access underlying numpy array. | | `to_numpy` | `(*args, **kwargs)` | Convert to numpy array. | ### Inequality and distribution @@ -88,7 +88,7 @@ the series can compute, including the Gini coefficient and quantiles. | Method | Signature | Description | |---|---|---| -| `set_weights` | `(weights: , preserve_old: Optional[bool] = False) -> None` | Sets the weight values. | +| `set_weights` | `(weights: numpy.ndarray, preserve_old: Optional[bool] = False) -> None` | Sets the weight values. | | `nullify_weights` | `() -> None` | Set all weights to 1, effectively making the Series unweighted. | | `weight` | `() -> pandas.core.series.Series` | Calculates the weighted value of the MicroSeries. | @@ -99,8 +99,18 @@ the series can compute, including the Gini coefficient and quantiles. | Method | Signature | Description | |---|---|---| | `sum` | `(axis: Union[int, str, NoneType] = 0, skipna: bool = True, numeric_only: bool = False, min_count: int = 0, **kwargs) -> Union[pandas.core.series.Series, microdf.microseries.MicroSeries, float]` | Sum numeric columns, weighting reductions across observations. | -| `cov` | `(min_periods: 'int \| None' = None, ddof: 'int \| None' = 1, numeric_only: 'bool' = False) -> 'DataFrame'` | Pairwise frequency-weighted covariance of the columns. | -| `corr` | `(method: 'CorrelationMethod' = 'pearson', min_periods: 'int' = 1, numeric_only: 'bool' = False) -> 'DataFrame'` | Pairwise frequency-weighted Pearson correlation of the columns. | + +```{warning} +`MicroDataFrame.cov()` and `MicroDataFrame.corr()` return pandas' **unweighted** +results; the weights are ignored. Use `MicroSeries.cov()` and `MicroSeries.corr()` +on a pair of columns for the frequency-weighted values. See +[#327](https://github.com/PolicyEngine/microdf/issues/327). +``` + +| Method | Signature | Description | +|---|---|---| +| `cov` | `(min_periods: 'int \| None' = None, ddof: 'int \| None' = 1, numeric_only: 'bool' = False) -> 'DataFrame'` | Pairwise covariance of the columns, **unweighted**. | +| `corr` | `(method: 'CorrelationMethod' = 'pearson', min_periods: 'int' = 1, numeric_only: 'bool' = False) -> 'DataFrame'` | Pairwise Pearson correlation of the columns, **unweighted**. | ### Weight-preserving operations @@ -112,7 +122,7 @@ the series can compute, including the Gini coefficient and quantiles. | `drop` | `(labels=None, axis=0, index=None, columns=None, level=None, inplace=False, errors='raise')` | Drop rows or columns, keeping weights aligned to the remaining rows. | | `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> 'MicroDataFrame'` | Convert MicroDataFrame to specified data type while preserving weights. | | `copy` | `(deep: Optional[bool] = True) -> 'MicroDataFrame'` | Copy the frame and its weights. | -| `equals` | `(other: 'MicroDataFrame') -> bool` | Compare values; weights are not part of the comparison. | +| `equals` | `(other: 'MicroDataFrame') -> bool` | True when both the values and the weights are equal. | ### Poverty @@ -129,7 +139,7 @@ the series can compute, including the Gini coefficient and quantiles. | Method | Signature | Description | |---|---|---| -| `set_weights` | `(weights: Union[numpy.ndarray, str], preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame. | +| `set_weights` | `(weights: numpy.ndarray, preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame. | | `set_weight_col` | `(column: str, preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame by specifying the name of the weight column. | | `nullify_weights` | `() -> None` | Set all weights to 1, effectively making the DataFrame unweighted. | diff --git a/microdf/microdataframe.py b/microdf/microdataframe.py index e756d44e..9d5222ad 100644 --- a/microdf/microdataframe.py +++ b/microdf/microdataframe.py @@ -33,7 +33,7 @@ def __init__(self, *args, weights=None, **kwargs): set_weight_col. :param weights: Array of weights. - :type weights: np.array + :type weights: np.ndarray """ super().__init__(*args, **kwargs) # pandas normalizes mixed-dimensional concat inputs through this @@ -338,7 +338,7 @@ def set_weights( :param weights: Array of weights. :param preserve_old: If True, keeps the old weights as a column when new weights are provided. - :type weights: np.array + :type weights: np.ndarray """ if preserve_old and self.weights_col is not None: self["old_" + self.weights_col] = self.weights diff --git a/microdf/microseries.py b/microdf/microseries.py index 83b3cfbe..b0bf9cf2 100644 --- a/microdf/microseries.py +++ b/microdf/microseries.py @@ -103,13 +103,13 @@ class MicroSeries(WeightPropagationMixin, pd.Series): # Keep inherited fallback paths working after overriding that handler. _HANDLED_TYPES = pd.Series._HANDLED_TYPES + (pd.Series, pd.DataFrame) - def __init__(self, *args, weights: np.array = None, **kwargs): + def __init__(self, *args, weights: np.ndarray = None, **kwargs): """A Series-inheriting class for weighted microdata. Weights can be provided at initialisation, or using set_weights. :param weights: Array of weights. - :type weights: np.array + :type weights: np.ndarray """ super().__init__(*args, **kwargs) self.set_weights(weights) @@ -593,7 +593,7 @@ def corr( result = np.sum(x * y) / (np.sqrt(x_ss) * np.sqrt(y_ss)) return float(np.clip(result, -1.0, 1.0)) - def quantile(self, q: np.array, skipna: bool = True) -> pd.Series: + def quantile(self, q: np.ndarray, skipna: bool = True) -> pd.Series: """Calculates weighted quantiles of the MicroSeries. Uses the inverse CDF method: the q-th quantile is the smallest @@ -601,7 +601,7 @@ def quantile(self, q: np.array, skipna: bool = True) -> pd.Series: the default behavior of R's survey::svyquantile. :param q: Quantile(s) to calculate, must be in [0, 1]. - :type q: float or np.array + :type q: float or np.ndarray :param skipna: Exclude NaN values (default True). NaN sorts to the end of the array, so leaving NaN rows in would let their weight inflate the cumulative distribution and push the cutoff upward. diff --git a/microdf/tests/test_docs_api_coverage.py b/microdf/tests/test_docs_api_coverage.py index bef55c84..99abb47e 100644 --- a/microdf/tests/test_docs_api_coverage.py +++ b/microdf/tests/test_docs_api_coverage.py @@ -4,14 +4,21 @@ weighted estimators came to be missing from it. """ -import inspect import re from pathlib import Path +import pytest + import microdf as mdf DOCS = Path(__file__).resolve().parents[2] / "docs" / "api.md" +# docs/ is not shipped in the sdist or the wheel, so these cannot run against an +# installed copy of the package. +pytestmark = pytest.mark.skipif( + not DOCS.exists(), reason="docs/api.md is not present in the installed package" +) + # Public names that read as internals rather than API a user would call. INTERNAL = { "scalar_function", @@ -26,7 +33,7 @@ def documented_names(): return set(re.findall(r"^\| `(\w+)` \|", DOCS.read_text(), re.M)) -def public_methods(cls, base=None): +def public_methods(cls): """Public names this class defines itself. Anything inherited unchanged from pandas is pandas' to document; what From e6df79a7f4893e25651db939250bcc83084617e6 Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Fri, 18 Sep 2026 12:30:01 +0100 Subject: [PATCH 4/7] Correct cumsum, and pin the hand-written claims in tests Audited every description on the page that was written by hand rather than taken from a docstring, since that is where all three errors found in review were. cumsum was the remaining one. It returns a plain pandas Series and warns that the weights have been applied and cannot be reused, so it is the one method in the weight-preserving section that does not preserve them. The page now says so. The rest hold: clip, round, sqrt, copy and astype carry weights through; repeat repeats them alongside the values; merge, reset_index and drop keep them aligned, and drop(index=) drops the matching weights; groupby returns the MicroSeriesGroupBy and MicroDataFrameGroupBy wrappers. Those checks are now a test rather than something I ran once, covering the claims no docstring enforces: the unweighted frame cov and corr, the weighted MicroSeries cov, equals comparing weights, cumsum dropping them, and repeat carrying them. --- docs/api.md | 2 +- microdf/tests/test_docs_api_coverage.py | 38 +++++++++++++++++++++++++ 2 files changed, 39 insertions(+), 1 deletion(-) diff --git a/docs/api.md b/docs/api.md index c93787b2..8dab4aad 100644 --- a/docs/api.md +++ b/docs/api.md @@ -39,7 +39,7 @@ Operations that change the shape or type of the data, overridden so weights stay | Method | Signature | Description | |---|---|---| | `groupby` | `(*args, **kwargs) -> 'MicroSeriesGroupBy'` | Group into `MicroSeriesGroupBy`, carrying weights into each group. | -| `cumsum` | `() -> pandas.core.series.Series` | Weighted cumulative sum, i.e. the cumulative sum of value times weight. | +| `cumsum` | `() -> pandas.core.series.Series` | Cumulative sum of value times weight. Returns a plain `pandas.Series`: the weights have been applied and are not carried forward, so this is the one method here that does not preserve them. | | `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> 'MicroSeries'` | Convert MicroSeries to specified data type while preserving weights. | | `clip` | `(lower: Optional[float] = None, upper: Optional[float] = None, axis: Optional[int] = None, inplace: Optional[bool] = False, *args, **kwargs) -> 'MicroSeries'` | Trim values at the given thresholds, preserving weights. | | `round` | `(decimals: Optional[int] = 0, *args, **kwargs) -> 'MicroSeries'` | Round each value, preserving weights. | diff --git a/microdf/tests/test_docs_api_coverage.py b/microdf/tests/test_docs_api_coverage.py index 99abb47e..dcc98d3c 100644 --- a/microdf/tests/test_docs_api_coverage.py +++ b/microdf/tests/test_docs_api_coverage.py @@ -65,3 +65,41 @@ def test_every_public_method_is_documented(): assert not missing, ( f"{cls.__name__} methods missing from docs/api.md: {sorted(missing)}" ) + + +def test_documented_weight_behaviour_holds(): + """Pin the claims the page makes that a docstring does not enforce. + + Every error found in review was in a description written by hand rather + than taken from a docstring, so the ones that remain are asserted here. + """ + import numpy as np + import pandas as pd + + frame = mdf.MicroDataFrame( + {"x": [1.0, 2.0, 3.0, 4.0], "y": [1.0, 4.0, 2.0, 8.0]}, + weights=[1.0, 1.0, 1.0, 5.0], + ) + replicated = pd.DataFrame( + {"x": [1.0, 2.0, 3.0] + [4.0] * 5, "y": [1.0, 4.0, 2.0] + [8.0] * 5} + ) + + # The page says these are unweighted, and points at #327. + plain = pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0], "y": [1.0, 4.0, 2.0, 8.0]}) + assert frame.cov().loc["x", "y"] == pytest.approx(plain.cov().loc["x", "y"]) + assert frame.corr().loc["x", "y"] == pytest.approx(plain.corr().loc["x", "y"]) + + # The page says the MicroSeries versions are frequency-weighted. + assert frame.x.cov(frame.y) == pytest.approx(replicated.cov().loc["x", "y"]) + + # The page says equals compares weights. + light = mdf.MicroSeries([1, 2, 3], weights=[1, 1, 1]) + heavy = mdf.MicroSeries([1, 2, 3], weights=[9, 9, 9]) + assert not light.equals(heavy) + + # The page says cumsum drops the weights. + assert not hasattr(frame.x.cumsum(), "weights") + + # The page says repeat repeats the weights alongside the values. + repeated = mdf.MicroSeries([1.0, 2.0], weights=[3.0, 4.0]).repeat(2) + assert list(np.asarray(repeated.weights)) == [3.0, 3.0, 4.0, 4.0] From 552dc4c316790c2a2b9547eefd013b090edd01dd Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Fri, 18 Sep 2026 12:39:54 +0100 Subject: [PATCH 5/7] Regenerate the rows the source outran, and test that it cannot recur The three rank rows were still blank: they got docstrings in the first round but the page was never regenerated, the same miss as set_weights in the second. Patching individual rows by hand is what kept producing this, so the rows are now regenerated from the live signatures and docstrings, and two tests hold the page to the code. test_no_row_is_missing_its_description fails on any blank description cell. test_signatures_match_the_live_ones compares every rendered signature against inspect.signature, attributing each row to the class whose heading it falls under, since several names exist on both. The second test immediately found a real error: the MicroDataFrame set_weights row carried the MicroSeries signature, because the earlier regex fix matched both rows. Regenerated. --- docs/api.md | 8 ++--- microdf/tests/test_docs_api_coverage.py | 45 +++++++++++++++++++++++++ 2 files changed, 49 insertions(+), 4 deletions(-) diff --git a/docs/api.md b/docs/api.md index 8dab4aad..7e1786df 100644 --- a/docs/api.md +++ b/docs/api.md @@ -69,9 +69,9 @@ Operations that change the shape or type of the data, overridden so weights stay | Method | Signature | Description | |---|---|---| | `decile_rank` | `(negatives_in_zero: Optional[bool] = False)` | Calculate decile ranks (1-10) with optional zero decile for negatives. | -| `quintile_rank` | `() -> 'MicroSeries'` | | -| `quartile_rank` | `() -> 'MicroSeries'` | | -| `percentile_rank` | `() -> 'MicroSeries'` | | +| `quintile_rank` | `() -> 'MicroSeries'` | Calculate weighted quintile ranks (1-5). | +| `quartile_rank` | `() -> 'MicroSeries'` | Calculate weighted quartile ranks (1-4). | +| `percentile_rank` | `() -> 'MicroSeries'` | Calculate weighted percentile ranks (1-100). | ### Variance from replicate weights @@ -139,7 +139,7 @@ on a pair of columns for the frequency-weighted values. See | Method | Signature | Description | |---|---|---| -| `set_weights` | `(weights: numpy.ndarray, preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame. | +| `set_weights` | `(weights: Union[numpy.ndarray, str], preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame. | | `set_weight_col` | `(column: str, preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame by specifying the name of the weight column. | | `nullify_weights` | `() -> None` | Set all weights to 1, effectively making the DataFrame unweighted. | diff --git a/microdf/tests/test_docs_api_coverage.py b/microdf/tests/test_docs_api_coverage.py index dcc98d3c..a0bd4f5c 100644 --- a/microdf/tests/test_docs_api_coverage.py +++ b/microdf/tests/test_docs_api_coverage.py @@ -4,6 +4,7 @@ weighted estimators came to be missing from it. """ +import inspect import re from pathlib import Path @@ -103,3 +104,47 @@ def test_documented_weight_behaviour_holds(): # The page says repeat repeats the weights alongside the values. repeated = mdf.MicroSeries([1.0, 2.0], weights=[3.0, 4.0]).repeat(2) assert list(np.asarray(repeated.weights)) == [3.0, 3.0, 4.0, 4.0] + + +def test_no_row_is_missing_its_description(): + """Every documented method needs a description. + + A row whose description cell is blank is the signature of a source change + that the page was never regenerated for, which happened three times in + review before this test existed. + """ + blank = re.findall( + r"^\| `(\w+)` \| (?:`[^`]*`|\*attribute\*) \|\s*\|$", DOCS.read_text(), re.M + ) + assert not blank, f"rows with no description: {blank}" + + +def test_signatures_match_the_live_ones(): + """The rendered signature must be the one the code actually has. + + Rows are attributed to the class whose `## ` heading they fall under, since + several names exist on both. + """ + current, mismatches = None, [] + for line in DOCS.read_text().split("\n"): + if line.startswith("## MicroSeries"): + current = mdf.MicroSeries + elif line.startswith("## MicroDataFrame"): + current = mdf.MicroDataFrame + elif line.startswith("## "): + current = None + row = re.match(r"^\| `(\w+)` \| `([^`]*)` \|", line) + if not row or current is None: + continue + name, rendered = row.groups() + func = inspect.getattr_static(current, name, None) + if func is None or isinstance(func, property): + continue + try: + live = str(inspect.signature(func)) + except (TypeError, ValueError): + continue + live = live.replace("(self, ", "(").replace("(self)", "()").replace("|", "\\|") + if live != rendered: + mismatches.append((current.__name__, name, rendered, live)) + assert not mismatches, f"page is out of date with the code: {mismatches}" From e22b7421cb3f282d079ea0e17b5592726522731e Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Fri, 18 Sep 2026 13:03:58 +0100 Subject: [PATCH 6/7] Generate the API page from a committed script, version-stably The script that produced docs/api.md was run by hand and lived outside the repo, which is why the source outran the page four times in review. It is now docs/build_api.py: the prose is held verbatim, the tables are generated from the live classes, and the hand-written descriptions that are not docstrings (cumsum, merge, reset_index, drop, equals, copy, groupby, clip, round, repeat, sqrt, and the two unweighted cov/corr rows) are an explicit override table so regeneration cannot lose them. str(inspect.signature(...)) was never safe to compare as text: pandas 2 renders a Series annotation as pandas.core.series.Series and pandas 3 renders it as pandas.Series, so a page generated under one fails CI on the other's jobs, and from Python 3.14 Optional[X] reprs as X | None. microdf/_docs.py renders signatures from parameter names, kinds and defaults, strips module qualifiers and puts every union into one form. The generator and the test both import it, so the page cannot disagree with the test, and the same committed page passes under pandas 2.3 and 3.0 and on Python 3.9 through 3.14. --- docs/api.md | 52 ++--- docs/build_api.py | 264 ++++++++++++++++++++++++ microdf/_docs.py | 144 +++++++++++++ microdf/tests/test_docs_api_coverage.py | 16 +- 4 files changed, 447 insertions(+), 29 deletions(-) create mode 100644 docs/build_api.py create mode 100644 microdf/_docs.py diff --git a/docs/api.md b/docs/api.md index 7e1786df..324b2f4c 100644 --- a/docs/api.md +++ b/docs/api.md @@ -25,12 +25,12 @@ These have the same names as their pandas equivalents and return weighted result | `count` | `(skipna: bool = True) -> float` | Calculates the weighted count of the MicroSeries. | | `mean` | `(skipna: bool = True) -> float` | Calculates the weighted mean of the MicroSeries. | | `median` | `(skipna: bool = True) -> float` | Calculates the weighted median of the MicroSeries. | -| `quantile` | `(q: numpy.ndarray, skipna: bool = True) -> pandas.core.series.Series` | Calculates weighted quantiles of the MicroSeries. | +| `quantile` | `(q: ndarray, skipna: bool = True) -> Series` | Calculates weighted quantiles of the MicroSeries. | | `var` | `(ddof: int = 1, skipna: bool = True) -> float` | Calculates the weighted variance of the MicroSeries. | | `std` | `(ddof: int = 1, skipna: bool = True) -> float` | Calculates the weighted standard deviation of the MicroSeries. | -| `cov` | `(other: pandas.core.series.Series, min_periods: Optional[int] = None, ddof: int = 1, *, skipna: bool = True) -> float` | Calculate frequency-weighted covariance with another Series. | -| `corr` | `(other: pandas.core.series.Series, method: str = 'pearson', min_periods: Optional[int] = None, *, ddof: int = 1, skipna: bool = True) -> float` | Calculate frequency-weighted Pearson correlation. | -| `rank` | `(pct: Optional[bool] = False) -> pandas.core.series.Series` | Weighted rank of each element. | +| `cov` | `(other: Series, min_periods: Optional[int] = None, ddof: int = 1, *, skipna: bool = True) -> float` | Calculate frequency-weighted covariance with another Series. | +| `corr` | `(other: Series, method: str = 'pearson', min_periods: Optional[int] = None, *, ddof: int = 1, skipna: bool = True) -> float` | Calculate frequency-weighted Pearson correlation. | +| `rank` | `(pct: Optional[bool] = False) -> Series` | Weighted rank of each element. | ### Weight-preserving operations @@ -38,15 +38,15 @@ Operations that change the shape or type of the data, overridden so weights stay | Method | Signature | Description | |---|---|---| -| `groupby` | `(*args, **kwargs) -> 'MicroSeriesGroupBy'` | Group into `MicroSeriesGroupBy`, carrying weights into each group. | -| `cumsum` | `() -> pandas.core.series.Series` | Cumulative sum of value times weight. Returns a plain `pandas.Series`: the weights have been applied and are not carried forward, so this is the one method here that does not preserve them. | -| `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> 'MicroSeries'` | Convert MicroSeries to specified data type while preserving weights. | -| `clip` | `(lower: Optional[float] = None, upper: Optional[float] = None, axis: Optional[int] = None, inplace: Optional[bool] = False, *args, **kwargs) -> 'MicroSeries'` | Trim values at the given thresholds, preserving weights. | -| `round` | `(decimals: Optional[int] = 0, *args, **kwargs) -> 'MicroSeries'` | Round each value, preserving weights. | +| `groupby` | `(*args, **kwargs) -> MicroSeriesGroupBy` | Group into `MicroSeriesGroupBy`, carrying weights into each group. | +| `cumsum` | `() -> Series` | Cumulative sum of value times weight. Returns a plain `pandas.Series`: the weights have been applied and are not carried forward, so this is the one method here that does not preserve them. | +| `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> MicroSeries` | Convert MicroSeries to specified data type while preserving weights. | +| `clip` | `(lower: Optional[float] = None, upper: Optional[float] = None, axis: Optional[int] = None, inplace: Optional[bool] = False, *args, **kwargs) -> MicroSeries` | Trim values at the given thresholds, preserving weights. | +| `round` | `(decimals: Optional[int] = 0, *args, **kwargs) -> MicroSeries` | Round each value, preserving weights. | | `repeat` | `(repeats, axis=None)` | Repeat elements, repeating their weights alongside. | -| `sqrt` | `() -> 'MicroSeries'` | Element-wise square root, preserving weights. | +| `sqrt` | `() -> MicroSeries` | Element-wise square root, preserving weights. | | `copy` | `(deep: Optional[bool] = True)` | Copy the series and its weights. | -| `equals` | `(other: 'MicroSeries') -> bool` | True when both the values and the weights are equal. | +| `equals` | `(other: MicroSeries) -> bool` | True when both the values and the weights are equal. | | `values` | *attribute* | Access underlying numpy array. | | `to_numpy` | `(*args, **kwargs)` | Convert to numpy array. | @@ -69,9 +69,9 @@ Operations that change the shape or type of the data, overridden so weights stay | Method | Signature | Description | |---|---|---| | `decile_rank` | `(negatives_in_zero: Optional[bool] = False)` | Calculate decile ranks (1-10) with optional zero decile for negatives. | -| `quintile_rank` | `() -> 'MicroSeries'` | Calculate weighted quintile ranks (1-5). | -| `quartile_rank` | `() -> 'MicroSeries'` | Calculate weighted quartile ranks (1-4). | -| `percentile_rank` | `() -> 'MicroSeries'` | Calculate weighted percentile ranks (1-100). | +| `quintile_rank` | `() -> MicroSeries` | Calculate weighted quintile ranks (1-5). | +| `quartile_rank` | `() -> MicroSeries` | Calculate weighted quartile ranks (1-4). | +| `percentile_rank` | `() -> MicroSeries` | Calculate weighted percentile ranks (1-100). | ### Variance from replicate weights @@ -88,9 +88,9 @@ the series can compute, including the Gini coefficient and quantiles. | Method | Signature | Description | |---|---|---| -| `set_weights` | `(weights: numpy.ndarray, preserve_old: Optional[bool] = False) -> None` | Sets the weight values. | +| `set_weights` | `(weights: ndarray, preserve_old: Optional[bool] = False) -> None` | Sets the weight values. | | `nullify_weights` | `() -> None` | Set all weights to 1, effectively making the Series unweighted. | -| `weight` | `() -> pandas.core.series.Series` | Calculates the weighted value of the MicroSeries. | +| `weight` | `() -> Series` | Calculates the weighted value of the MicroSeries. | ## MicroDataFrame @@ -98,7 +98,7 @@ the series can compute, including the Gini coefficient and quantiles. | Method | Signature | Description | |---|---|---| -| `sum` | `(axis: Union[int, str, NoneType] = 0, skipna: bool = True, numeric_only: bool = False, min_count: int = 0, **kwargs) -> Union[pandas.core.series.Series, microdf.microseries.MicroSeries, float]` | Sum numeric columns, weighting reductions across observations. | +| `sum` | `(axis: Union[int, str, NoneType] = 0, skipna: bool = True, numeric_only: bool = False, min_count: int = 0, **kwargs) -> Union[Series, MicroSeries, float]` | Sum numeric columns, weighting reductions across observations. | ```{warning} `MicroDataFrame.cov()` and `MicroDataFrame.corr()` return pandas' **unweighted** @@ -109,20 +109,20 @@ on a pair of columns for the frequency-weighted values. See | Method | Signature | Description | |---|---|---| -| `cov` | `(min_periods: 'int \| None' = None, ddof: 'int \| None' = 1, numeric_only: 'bool' = False) -> 'DataFrame'` | Pairwise covariance of the columns, **unweighted**. | -| `corr` | `(method: 'CorrelationMethod' = 'pearson', min_periods: 'int' = 1, numeric_only: 'bool' = False) -> 'DataFrame'` | Pairwise Pearson correlation of the columns, **unweighted**. | +| `cov` | `(min_periods: Optional[int] = None, ddof: Optional[int] = 1, numeric_only: bool = False) -> DataFrame` | Pairwise covariance of the columns, **unweighted**. | +| `corr` | `(method: CorrelationMethod = 'pearson', min_periods: int = 1, numeric_only: bool = False) -> DataFrame` | Pairwise Pearson correlation of the columns, **unweighted**. | ### Weight-preserving operations | Method | Signature | Description | |---|---|---| -| `groupby` | `(by: Union[str, List], *args, **kwargs) -> 'MicroDataFrameGroupBy'` | Returns a GroupBy object with MicroSeriesGroupBy objects for each column. | +| `groupby` | `(by: Union[str, list], *args, **kwargs) -> MicroDataFrameGroupBy` | Returns a GroupBy object with MicroSeriesGroupBy objects for each column. | | `merge` | `(right, how='inner', on=None, left_on=None, right_on=None, left_index=False, right_index=False, sort=False, suffixes=('_x', '_y'), copy=True, indicator=False, validate=None)` | Database-style join that carries the weight column through. | -| `reset_index` | `(level: Optional[int] = None, drop: Optional[bool] = False, inplace: Optional[bool] = False, col_level: Optional[int] = 0, col_fill: Optional[str] = '', allow_duplicates: Optional[bool] = None, names: Optional[List[str]] = None) -> Optional[ForwardRef('MicroDataFrame')]` | Reset the index, keeping weights aligned to their rows. | +| `reset_index` | `(level: Optional[int] = None, drop: Optional[bool] = False, inplace: Optional[bool] = False, col_level: Optional[int] = 0, col_fill: Optional[str] = '', allow_duplicates: Optional[bool] = None, names: Optional[list[str]] = None) -> Optional[MicroDataFrame]` | Reset the index, keeping weights aligned to their rows. | | `drop` | `(labels=None, axis=0, index=None, columns=None, level=None, inplace=False, errors='raise')` | Drop rows or columns, keeping weights aligned to the remaining rows. | -| `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> 'MicroDataFrame'` | Convert MicroDataFrame to specified data type while preserving weights. | -| `copy` | `(deep: Optional[bool] = True) -> 'MicroDataFrame'` | Copy the frame and its weights. | -| `equals` | `(other: 'MicroDataFrame') -> bool` | True when both the values and the weights are equal. | +| `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> MicroDataFrame` | Convert MicroDataFrame to specified data type while preserving weights. | +| `copy` | `(deep: Optional[bool] = True) -> MicroDataFrame` | Copy the frame and its weights. | +| `equals` | `(other: MicroDataFrame) -> bool` | True when both the values and the weights are equal. | ### Poverty @@ -130,7 +130,7 @@ on a pair of columns for the frequency-weighted values. See |---|---|---| | `poverty_rate` | `(income: str, threshold: str) -> float` | Calculate poverty rate, i.e., the population share with income below their poverty threshold. | | `poverty_gap` | `(income: str, threshold: str) -> float` | Calculate poverty gap, i.e., the total gap between income and poverty thresholds for all people in poverty. | -| `poverty_count` | `(income: Union[microdf.microseries.MicroSeries, str], threshold: Union[microdf.microseries.MicroSeries, str]) -> int` | Calculates the number of entities with income below a poverty threshold. | +| `poverty_count` | `(income: Union[MicroSeries, str], threshold: Union[MicroSeries, str]) -> int` | Calculates the number of entities with income below a poverty threshold. | | `deep_poverty_rate` | `(income: str, threshold: str) -> float` | Calculate deep poverty rate, i.e., the population share with income below half their poverty threshold. | | `deep_poverty_gap` | `(income: str, threshold: str) -> float` | Calculate deep poverty gap, i.e., the total gap between income and half of poverty thresholds for all people in deep poverty. | | `squared_poverty_gap` | `(income: str, threshold: str) -> float` | Calculate squared poverty gap, i.e., the total squared gap between income and poverty thresholds for all people in poverty. Also known as the poverty severity index. | @@ -139,7 +139,7 @@ on a pair of columns for the frequency-weighted values. See | Method | Signature | Description | |---|---|---| -| `set_weights` | `(weights: Union[numpy.ndarray, str], preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame. | +| `set_weights` | `(weights: Union[ndarray, str], preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame. | | `set_weight_col` | `(column: str, preserve_old: Optional[bool] = False) -> None` | Sets the weights for the MicroDataFrame by specifying the name of the weight column. | | `nullify_weights` | `() -> None` | Set all weights to 1, effectively making the DataFrame unweighted. | diff --git a/docs/build_api.py b/docs/build_api.py new file mode 100644 index 00000000..5bea67e3 --- /dev/null +++ b/docs/build_api.py @@ -0,0 +1,264 @@ +#!/usr/bin/env python +"""Regenerate ``docs/api.md`` from the live package. + +Run it from anywhere: + + uv run --with . --no-project python docs/build_api.py + +The page used to be produced by a script that lived outside the repo, which is +how the source and the page drifted apart four separate times in review. This +is that script, committed, so the page can always be rebuilt and a stale page is +a visible diff. + +Signatures are rendered through ``microdf._docs``, the same module the test +imports, so the page reads the same under pandas 2 and pandas 3 and the test +cannot disagree with the generator. + +The prose is held verbatim here; only the tables are generated. +""" + +import inspect +from pathlib import Path + +import microdf as mdf +from microdf._docs import markdown_signature + +PAGE = Path(__file__).resolve().parent / "api.md" + +# Descriptions that are not the method's own docstring: methods inherited or +# overridden from pandas keep pandas' docstring, which describes pandas' +# behaviour rather than what microdf does with the weights. These are written +# by hand and must survive regeneration. +OVERRIDES = { + ("MicroSeries", "groupby"): ( + "Group into `MicroSeriesGroupBy`, carrying weights into each group." + ), + ("MicroSeries", "cumsum"): ( + "Cumulative sum of value times weight. Returns a plain `pandas.Series`: " + "the weights have been applied and are not carried forward, so this is " + "the one method here that does not preserve them." + ), + ("MicroSeries", "clip"): "Trim values at the given thresholds, preserving weights.", + ("MicroSeries", "round"): "Round each value, preserving weights.", + ("MicroSeries", "repeat"): "Repeat elements, repeating their weights alongside.", + ("MicroSeries", "sqrt"): "Element-wise square root, preserving weights.", + ("MicroSeries", "copy"): "Copy the series and its weights.", + ("MicroSeries", "equals"): "True when both the values and the weights are equal.", + ("MicroDataFrame", "cov"): "Pairwise covariance of the columns, **unweighted**.", + ("MicroDataFrame", "corr"): ( + "Pairwise Pearson correlation of the columns, **unweighted**." + ), + ("MicroDataFrame", "merge"): ( + "Database-style join that carries the weight column through." + ), + ("MicroDataFrame", "reset_index"): ( + "Reset the index, keeping weights aligned to their rows." + ), + ("MicroDataFrame", "drop"): ( + "Drop rows or columns, keeping weights aligned to the remaining rows." + ), + ("MicroDataFrame", "copy"): "Copy the frame and its weights.", + ("MicroDataFrame", "equals"): ( + "True when both the values and the weights are equal." + ), +} + +HEADER = "| Method | Signature | Description |\n|---|---|---|" + + +def describe(cls, name): + key = (cls.__name__, name) + if key in OVERRIDES: + return OVERRIDES[key] + attr = inspect.getattr_static(cls, name) + if isinstance(attr, property): + attr = attr.fget + doc = inspect.getdoc(attr) or "" + first = " ".join(doc.split("\n\n")[0].split()).split(":param")[0].strip() + if not first: + raise SystemExit( + f"{cls.__name__}.{name} has no docstring and no description override" + ) + return first + + +def row(cls, name): + attr = inspect.getattr_static(cls, name) + if isinstance(attr, property): + signature = "*attribute*" + else: + signature = f"`{markdown_signature(attr)}`" + return f"| `{name}` | {signature} | {describe(cls, name)} |" + + +def table(cls, names): + return "\n".join([HEADER] + [row(cls, name) for name in names]) + + +SERIES = mdf.MicroSeries +FRAME = mdf.MicroDataFrame + + +def build(): + return f"""# API reference + +`microdf` exposes two classes. `MicroSeries` is a `pandas.Series` carrying a weight +vector; `MicroDataFrame` is a `pandas.DataFrame` carrying a weight column. Both +behave like their pandas counterparts, and the methods below either add a +weighted estimator or preserve weights through an operation that would otherwise +drop them. + +```python +import microdf as mdf + +df = mdf.MicroDataFrame({{"income": [10_000, 30_000, 120_000]}}, weights=[800, 1_200, 50]) +df.income.gini() +``` + +## MicroSeries + +### Weighted aggregation + +These have the same names as their pandas equivalents and return weighted results. + +{ + table( + SERIES, + [ + "sum", + "count", + "mean", + "median", + "quantile", + "var", + "std", + "cov", + "corr", + "rank", + ], + ) + } + +### Weight-preserving operations + +Operations that change the shape or type of the data, overridden so weights stay aligned with their rows. + +{ + table( + SERIES, + [ + "groupby", + "cumsum", + "astype", + "clip", + "round", + "repeat", + "sqrt", + "copy", + "equals", + "values", + "to_numpy", + ], + ) + } + +### Inequality and distribution + +{ + table( + SERIES, + [ + "gini", + "top_1_pct_share", + "top_10_pct_share", + "top_50_pct_share", + "bottom_50_pct_share", + "top_0_1_pct_share", + "top_x_pct_share", + "bottom_x_pct_share", + "t10_b50", + ], + ) + } + +### Ranking + +{table(SERIES, ["decile_rank", "quintile_rank", "quartile_rank", "percentile_rank"])} + +### Variance from replicate weights + +{table(SERIES, ["replicate_standard_error"])} + +`replicate_standard_error` accepts `method` of `jackknife`, `brr`, `bootstrap`, +`successive-difference`, or `fay` (which also requires `fay_k`). Because it +resamples rather than applying an analytic formula, it works for any statistic +the series can compute, including the Gini coefficient and quantiles. + +### Weights + +{table(SERIES, ["set_weights", "nullify_weights", "weight"])} + +## MicroDataFrame + +### Weighted aggregation + +{table(FRAME, ["sum"])} + +```{{warning}} +`MicroDataFrame.cov()` and `MicroDataFrame.corr()` return pandas' **unweighted** +results; the weights are ignored. Use `MicroSeries.cov()` and `MicroSeries.corr()` +on a pair of columns for the frequency-weighted values. See +[#327](https://github.com/PolicyEngine/microdf/issues/327). +``` + +{table(FRAME, ["cov", "corr"])} + +### Weight-preserving operations + +{ + table( + FRAME, + ["groupby", "merge", "reset_index", "drop", "astype", "copy", "equals"], + ) + } + +### Poverty + +{ + table( + FRAME, + [ + "poverty_rate", + "poverty_gap", + "poverty_count", + "deep_poverty_rate", + "deep_poverty_gap", + "squared_poverty_gap", + ], + ) + } + +### Weights + +{table(FRAME, ["set_weights", "set_weight_col", "nullify_weights"])} + +## Module-level functions + +| Function | Description | +|---|---| +| `microdf.replicate_variance` | Variance of a statistic from replicate weights. | +| `microdf.replicate_standard_error` | Square root of the above. | + +## A note on estimator conventions + +Quantiles follow the inverse cumulative distribution function, so results can be +checked against `survey::svyquantile` in R. Weighted variance treats weights as +frequency weights, so integer weights agree with `numpy` computed on the +replicated sample. Top-share cutoffs split a record that straddles the boundary +in proportion, rather than assigning it wholly to one side. +""" + + +if __name__ == "__main__": + PAGE.write_text(build()) + print(f"wrote {PAGE}") diff --git a/microdf/_docs.py b/microdf/_docs.py new file mode 100644 index 00000000..16d6a60d --- /dev/null +++ b/microdf/_docs.py @@ -0,0 +1,144 @@ +"""Version-stable rendering of signatures for the API reference. + +`str(inspect.signature(...))` is not stable across versions: pandas 2 renders a +Series annotation as ``pandas.core.series.Series`` and pandas 3 renders the same +annotation as ``pandas.Series``, and from Python 3.14 ``Optional[int]`` reprs as +``int | None``. Generating ``docs/api.md`` from one of those and testing it +against another is a guaranteed CI failure on some job. + +So the page and its test both render signatures through the functions here, +which strip module qualifiers and put unions into a single form. There is one +renderer, imported from one place, so the page cannot disagree with the test. +""" + +import inspect +import re +import types +import typing + +__all__ = ["render_annotation", "render_signature", "markdown_signature"] + +_NoneType = type(None) + +# ``pandas.core.series.Series`` -> ``Series``. Requires a dot, so string +# literals inside e.g. ``Literal['a', 'b']`` are untouched. +_QUALIFIER = re.compile(r"\b(?:[A-Za-z_][A-Za-z0-9_]*\.)+([A-Za-z_][A-Za-z0-9_]*)\b") + + +def _split_top_level(text, sep="|"): + parts, depth, current = [], 0, "" + for char in text: + if char in "[({": + depth += 1 + elif char in "])}": + depth -= 1 + if char == sep and depth == 0: + parts.append(current) + current = "" + else: + current += char + parts.append(current) + return [part.strip() for part in parts] + + +def _union(rendered): + """One spelling for a union, whatever the source spelling was.""" + non_none = [part for part in rendered if part not in ("None", "NoneType")] + nones = len(rendered) - len(non_none) + if nones and len(non_none) == 1: + return f"Optional[{non_none[0]}]" + if nones: + return "Union[" + ", ".join(non_none + ["NoneType"]) + "]" + return "Union[" + ", ".join(non_none) + "]" + + +def _clean_text(text): + """Normalise an annotation that reaches us as a string. + + ``from __future__ import annotations`` in pandas means many of its + signatures carry string annotations such as ``'int | None'``. + """ + text = text.strip() + if len(text) > 1 and text[0] == text[-1] and text[0] in "\"'": + text = text[1:-1].strip() + text = _QUALIFIER.sub(r"\1", text) + if "|" in text: + parts = _split_top_level(text) + if len(parts) > 1: + return _union([_clean_text(part) for part in parts]) + return text + + +def render_annotation(annotation): + """Render an annotation identically under any supported pandas/Python.""" + if isinstance(annotation, str): + return _clean_text(annotation) + if isinstance(annotation, typing.ForwardRef): + return _clean_text(annotation.__forward_arg__) + if annotation is _NoneType: + return "NoneType" + + origin = typing.get_origin(annotation) + if origin is not None: + args = typing.get_args(annotation) + if origin is typing.Union or origin is getattr(types, "UnionType", ()): + return _union([render_annotation(arg) for arg in args]) + name = getattr(origin, "__name__", None) or _clean_text(repr(origin)) + if not args: + return name + return f"{name}[{', '.join(render_annotation(arg) for arg in args)}]" + + if isinstance(annotation, type): + return annotation.__name__ + return _clean_text(repr(annotation)) + + +def render_signature(func): + """Render ``func``'s signature, without ``self``, version-stably. + + Parameter names, kinds and defaults come straight from + ``inspect.signature``; only the annotation text is normalised. + """ + signature = inspect.signature(func) + parts, previous_kind = [], None + for parameter in signature.parameters.values(): + if parameter.name == "self" and not parts: + previous_kind = parameter.kind + continue + if ( + previous_kind is inspect.Parameter.POSITIONAL_ONLY + and parameter.kind is not inspect.Parameter.POSITIONAL_ONLY + ): + parts.append("/") + if parameter.kind is inspect.Parameter.KEYWORD_ONLY and previous_kind not in ( + inspect.Parameter.VAR_POSITIONAL, + inspect.Parameter.KEYWORD_ONLY, + ): + parts.append("*") + + rendered = parameter.name + if parameter.kind is inspect.Parameter.VAR_POSITIONAL: + rendered = "*" + rendered + elif parameter.kind is inspect.Parameter.VAR_KEYWORD: + rendered = "**" + rendered + if parameter.annotation is not inspect.Parameter.empty: + rendered += f": {render_annotation(parameter.annotation)}" + if parameter.default is not inspect.Parameter.empty: + rendered += f" = {parameter.default!r}" + elif parameter.default is not inspect.Parameter.empty: + rendered += f"={parameter.default!r}" + parts.append(rendered) + previous_kind = parameter.kind + + if previous_kind is inspect.Parameter.POSITIONAL_ONLY: + parts.append("/") + + text = "(" + ", ".join(parts) + ")" + if signature.return_annotation is not inspect.Signature.empty: + text += f" -> {render_annotation(signature.return_annotation)}" + return text + + +def markdown_signature(func): + """``render_signature`` escaped for a markdown table cell.""" + return render_signature(func).replace("|", r"\|") diff --git a/microdf/tests/test_docs_api_coverage.py b/microdf/tests/test_docs_api_coverage.py index a0bd4f5c..e835e9b5 100644 --- a/microdf/tests/test_docs_api_coverage.py +++ b/microdf/tests/test_docs_api_coverage.py @@ -11,6 +11,7 @@ import pytest import microdf as mdf +from microdf._docs import markdown_signature DOCS = Path(__file__).resolve().parents[2] / "docs" / "api.md" @@ -122,6 +123,14 @@ def test_no_row_is_missing_its_description(): def test_signatures_match_the_live_ones(): """The rendered signature must be the one the code actually has. + The comparison is structural, not textual: both sides go through + `microdf._docs.markdown_signature`, which builds the text from parameter + names, kinds and defaults (stable across versions) and normalises the + annotations, so `pandas.core.series.Series` under pandas 2 and + `pandas.Series` under pandas 3 render alike, as do `Optional[int]` and the + `int | None` that Python 3.14 reprs it as. `docs/build_api.py` renders the + page with the same function, so the page and this test cannot disagree. + Rows are attributed to the class whose `## ` heading they fall under, since several names exist on both. """ @@ -141,10 +150,11 @@ def test_signatures_match_the_live_ones(): if func is None or isinstance(func, property): continue try: - live = str(inspect.signature(func)) + live = markdown_signature(func) except (TypeError, ValueError): continue - live = live.replace("(self, ", "(").replace("(self)", "()").replace("|", "\\|") if live != rendered: mismatches.append((current.__name__, name, rendered, live)) - assert not mismatches, f"page is out of date with the code: {mismatches}" + assert not mismatches, ( + f"page is out of date with the code; rerun docs/build_api.py: {mismatches}" + ) From a37aab028a91c9bf5aa62f9853805dd83a07fec1 Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Fri, 18 Sep 2026 13:08:06 +0100 Subject: [PATCH 7/7] Follow #330: cov and corr are weighted, and satisfy docformatter #330 merged, so MicroDataFrame.cov and corr now use the weights. Drops the warning box and folds both back into the weighted aggregation table, with the descriptions restored to frequency-weighted, and flips the assertion in test_documented_weight_behaviour_holds to the replicated sample. It also now asserts each frame cell equals the corresponding MicroSeries value, which is the property #330 introduced. The CI Lint failure was docformatter rather than ruff - make lint runs both, and the new _docs.py needed its docstrings rewrapped. Merges main in, so the branch carries #330 and #331. --- docs/api.md | 14 ++------------ docs/build_api.py | 15 +++------------ microdf/_docs.py | 8 ++++---- microdf/tests/test_docs_api_coverage.py | 10 +++++----- 4 files changed, 14 insertions(+), 33 deletions(-) diff --git a/docs/api.md b/docs/api.md index 324b2f4c..802ff917 100644 --- a/docs/api.md +++ b/docs/api.md @@ -99,18 +99,8 @@ the series can compute, including the Gini coefficient and quantiles. | Method | Signature | Description | |---|---|---| | `sum` | `(axis: Union[int, str, NoneType] = 0, skipna: bool = True, numeric_only: bool = False, min_count: int = 0, **kwargs) -> Union[Series, MicroSeries, float]` | Sum numeric columns, weighting reductions across observations. | - -```{warning} -`MicroDataFrame.cov()` and `MicroDataFrame.corr()` return pandas' **unweighted** -results; the weights are ignored. Use `MicroSeries.cov()` and `MicroSeries.corr()` -on a pair of columns for the frequency-weighted values. See -[#327](https://github.com/PolicyEngine/microdf/issues/327). -``` - -| Method | Signature | Description | -|---|---|---| -| `cov` | `(min_periods: Optional[int] = None, ddof: Optional[int] = 1, numeric_only: bool = False) -> DataFrame` | Pairwise covariance of the columns, **unweighted**. | -| `corr` | `(method: CorrelationMethod = 'pearson', min_periods: int = 1, numeric_only: bool = False) -> DataFrame` | Pairwise Pearson correlation of the columns, **unweighted**. | +| `cov` | `(min_periods: Optional[int] = None, ddof: int = 1, numeric_only: bool = False) -> DataFrame` | Pairwise frequency-weighted covariance of the columns. | +| `corr` | `(method: str = 'pearson', min_periods: int = 1, numeric_only: bool = False) -> DataFrame` | Pairwise frequency-weighted Pearson correlation of the columns. | ### Weight-preserving operations diff --git a/docs/build_api.py b/docs/build_api.py index 5bea67e3..083709d8 100644 --- a/docs/build_api.py +++ b/docs/build_api.py @@ -44,9 +44,9 @@ ("MicroSeries", "sqrt"): "Element-wise square root, preserving weights.", ("MicroSeries", "copy"): "Copy the series and its weights.", ("MicroSeries", "equals"): "True when both the values and the weights are equal.", - ("MicroDataFrame", "cov"): "Pairwise covariance of the columns, **unweighted**.", + ("MicroDataFrame", "cov"): "Pairwise frequency-weighted covariance of the columns.", ("MicroDataFrame", "corr"): ( - "Pairwise Pearson correlation of the columns, **unweighted**." + "Pairwise frequency-weighted Pearson correlation of the columns." ), ("MicroDataFrame", "merge"): ( "Database-style join that carries the weight column through." @@ -202,16 +202,7 @@ def build(): ### Weighted aggregation -{table(FRAME, ["sum"])} - -```{{warning}} -`MicroDataFrame.cov()` and `MicroDataFrame.corr()` return pandas' **unweighted** -results; the weights are ignored. Use `MicroSeries.cov()` and `MicroSeries.corr()` -on a pair of columns for the frequency-weighted values. See -[#327](https://github.com/PolicyEngine/microdf/issues/327). -``` - -{table(FRAME, ["cov", "corr"])} +{table(FRAME, ["sum", "cov", "corr"])} ### Weight-preserving operations diff --git a/microdf/_docs.py b/microdf/_docs.py index 16d6a60d..4c985f7e 100644 --- a/microdf/_docs.py +++ b/microdf/_docs.py @@ -1,10 +1,10 @@ """Version-stable rendering of signatures for the API reference. `str(inspect.signature(...))` is not stable across versions: pandas 2 renders a -Series annotation as ``pandas.core.series.Series`` and pandas 3 renders the same -annotation as ``pandas.Series``, and from Python 3.14 ``Optional[int]`` reprs as -``int | None``. Generating ``docs/api.md`` from one of those and testing it -against another is a guaranteed CI failure on some job. +Series annotation as ``pandas.core.series.Series`` and pandas 3 renders the +same annotation as ``pandas.Series``, and from Python 3.14 ``Optional[int]`` +reprs as ``int | None``. Generating ``docs/api.md`` from one of those and +testing it against another is a guaranteed CI failure on some job. So the page and its test both render signatures through the functions here, which strip module qualifiers and put unions into a single form. There is one diff --git a/microdf/tests/test_docs_api_coverage.py b/microdf/tests/test_docs_api_coverage.py index e835e9b5..2a148201 100644 --- a/microdf/tests/test_docs_api_coverage.py +++ b/microdf/tests/test_docs_api_coverage.py @@ -86,12 +86,12 @@ def test_documented_weight_behaviour_holds(): {"x": [1.0, 2.0, 3.0] + [4.0] * 5, "y": [1.0, 4.0, 2.0] + [8.0] * 5} ) - # The page says these are unweighted, and points at #327. - plain = pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0], "y": [1.0, 4.0, 2.0, 8.0]}) - assert frame.cov().loc["x", "y"] == pytest.approx(plain.cov().loc["x", "y"]) - assert frame.corr().loc["x", "y"] == pytest.approx(plain.corr().loc["x", "y"]) + # The page says these are frequency-weighted, as of #330. + assert frame.cov().loc["x", "y"] == pytest.approx(replicated.cov().loc["x", "y"]) + assert frame.corr().loc["x", "y"] == pytest.approx(replicated.corr().loc["x", "y"]) - # The page says the MicroSeries versions are frequency-weighted. + # Each frame cell is the corresponding MicroSeries value. + assert frame.cov().loc["x", "y"] == pytest.approx(frame.x.cov(frame.y)) assert frame.x.cov(frame.y) == pytest.approx(replicated.cov().loc["x", "y"]) # The page says equals compares weights.