diff --git a/.github/workflows/pr.yaml b/.github/workflows/pr.yaml index 9226ece4..a06edc2c 100644 --- a/.github/workflows/pr.yaml +++ b/.github/workflows/pr.yaml @@ -26,11 +26,17 @@ jobs: run: make lint test: - name: Test Python ${{ matrix.python-version }} + name: Test Python ${{ matrix.python-version }} / pandas ${{ matrix.pandas-version }} runs-on: ubuntu-latest strategy: matrix: python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"] + pandas-version: ["latest"] + include: + - python-version: "3.13" + pandas-version: "2" + - python-version: "3.13" + pandas-version: "3" steps: - uses: actions/checkout@v6 @@ -46,6 +52,9 @@ jobs: - name: Install dependencies run: | uv pip install -e ".[dev]" --system + - name: Select pandas major version + if: matrix.pandas-version != 'latest' + run: uv pip install --system "pandas==${{ matrix.pandas-version }}.*" - name: Run tests with coverage run: make test - name: Upload coverage to Codecov diff --git a/README.md b/README.md index 8b0cc96e..af45166e 100644 --- a/README.md +++ b/README.md @@ -31,10 +31,16 @@ the two ways to get a believable-looking wrong answer. ## Key Features - **MicroDataFrame**: A pandas DataFrame with an integrated weight column - **MicroSeries**: A pandas Series with integrated weights -- **Weighted operations**: All aggregations (sum, mean, median, etc.) automatically use weights +- **Weighted operations**: Supported aggregations (sum, mean, median, etc.) use weights; unsupported estimators raise - **Inequality metrics**: Built-in Gini coefficient calculation - **Poverty analysis**: Integrated poverty rate and gap calculations +Weight preservation is limited to the tested operations in the +[support matrix](docs/support.md). Convert explicitly to `pd.Series(s)` or +`pd.DataFrame(df)` to request unweighted pandas behaviour. Use `microdf.concat` +to reject mixed weighted and plain inputs regardless of their order; pandas can +bypass microdf's checks when a plain input comes first in `pd.concat`. + ## Installation Install with: @@ -57,7 +63,7 @@ df = pd.DataFrame( # Create a MicroDataFrame mdf_df = mdf.MicroDataFrame(df, weights="weights") -# All operations are weight-aware +# Supported estimators use the row weights print(mdf_df.income.mean()) # Weighted mean print(mdf_df.income.gini()) # Gini coefficient ``` diff --git a/changelog.d/333.fixed.md b/changelog.d/333.fixed.md new file mode 100644 index 00000000..4d2a3b38 --- /dev/null +++ b/changelog.d/333.fixed.md @@ -0,0 +1 @@ +Preserve weights through aggregation, construction, pandas conversions and row operations; reject conflicting arithmetic weights and unsupported estimators. Add checked microdf.concat and a regression-backed support matrix. diff --git a/changelog.d/334.fixed.md b/changelog.d/334.fixed.md new file mode 100644 index 00000000..ea9c5050 --- /dev/null +++ b/changelog.d/334.fixed.md @@ -0,0 +1 @@ +Clarify the units of poverty gap totals and test all poverty estimators with non-uniform and zero weights and threshold boundaries. diff --git a/changelog.d/335.fixed.md b/changelog.d/335.fixed.md new file mode 100644 index 00000000..133e697e --- /dev/null +++ b/changelog.d/335.fixed.md @@ -0,0 +1 @@ +Describe frequency-weighted covariance and Pearson correlation in the examples, with a checked NumPy example. diff --git a/docs/api.md b/docs/api.md index 802ff917..a169ec70 100644 --- a/docs/api.md +++ b/docs/api.md @@ -4,7 +4,8 @@ vector; `MicroDataFrame` is a `pandas.DataFrame` carrying a weight column. Both behave like their pandas counterparts, and the methods below either add a weighted estimator or preserve weights through an operation that would otherwise -drop them. +drop them. The [support matrix](support.md) lists tested operations and explicit +rejections; arbitrary pandas operations do not necessarily preserve weights. ```python import microdf as mdf @@ -31,6 +32,8 @@ These have the same names as their pandas equivalents and return weighted result | `cov` | `(other: Series, min_periods: Optional[int] = None, ddof: int = 1, *, skipna: bool = True) -> float` | Calculate frequency-weighted covariance with another Series. | | `corr` | `(other: Series, method: str = 'pearson', min_periods: Optional[int] = None, *, ddof: int = 1, skipna: bool = True) -> float` | Calculate frequency-weighted Pearson correlation. | | `rank` | `(pct: Optional[bool] = False) -> Series` | Weighted rank of each element. | +| `value_counts` | `(normalize=False, sort=True, ascending=False, bins=None, dropna=True) -> Series` | Sum weights by value; optionally divide by the included weight total. | +| `mode` | `(dropna: bool = True) -> Series` | Return values with the greatest positive total weight as a plain Series. | ### Weight-preserving operations @@ -44,6 +47,7 @@ Operations that change the shape or type of the data, overridden so weights stay | `clip` | `(lower: Optional[float] = None, upper: Optional[float] = None, axis: Optional[int] = None, inplace: Optional[bool] = False, *args, **kwargs) -> MicroSeries` | Trim values at the given thresholds, preserving weights. | | `round` | `(decimals: Optional[int] = 0, *args, **kwargs) -> MicroSeries` | Round each value, preserving weights. | | `repeat` | `(repeats, axis=None)` | Repeat elements, repeating their weights alongside. | +| `explode` | `(ignore_index: bool = False) -> MicroSeries` | Expand list entries, repeating each observation's weight. | | `sqrt` | `() -> MicroSeries` | Element-wise square root, preserving weights. | | `copy` | `(deep: Optional[bool] = True)` | Copy the series and its weights. | | `equals` | `(other: MicroSeries) -> bool` | True when both the values and the weights are equal. | @@ -101,6 +105,7 @@ the series can compute, including the Gini coefficient and quantiles. | `sum` | `(axis: Union[int, str, NoneType] = 0, skipna: bool = True, numeric_only: bool = False, min_count: int = 0, **kwargs) -> Union[Series, MicroSeries, float]` | Sum numeric columns, weighting reductions across observations. | | `cov` | `(min_periods: Optional[int] = None, ddof: int = 1, numeric_only: bool = False) -> DataFrame` | Pairwise frequency-weighted covariance of the columns. | | `corr` | `(method: str = 'pearson', min_periods: int = 1, numeric_only: bool = False) -> DataFrame` | Pairwise frequency-weighted Pearson correlation of the columns. | +| `pivot_table` | `(values=None, index=None, columns=None, aggfunc='mean', fill_value=None, margins=False, dropna=True, margins_name='All', observed=True, sort=True, **kwargs)` | Build a pivot table by applying estimators to weighted column groups. | ### Weight-preserving operations @@ -110,6 +115,8 @@ the series can compute, including the Gini coefficient and quantiles. | `merge` | `(right, how='inner', on=None, left_on=None, right_on=None, left_index=False, right_index=False, sort=False, suffixes=('_x', '_y'), copy=True, indicator=False, validate=None)` | Database-style join that carries the weight column through. | | `reset_index` | `(level: Optional[int] = None, drop: Optional[bool] = False, inplace: Optional[bool] = False, col_level: Optional[int] = 0, col_fill: Optional[str] = '', allow_duplicates: Optional[bool] = None, names: Optional[list[str]] = None) -> Optional[MicroDataFrame]` | Reset the index, keeping weights aligned to their rows. | | `drop` | `(labels=None, axis=0, index=None, columns=None, level=None, inplace=False, errors='raise')` | Drop rows or columns, keeping weights aligned to the remaining rows. | +| `dropna` | `(*, axis=0, how=None, thresh=None, subset=None, inplace=False, ignore_index=False)` | Drop missing observations and their weights using row positions. | +| `apply` | `(func, axis=0, raw=False, result_type=None, args=(), **kwargs)` | Apply row functions while retaining row weights on the result. | | `astype` | `(dtype, copy: Optional[bool] = True, errors: Optional[str] = 'raise') -> MicroDataFrame` | Convert MicroDataFrame to specified data type while preserving weights. | | `copy` | `(deep: Optional[bool] = True) -> MicroDataFrame` | Copy the frame and its weights. | | `equals` | `(other: MicroDataFrame) -> bool` | True when both the values and the weights are equal. | @@ -118,12 +125,18 @@ the series can compute, including the Gini coefficient and quantiles. | Method | Signature | Description | |---|---|---| -| `poverty_rate` | `(income: str, threshold: str) -> float` | Calculate poverty rate, i.e., the population share with income below their poverty threshold. | -| `poverty_gap` | `(income: str, threshold: str) -> float` | Calculate poverty gap, i.e., the total gap between income and poverty thresholds for all people in poverty. | +| `poverty_rate` | `(income: str, threshold: str) -> float` | Return the weighted headcount share strictly below the poverty threshold. | +| `poverty_gap` | `(income: str, threshold: str) -> float` | Return the weighted aggregate poverty gap in income currency units. | | `poverty_count` | `(income: Union[MicroSeries, str], threshold: Union[MicroSeries, str]) -> int` | Calculates the number of entities with income below a poverty threshold. | -| `deep_poverty_rate` | `(income: str, threshold: str) -> float` | Calculate deep poverty rate, i.e., the population share with income below half their poverty threshold. | -| `deep_poverty_gap` | `(income: str, threshold: str) -> float` | Calculate deep poverty gap, i.e., the total gap between income and half of poverty thresholds for all people in deep poverty. | -| `squared_poverty_gap` | `(income: str, threshold: str) -> float` | Calculate squared poverty gap, i.e., the total squared gap between income and poverty thresholds for all people in poverty. Also known as the poverty severity index. | +| `deep_poverty_rate` | `(income: str, threshold: str) -> float` | Return the weighted headcount share strictly below half the threshold. | +| `deep_poverty_gap` | `(income: str, threshold: str) -> float` | Return the weighted aggregate deep poverty gap in income currency units. | +| `squared_poverty_gap` | `(income: str, threshold: str) -> float` | Return the weighted aggregate squared gap in squared currency units. | + +The gap methods return currency totals. Normalised FGT(1) (poverty gap index) +and FGT(2) (poverty severity index) are outside this API's scope. In those +indices, divide each positive gap by that row's threshold before raising to +the first or second power, then take the population-weighted mean. People at +the threshold contribute zero, and people with zero weight do not contribute. ### Weights @@ -137,6 +150,7 @@ the series can compute, including the Gini coefficient and quantiles. | Function | Description | |---|---| +| `microdf.concat` | Concatenate Micro objects; reject plain pandas inputs in either order. | | `microdf.replicate_variance` | Variance of a statistic from replicate weights. | | `microdf.replicate_standard_error` | Square root of the above. | diff --git a/docs/build_api.py b/docs/build_api.py index 083709d8..e0bc31ee 100644 --- a/docs/build_api.py +++ b/docs/build_api.py @@ -106,7 +106,8 @@ def build(): vector; `MicroDataFrame` is a `pandas.DataFrame` carrying a weight column. Both behave like their pandas counterparts, and the methods below either add a weighted estimator or preserve weights through an operation that would otherwise -drop them. +drop them. The [support matrix](support.md) lists tested operations and explicit +rejections; arbitrary pandas operations do not necessarily preserve weights. ```python import microdf as mdf @@ -135,6 +136,8 @@ def build(): "cov", "corr", "rank", + "value_counts", + "mode", ], ) } @@ -153,6 +156,7 @@ def build(): "clip", "round", "repeat", + "explode", "sqrt", "copy", "equals", @@ -202,14 +206,24 @@ def build(): ### Weighted aggregation -{table(FRAME, ["sum", "cov", "corr"])} +{table(FRAME, ["sum", "cov", "corr", "pivot_table"])} ### Weight-preserving operations { table( FRAME, - ["groupby", "merge", "reset_index", "drop", "astype", "copy", "equals"], + [ + "groupby", + "merge", + "reset_index", + "drop", + "dropna", + "apply", + "astype", + "copy", + "equals", + ], ) } @@ -229,6 +243,12 @@ def build(): ) } +The gap methods return currency totals. Normalised FGT(1) (poverty gap index) +and FGT(2) (poverty severity index) are outside this API's scope. In those +indices, divide each positive gap by that row's threshold before raising to +the first or second power, then take the population-weighted mean. People at +the threshold contribute zero, and people with zero weight do not contribute. + ### Weights {table(FRAME, ["set_weights", "set_weight_col", "nullify_weights"])} @@ -237,6 +257,7 @@ def build(): | Function | Description | |---|---| +| `microdf.concat` | Concatenate Micro objects; reject plain pandas inputs in either order. | | `microdf.replicate_variance` | Variance of a statistic from replicate weights. | | `microdf.replicate_standard_error` | Square root of the above. | diff --git a/docs/build_support.py b/docs/build_support.py new file mode 100644 index 00000000..1f05f13a --- /dev/null +++ b/docs/build_support.py @@ -0,0 +1,7 @@ +"""Generate the support matrix from the executable regression cases.""" + +from pathlib import Path +from microdf.tests.test_fail_closed import support_matrix + +if __name__ == "__main__": + Path(__file__).with_name("support.md").write_text(support_matrix()) diff --git a/docs/examples.md b/docs/examples.md index a5ee4d66..a25d2166 100644 --- a/docs/examples.md +++ b/docs/examples.md @@ -41,14 +41,24 @@ new Micro object with appropriate weights when that choice is intentional. A single DataFrame row is a plain pandas `Series`, because its entries are columns rather than weighted observations. -`MicroDataFrame.cov()` and `.corr()` retain pandas' unweighted calculations -and return plain pandas `DataFrame` matrices. Their rows describe columns, -so observation weights do not apply to the result or subsequent operations -such as `.sum()`. These methods accept the installed pandas version's -arguments and defaults, including missing-value handling and correlation -methods. - -Use Micro objects for **every input** to `pd.concat`. A mixed concat raises +`MicroDataFrame.cov()` computes frequency-weighted covariance, and `.corr()` +computes frequency-weighted Pearson correlation. Each matrix cell applies the +corresponding `MicroSeries` estimator to the pair of columns, excluding missing +pairs and zero-weight rows. Integer weights agree with the replicated sample. +Both methods return plain pandas `DataFrame` matrices: their rows describe +columns, so subsequent matrix operations have no observation weights. Other +correlation methods are unsupported. The example below returns the covariance +matrix `[[8/3, 4/3], [4/3, 2]]`. + +```python +import numpy as np + +xy = MicroDataFrame({"x": [1, 3, 5], "y": [2, 5, 4]}, weights=[1, 2, 1]) +assert np.allclose(xy.cov(), np.cov(xy.to_numpy().T, fweights=[1, 2, 1])) +``` + +Use `microdf.concat` to reject plain pandas inputs in either order, or use Micro +objects for **every input** to `pd.concat`. A mixed concat raises `ValueError` when pandas calls the Micro object's hooks. If a plain pandas object comes first, pandas can bypass those hooks and return an unweighted object; microdf cannot intercept that dispatch. Convert each input to diff --git a/docs/myst.yml b/docs/myst.yml index d1b54907..21ea40bb 100644 --- a/docs/myst.yml +++ b/docs/myst.yml @@ -21,6 +21,7 @@ project: children: - file: gini.ipynb - file: api.md + - file: support.md site: options: logo: microdf_logo.png diff --git a/docs/support.md b/docs/support.md new file mode 100644 index 00000000..2e7cc4d6 --- /dev/null +++ b/docs/support.md @@ -0,0 +1,54 @@ +# Supported operations + +Generated from `microdf/tests/test_fail_closed.py` by `uv run python docs/build_support.py`. + +The regression suite checks these contracts on pandas 2 and 3. Aggregated results carry no row weights; transformations retain independent copies of the input weights. + +| Operation | Behaviour | +|---|---| +| groupby column mean | Weighted result / preserved weights | +| groupby dict agg | Weighted result / preserved weights | +| groupby named agg | Weighted result / preserved weights | +| groupby callable agg | Weighted result / preserved weights | +| SeriesGroupBy callable agg | Weighted result / preserved weights | +| callable pivot_table | Weighted result / preserved weights | +| row apply | Weighted result / preserved weights | +| frame numeric mean | Weighted result / preserved weights | +| frame numeric median | Weighted result / preserved weights | +| groupby numeric mean | Weighted result / preserved weights | +| groupby numeric median | Weighted result / preserved weights | +| frame constructor | Weighted result / preserved weights | +| series constructor | Weighted result / preserved weights | +| pd.cut preserves weights | Weighted result / preserved weights | +| pd.qcut preserves weights | Weighted result / preserved weights | +| pd.to_numeric preserves weights | Weighted result / preserved weights | +| series explode | Weighted result / preserved weights | +| np.average | Weighted result / preserved weights | +| np.mean | Weighted result / preserved weights | +| np.median | Weighted result / preserved weights | +| weighted value_counts | Weighted result / preserved weights | +| weighted mode | Weighted result / preserved weights | +| numeric frame dropna | Weighted result / preserved weights | +| `rolling` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `expanding` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `ewm` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `sem` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `skew` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `kurt` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `kurtosis` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `prod` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `product` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `idxmax` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| `idxmin` | Raises; use `pd.Series(s)` for unweighted pandas behaviour | +| Arithmetic with conflicting weights | Raises `ValueError` in either operand order | + +`pd.cut` and `pd.qcut` retain row weights, but choose bin edges using pandas' unweighted rules. Supply explicit bin edges for weighted quantile bins. + +`pivot_table` grouping keys must name columns; external Series, callable +groupers and index-level groupers raise. Weighted `Series.value_counts` and +`Series.mode` return plain summary Series; frame and grouped variants raise. + +`microdf.concat` rejects mixed weighted/plain inputs in either order. Direct +`pd.concat` still bypasses microdf when its first input is plain pandas; the +regression suite records this upstream dispatch limitation as an expected failure. +Use Micro objects for every input to `pd.concat`, or use `microdf.concat`. diff --git a/microdf/__init__.py b/microdf/__init__.py index 2e2d83a6..18f342c4 100644 --- a/microdf/__init__.py +++ b/microdf/__init__.py @@ -1,5 +1,6 @@ from importlib.metadata import PackageNotFoundError, version +from .concat import concat from .microdataframe import MicroDataFrame, MicroDataFrameGroupBy from .microseries import MicroSeries, MicroSeriesGroupBy from .replication import replicate_standard_error, replicate_variance @@ -14,6 +15,7 @@ __version__ = "unknown" __all__ = [ + "concat", # microseries.py "MicroSeries", "MicroSeriesGroupBy", diff --git a/microdf/_weights.py b/microdf/_weights.py index 3dd72456..6109403b 100644 --- a/microdf/_weights.py +++ b/microdf/_weights.py @@ -23,6 +23,30 @@ def aligned_weights(source, index): ) +def require_equal_weights(left, right): + """Reject conflicting weights before pandas discards operand metadata.""" + if isinstance(right, WeightPropagationMixin): + other = aligned_weights(right, left.index) + if not np.array_equal(left.weights, other, equal_nan=True): + raise ValueError("Arithmetic requires equal row weights on both operands.") + + +def unsupported_weighted(name): + """Build an explicit escape hatch for operations without weighted + semantics.""" + + def method(self, *args, **kwargs): + plain = "pd.Series(s)" if self.ndim == 1 else "pd.DataFrame(df)" + raise NotImplementedError( + f"{name} has no supported weighted definition; use {plain} " + "for an explicitly unweighted operation." + ) + + method.__name__ = name + method.__doc__ = f"Reject {name}; convert to plain pandas for unweighted behaviour." + return method + + def finalize_weights(result, source, method, previous_weights): """Propagate row weights after pandas has finalized its own metadata.""" if method == "transpose" and result.ndim == 2: @@ -124,6 +148,46 @@ def concat_weights(result, objects, axis): class WeightPropagationMixin: """Use positional provenance for pandas operations that select rows.""" + mode = unsupported_weighted("mode") + value_counts = unsupported_weighted("value_counts") + rolling = unsupported_weighted("rolling") + expanding = unsupported_weighted("expanding") + ewm = unsupported_weighted("ewm") + sem = unsupported_weighted("sem") + skew = unsupported_weighted("skew") + kurt = unsupported_weighted("kurt") + kurtosis = unsupported_weighted("kurtosis") + prod = unsupported_weighted("prod") + product = unsupported_weighted("product") + idxmax = unsupported_weighted("idxmax") + idxmin = unsupported_weighted("idxmin") + + def _arith_method(self, other, op): + require_equal_weights(self, other) + return super()._arith_method(other, op) + + def _logical_method(self, other, op): + require_equal_weights(self, other) + return super()._logical_method(other, op) + + def _cmp_method(self, other, op): + if isinstance(other, pd.Series) and not self.index.equals(other.index): + return super()._cmp_method(other, op) + require_equal_weights(self, other) + return super()._cmp_method(other, op) + + def _binop(self, other, func, *args, **kwargs): + require_equal_weights(self, other) + return super()._binop(other, func, *args, **kwargs) + + def _flex_arith_method(self, other, op, *args, **kwargs): + require_equal_weights(self, other) + return super()._flex_arith_method(other, op, *args, **kwargs) + + def _flex_cmp_method(self, other, op, *args, **kwargs): + require_equal_weights(self, other) + return super()._flex_cmp_method(other, op, *args, **kwargs) + def _plain(self): return ( pd.Series(self, copy=False) diff --git a/microdf/concat.py b/microdf/concat.py new file mode 100644 index 00000000..25703f7d --- /dev/null +++ b/microdf/concat.py @@ -0,0 +1,33 @@ +"""Concatenation with validation before pandas chooses an output class.""" + +from collections.abc import Mapping + +import pandas as pd + +from ._weights import WeightPropagationMixin + + +def concat(objs, **kwargs): + """Concatenate weighted objects, rejecting plain inputs in either order. + + Accept pandas.concat keyword arguments. Every non-None input must carry + weights; construct Micro objects with explicit weights before calling. + """ + objs = dict(objs) if isinstance(objs, Mapping) else list(objs) + keys = kwargs.get("keys") + if keys is not None: + keys = list(keys) + kwargs["keys"] = keys + selected = ( + [objs[key] for key in (objs if keys is None else keys)] + if isinstance(objs, Mapping) + else objs + ) + if any( + obj is not None and not isinstance(obj, WeightPropagationMixin) + for obj in selected + ): + raise ValueError( + "Cannot concatenate weighted and unweighted objects: provide explicit weights for every input." + ) + return pd.concat(objs, **kwargs) diff --git a/microdf/microdataframe.py b/microdf/microdataframe.py index 32be4a00..dd9b758f 100644 --- a/microdf/microdataframe.py +++ b/microdf/microdataframe.py @@ -1,4 +1,3 @@ -import copy import logging import warnings from functools import wraps @@ -21,6 +20,7 @@ aligned_weights, finalize_weights, weight_series, + require_equal_weights, ) logger = logging.getLogger(__name__) @@ -50,7 +50,7 @@ def __init__(self, *args, weights=None, **kwargs): weight_source = args[0] if args else kwargs.get("data") if isinstance(weight_source, dict) and len(weight_source) == 1: weight_source = next(iter(weight_source.values())) - if weights is None and isinstance(weight_source, MicroSeries): + if weights is None and isinstance(weight_source, (MicroSeries, MicroDataFrame)): weights = aligned_weights(weight_source, self.index) self.weights = weight_series(np.ones(len(self)), self.index) self.weights_col = None @@ -82,6 +82,151 @@ def __finalize__(self, other, method=None, **kwargs): super().__finalize__(other, method=method, **kwargs) return finalize_weights(self, other, method, previous) + def __array_ufunc__(self, ufunc, method, *inputs, **kwargs): + for value in inputs: + if value is not self: + require_equal_weights(self, value) + if method != "__call__": + raise NotImplementedError( + "Use a weighted frame reduction or pd.DataFrame(df) explicitly" + ) + result = super().__array_ufunc__(ufunc, method, *inputs, **kwargs) + + def restore(value): + if isinstance(value, pd.DataFrame): + return self._weighted_result(value, aligned_weights(self, value.index)) + return value + + return ( + tuple(restore(value) for value in result) + if isinstance(result, tuple) + else restore(result) + ) + + def apply(self, func, axis=0, raw=False, result_type=None, args=(), **kwargs): + """Apply row functions while retaining row weights on the result.""" + if self._get_axis_number(axis) != 1: + # Column callbacks must receive the weighted series. + if raw or result_type is not None or not callable(func): + raise NotImplementedError("Use pd.DataFrame(df) for this apply form") + return pd.Series( + {col: func(self[col], *args, **kwargs) for col in self.columns} + ) + result = pd.DataFrame(self).apply( + func, axis=1, raw=raw, result_type=result_type, args=args, **kwargs + ) + cls = MicroSeries if isinstance(result, pd.Series) else MicroDataFrame + return cls(result, weights=self.weights) + + def dropna( + self, + *, + axis=0, + how=None, + thresh=None, + subset=None, + inplace=False, + ignore_index=False, + ): + """Drop missing observations and their weights using row positions.""" + plain = pd.DataFrame(self).copy(deep=False) + axis = self._get_axis_number(axis) + original_index = plain.index + plain.index = pd.RangeIndex(len(plain)) + options = {"axis": axis} + if subset is not None: + if axis == 1: + raise NotImplementedError( + "dropna(axis=1, subset=...) requires pd.DataFrame(df)" + ) + options["subset"] = subset + if how is not None: + options["how"] = how + if thresh is not None: + options["thresh"] = thresh + result = plain.dropna(**options) + positions = np.asarray(result.index, dtype=int) + result.index = ( + pd.RangeIndex(len(result)) + if ignore_index + else original_index.take(positions) + ) + return self._finish_row_operation(result, positions, inplace) + + def pivot_table( + self, + values=None, + index=None, + columns=None, + aggfunc="mean", + fill_value=None, + margins=False, + dropna=True, + margins_name="All", + observed=True, + sort=True, + **kwargs, + ): + """Build a pivot table by applying estimators to weighted column + groups. + + Grouping keys must name columns. External Series, callables and index + level groupers require an explicit conversion to a plain DataFrame. + """ + for keys in (index, columns): + if keys is None: + continue + keys = keys if isinstance(keys, list) else [keys] + if any( + not pd.api.types.is_hashable(key) or key not in self.columns + for key in keys + ): + raise NotImplementedError( + "Weighted pivot_table grouping keys must name columns; " + "use pd.DataFrame(df) for other grouping forms" + ) + plain = pd.DataFrame(self).reset_index(drop=True) + + def wrap(func): + if isinstance(func, dict): + return {key: wrap(value) for key, value in func.items()} + if isinstance(func, (list, tuple)): + return [wrap(value) for value in func] + + def aggregate(series): + positions = np.asarray(series.index, dtype=int) + weighted = MicroSeries( + series.to_numpy(), + index=self.index.take(positions), + name=series.name, + weights=self.weights.iloc[positions], + ) + return ( + getattr(weighted, func)(**kwargs) + if isinstance(func, str) + else func(weighted, **kwargs) + ) + + aggregate.__name__ = ( + func + if isinstance(func, str) + else getattr(func, "__name__", "aggregate") + ) + return aggregate + + return plain.pivot_table( + values=values, + index=index, + columns=columns, + aggfunc=wrap(aggfunc), + fill_value=fill_value, + margins=margins, + dropna=dropna, + margins_name=margins_name, + observed=observed, + sort=sort, + ) + def cov( self, min_periods: Optional[int] = None, @@ -255,19 +400,19 @@ def _create_scalar_function(self, name: str) -> Callable: """ def fn(*args, **kwargs) -> pd.Series: - results = {} - for col in self.columns: - if pd.api.types.is_numeric_dtype(self[col]): - try: - results[col] = getattr(self[col], name)(*args, **kwargs) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - return pd.Series(results) + kwargs.pop("numeric_only", None) + axis = kwargs.pop("axis", 0) + if axis not in (0, "index"): + raise NotImplementedError( + f"Weighted {name} only supports axis=0; use pd.DataFrame(df)" + ) + return pd.Series( + { + col: getattr(self[col], name)(*args, **kwargs) + for col in self.columns + if pd.api.types.is_numeric_dtype(self[col]) + } + ) return fn @@ -717,7 +862,6 @@ def equals(self, other: "MicroDataFrame") -> bool: equal_weights = self.weights.equals(other.weights) return equal_values and equal_weights - @get_args_as_micro_series() def groupby(self, by: Union[str, List], *args, **kwargs) -> "MicroDataFrameGroupBy": """Returns a GroupBy object with MicroSeriesGroupBy objects for each column. @@ -734,23 +878,21 @@ def groupby(self, by: Union[str, List], *args, **kwargs) -> "MicroDataFrameGroup # DataFrame — any later ``df.sum()`` or ``list(df.columns)`` # would then include it. staged = pd.DataFrame(self).copy() + if "__tmp_weights" in staged.columns: + raise ValueError("Rename the reserved __tmp_weights column before grouping") staged["__tmp_weights"] = np.asarray(self.weights.values, dtype=float) gb = staged.groupby(by, *args, **kwargs) - weights = copy.deepcopy(gb["__tmp_weights"]) - for col in staged.columns: # df.groupby(...)[col]s use weights - res = gb[col] - res.__class__ = MicroSeriesGroupBy - res._init() - res.weights = weights - setattr(gb, col, res) gb.__class__ = MicroDataFrameGroupBy gb._init(by) return gb @get_args_as_micro_series() def poverty_rate(self, income: str, threshold: str) -> float: - """Calculate poverty rate, i.e., the population share with income below - their poverty threshold. + """Return the weighted headcount share strictly below the poverty + threshold. + + Divide the weight of people in poverty by total population weight. This + is the Foster-Greer-Thorbecke (FGT) headcount index, FGT(0). :param income: Column indicating income. :type income: str @@ -764,8 +906,10 @@ def poverty_rate(self, income: str, threshold: str) -> float: @get_args_as_micro_series() def deep_poverty_rate(self, income: str, threshold: str) -> float: - """Calculate deep poverty rate, i.e., the population share with income - below half their poverty threshold. + """Return the weighted headcount share strictly below half the + threshold. + + Divide the weight of people in deep poverty by total population weight. :param income: Column indicating income. :type income: str @@ -779,14 +923,16 @@ def deep_poverty_rate(self, income: str, threshold: str) -> float: @get_args_as_micro_series() def poverty_gap(self, income: str, threshold: str) -> float: - """Calculate poverty gap, i.e., the total gap between income and - poverty thresholds for all people in poverty. + """Return the weighted aggregate poverty gap in income currency units. + + Sum weight times (threshold - income) over people strictly below their + threshold. This aggregate is not the normalised FGT(1) index. :param income: Column indicating income. :type income: str :param threshold: Column indicating threshold. :type threshold: str - :return: Poverty gap. + :return: Weighted aggregate gap in income currency units. :rtype: float """ gaps = (threshold - income)[threshold > income] @@ -794,14 +940,17 @@ def poverty_gap(self, income: str, threshold: str) -> float: @get_args_as_micro_series() def deep_poverty_gap(self, income: str, threshold: str) -> float: - """Calculate deep poverty gap, i.e., the total gap between income and - half of poverty thresholds for all people in deep poverty. + """Return the weighted aggregate deep poverty gap in income currency + units. + + Sum weight times (threshold / 2 - income) over people strictly below + half their threshold. :param income: Column indicating income. :type income: str :param threshold: Column indicating threshold. :type threshold: str - :return: Deep poverty gap. + :return: Weighted aggregate deep gap in income currency units. :rtype: float """ deep_threshold = threshold / 2 @@ -810,15 +959,17 @@ def deep_poverty_gap(self, income: str, threshold: str) -> float: @get_args_as_micro_series() def squared_poverty_gap(self, income: str, threshold: str) -> float: - """Calculate squared poverty gap, i.e., the total squared gap between - income and poverty thresholds for all people in poverty. Also known as - the poverty severity index. + """Return the weighted aggregate squared gap in squared currency units. + + Sum weight times (threshold - income) squared over people strictly + below their threshold. This aggregate is not the normalised FGT(2) + poverty severity index. :param income: Column indicating income. :type income: str :param threshold: Column indicating threshold. :type threshold: str - :return: Squared poverty gap. + :return: Weighted aggregate squared gap in squared income currency units. :rtype: float """ gaps = (threshold - income)[threshold > income] @@ -874,161 +1025,111 @@ def __repr__(self) -> str: class MicroDataFrameGroupBy(pd.core.groupby.generic.DataFrameGroupBy): - def _init(self, by: Union[str, List]): + def _init(self, by, columns=None, weights=None): self._by = by - self.columns = list(self.obj.columns) - if isinstance(by, list): - for column in by: - self.columns.remove(column) - elif isinstance(by, str): - self.columns.remove(by) - self.columns.remove("__tmp_weights") - # Filter to only numeric columns + self.columns = ( + columns + if columns is not None + else [ + col + for col in self.obj.columns + if col != "__tmp_weights" and col not in self.exclusions + ] + ) self.numeric_columns = [ col for col in self.columns if pd.api.types.is_numeric_dtype(self.obj[col]) ] - # Store reference to weights groupby for column selection - self._weights_groupby = copy.deepcopy(super().__getitem__("__tmp_weights")) - for fn_name in MicroSeries.SCALAR_FUNCTIONS: - - def get_fn(name): - def fn(*args, **kwargs): - results = {} - for col in self.numeric_columns: - try: - results[col] = getattr(getattr(self, col), name)( - *args, **kwargs - ) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - # Return plain DataFrame - aggregated results don't have - # per-row weights (weights were already applied) - return pd.DataFrame(results) if results else pd.DataFrame() - - return fn - - setattr(self, fn_name, get_fn(fn_name)) - for fn_name in MicroSeries.VECTOR_FUNCTIONS: - - def get_fn(name) -> Callable: - def fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: - results = {} - for col in self.numeric_columns: - try: - results[col] = getattr(getattr(self, col), name)( - *args, **kwargs - ) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug("skipping column %s in %s: %s", col, name, exc) - # Return plain DataFrame - aggregated results don't have - # per-row weights (weights were already applied) - return pd.DataFrame(results) if results else pd.DataFrame() - - return fn - - setattr(self, fn_name, get_fn(fn_name)) - - def __getitem__( - self, key: Union[str, List] - ) -> Union["MicroSeriesGroupBy", "MicroDataFrameGroupBy"]: - """Select columns from the groupby object while preserving weights. + self._weights_groupby = ( + weights if weights is not None else self._gotitem("__tmp_weights", ndim=1) + ) + for name in MicroSeries.FUNCTIONS + [ + "sem", + "skew", + "kurt", + "kurtosis", + "prod", + "product", + "idxmax", + "idxmin", + "rolling", + "expanding", + "ewm", + "mode", + "value_counts", + ]: + + def reduction(*args, _name=name, **kwargs): + kwargs.pop("numeric_only", None) + result = pd.DataFrame( + { + col: getattr(self[col], _name)(*args, **kwargs) + for col in self.numeric_columns + } + ) + return result if self.as_index else result.reset_index() - This ensures that operations like groupby(col)["y"].sum() or - groupby(col)[["y"]].sum() use weighted aggregation. + setattr(self, name, reduction) - :param key: Column name or list of column names - :return: MicroSeriesGroupBy for single column, MicroDataFrameGroupBy - for multiple columns - """ - if isinstance(key, str): - # Single column - return MicroSeriesGroupBy + def __getitem__(self, key): + if pd.api.types.is_hashable(key): + if key not in self.columns: + raise KeyError(key) + result = self._gotitem(key, ndim=1) + else: result = super().__getitem__(key) + if isinstance(result, pd.core.groupby.generic.SeriesGroupBy): result.__class__ = MicroSeriesGroupBy result._init() result.weights = self._weights_groupby - return result else: - # Multiple columns - return a new MicroDataFrameGroupBy - # with only the selected columns - result = super().__getitem__(key) result.__class__ = MicroDataFrameGroupBy - # Re-initialize with the subset of columns - result._by = self._by - result.columns = list(key) if hasattr(key, "__iter__") else [key] - result.numeric_columns = [ - col - for col in result.columns - if pd.api.types.is_numeric_dtype(result.obj[col]) - ] - result._weights_groupby = self._weights_groupby - # Set up the column attributes as MicroSeriesGroupBy - for col in result.columns: - col_gb = super().__getitem__(col) - col_gb.__class__ = MicroSeriesGroupBy - col_gb._init() - col_gb.weights = self._weights_groupby - setattr(result, col, col_gb) - # Set up the scalar and vector functions - for fn_name in MicroSeries.SCALAR_FUNCTIONS: - - def get_scalar_fn(name, res): - def fn(*args, **kwargs): - results = {} - for col in res.numeric_columns: - try: - results[col] = getattr(getattr(res, col), name)( - *args, **kwargs - ) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug( - "skipping column %s in %s: %s", col, name, exc - ) - # Return plain DataFrame - aggregated results don't - # have per-row weights (weights were already applied) - return pd.DataFrame(results) if results else pd.DataFrame() - - return fn - - setattr(result, fn_name, get_scalar_fn(fn_name, result)) - for fn_name in MicroSeries.VECTOR_FUNCTIONS: - - def get_vector_fn(name, res): - def fn(*args, **kwargs): - results = {} - for col in res.numeric_columns: - try: - results[col] = getattr(getattr(res, col), name)( - *args, **kwargs - ) - except TypeError as exc: - # Skip columns whose dtype can't take this aggregation. - # Deliberately narrow: catching every Exception here also - # swallowed real errors (e.g. the ValueError from - # gini(negatives=...)) and returned a silently truncated - # result instead of raising. - logger.debug( - "skipping column %s in %s: %s", col, name, exc - ) - # Return plain DataFrame - aggregated results don't - # have per-row weights (weights were already applied) - return pd.DataFrame(results) if results else pd.DataFrame() - - return fn - - setattr(result, fn_name, get_vector_fn(fn_name, result)) - return result + result._init(self._by, list(key), self._weights_groupby) + return result + + def aggregate(self, func=None, *args, **kwargs): + """Apply named, dictionary, list and callable weighted aggregations.""" + if func is not None and ( + kwargs.get("engine") is not None or "engine_kwargs" in kwargs + ): + raise NotImplementedError( + "Weighted aggregation does not support engine overrides" + ) + numeric_only = kwargs.pop("numeric_only", False) if func is not None else False + if func is None: + if not kwargs or not all( + isinstance(value, tuple) and len(value) == 2 + for value in kwargs.values() + ): + raise TypeError("Named aggregation requires output=(column, function)") + results = { + label: self[column].agg(reducer, *args) + for label, (column, reducer) in kwargs.items() + } + elif isinstance(func, dict): + results = { + column: self[column].agg(reducer, *args, **kwargs) + for column, reducer in func.items() + } + else: + columns = ( + self.numeric_columns + if numeric_only or isinstance(func, str) + else self.columns + ) + results = { + column: self[column].agg(func, *args, **kwargs) for column in columns + } + if any(isinstance(value, pd.DataFrame) for value in results.values()): + # pandas uses a second level for all columns when any reducer is a list. + tables = { + key: value + if isinstance(value, pd.DataFrame) + else value.to_frame(func[key] if isinstance(func, dict) else func) + for key, value in results.items() + } + result = pd.concat(tables, axis=1) + else: + result = pd.DataFrame(results) + return result if self.as_index else result.reset_index() + + agg = aggregate diff --git a/microdf/microseries.py b/microdf/microseries.py index 36ece1da..bcce88a2 100644 --- a/microdf/microseries.py +++ b/microdf/microseries.py @@ -1,7 +1,8 @@ +import inspect import logging import warnings from functools import wraps -from typing import Callable, List, Optional, Union +from typing import Callable, Optional, Union import numpy as np import pandas as pd @@ -11,6 +12,7 @@ aligned_weights, finalize_weights, weight_series, + require_equal_weights, ) logger = logging.getLogger(__name__) @@ -190,11 +192,24 @@ def __init__(self, *args, weights: np.ndarray = None, **kwargs): :type weights: np.ndarray """ super().__init__(*args, **kwargs) + source = args[0] if args else kwargs.get("data") + if weights is None and isinstance(source, MicroSeries): + weights = aligned_weights(source, self.index) self.set_weights(weights) @property def _constructor(self): - return MicroSeries + def construct(*args, **kwargs): + result = MicroSeries(*args, **kwargs) + # pandas.cut/qcut/to_numeric reconstruct from arrays with an + # explicit row index, without calling __finalize__ afterwards. + if result.index.equals(self.index) and ( + "index" in kwargs or (len(args) > 1 and args[1] is self.index) + ): + result.weights = weight_series(self.weights, result.index) + return result + + return construct @property def _constructor_expanddim(self): @@ -228,6 +243,9 @@ def __array_ufunc__(self, ufunc, method, *inputs, **kwargs): else: return NotImplemented + for value in inputs: + if value is not self: + require_equal_weights(self, value) out = kwargs.get("out") has_output = out is not None and any(value is not None for value in out) if ( @@ -266,6 +284,123 @@ def restore_weights(value): return restore_weights(result) return super().__array_ufunc__(ufunc, method, *inputs, **kwargs) + def __array_function__(self, func, types, args, kwargs): + """Dispatch NumPy reductions without discarding observation weights.""" + reductions = { + np.mean: "mean", + np.median: "median", + np.sum: "sum", + np.var: "var", + np.std: "std", + } + if func is np.average: + options = inspect.signature(func).bind(*args, **kwargs).arguments + if options.get("axis") not in (None, 0): + raise ValueError("MicroSeries has only axis 0") + if options.get("keepdims", False): + raise NotImplementedError("np.average keepdims is not supported") + weights = options.get("weights") + if weights is not None and not np.array_equal(weights, self.weights): + raise ValueError( + "np.average weights must match the MicroSeries weights" + ) + value = self.mean(skipna=False) + return ( + (value, self.weights.sum()) if options.get("returned", False) else value + ) + if func in reductions: + options = inspect.signature(func).bind(*args, **kwargs).arguments + options.pop("a", None) + axis = options.pop("axis", None) + if axis not in (None, 0): + raise ValueError("MicroSeries has only axis 0") + for name in ( + "out", + "overwrite_input", + "keepdims", + "dtype", + "where", + "initial", + "mean", + "correction", + ): + if name in options: + value = options.pop(name) + if value is not None and value is not False: + raise NotImplementedError( + f"{func.__name__} {name} is not supported on weighted data" + ) + if func in (np.var, np.std): + options.setdefault("ddof", 0) + return getattr(self, reductions[func])(skipna=False, **options) + # Shape queries and equality checks do not estimate a statistic. + if func is np.putmask and not isinstance(args[0], MicroSeries): + # pandas' ufunc out= machinery writes weighted values into a caller's + # explicitly supplied ndarray. No weighted object is reconstructed. + return np.putmask( + *(np.asarray(a) if isinstance(a, MicroSeries) else a for a in args), + **kwargs, + ) + if func in ( + np.shape, + np.ndim, + np.size, + np.array_equal, + np.allclose, + np.isclose, + ): + plain = tuple( + np.asarray(a) if isinstance(a, MicroSeries) else a for a in args + ) + return func(*plain, **kwargs) + raise NotImplementedError( + f"{func.__name__} is not supported on weighted data; use pd.Series(s) " + "for an explicitly unweighted operation." + ) + + def explode(self, ignore_index: bool = False) -> "MicroSeries": + """Expand list entries, repeating each observation's weight.""" + plain = pd.Series(self).reset_index(drop=True).explode() + positions = np.asarray(plain.index, dtype=int) + plain.index = ( + pd.RangeIndex(len(plain)) if ignore_index else self.index.take(positions) + ) + return self._weighted_result(plain, self.weights.iloc[positions]) + + def value_counts( + self, normalize=False, sort=True, ascending=False, bins=None, dropna=True + ) -> pd.Series: + """Sum weights by value; optionally divide by the included weight + total.""" + if bins is not None: + raise NotImplementedError( + "Weighted value_counts bins are unsupported; use pd.Series(s)" + ) + values = pd.Series(self, copy=False) + weights = np.asarray(self.weights) + _validate_frequency_weights(weights) + result = ( + pd.Series(weights) + .groupby( + values.reset_index(drop=True), dropna=dropna, observed=False, sort=False + ) + .sum() + ) + result.index.name = self.name + result.name = "proportion" if normalize else "count" + if normalize: + result = result / result.sum() + if sort: + result = result.sort_values(ascending=ascending, kind="stable") + return result + + def mode(self, dropna: bool = True) -> pd.Series: + """Return values with the greatest positive total weight as a plain + Series.""" + counts = self.value_counts(sort=False, dropna=dropna) + modes = counts.index[(counts == counts.max()) & (counts > 0)] + return pd.Series(modes, name=self.name).sort_values(ignore_index=True) + def __rdivmod__(self, other) -> tuple["MicroSeries", "MicroSeries"]: # An explicit override gives the weighted subclass priority over a # plain Series on the left, as for the other reverse operators. @@ -1191,6 +1326,36 @@ def __repr__(self) -> str: class MicroSeriesGroupBy(pd.core.groupby.generic.SeriesGroupBy): + def aggregate(self, func=None, *args, **kwargs): + """Apply named or callable estimators to weighted groups.""" + if isinstance(func, str): + if func in MicroSeries.FUNCTIONS: + return getattr(self, func)(*args, **kwargs) + # Unsupported inherited reductions must never reach pandas kernels. + func_name = func + func = lambda series, *a, **k: getattr(series, func_name)(*a, **k) + if isinstance(func, (list, tuple)): + return pd.concat( + [self.aggregate(f, *args, **kwargs) for f in func], + axis=1, + keys=[f if isinstance(f, str) else f.__name__ for f in func], + ) + if not callable(func): + raise TypeError("Weighted SeriesGroupBy.agg requires a name or callable") + grouper = self._grouper if hasattr(self, "_grouper") else self.grouper + positions = pd.Series(np.arange(len(self.obj)), index=self.obj.index).groupby( + grouper + ) + + def apply_group(rows): + ids = np.asarray(rows, dtype=int) + series = MicroSeries(self.obj.iloc[ids], weights=self.weights.obj.iloc[ids]) + return func(series, *args, **kwargs) + + return positions.agg(apply_group) + + agg = aggregate + def _init(self): def _weighted_agg(name) -> Callable: def via_micro_series(row, *args, **kwargs): @@ -1200,6 +1365,11 @@ def via_micro_series(row, *args, **kwargs): @wraps(fn) def _weighted_agg_fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: + kwargs.pop("numeric_only", None) + if name in MicroSeries.SCALAR_FUNCTIONS: + return self.aggregate( + lambda series: getattr(series, name)(*args, **kwargs) + ) arrays = self.apply(np.array) weights = self.weights.apply(np.array) df = pd.DataFrame(dict(a=arrays, w=weights)) @@ -1260,3 +1430,25 @@ def _weighted_agg_fn(*args, **kwargs) -> Union[pd.Series, pd.DataFrame]: for fn_name in MicroSeries.FUNCTIONS: setattr(self, fn_name, _weighted_agg(fn_name)) + for name in ( + "sem", + "skew", + "kurt", + "kurtosis", + "prod", + "product", + "idxmax", + "idxmin", + "rolling", + "expanding", + "ewm", + "value_counts", + "mode", + ): + + def reject(*args, _name=name, **kwargs): + raise NotImplementedError( + f"Weighted GroupBy.{_name} is unsupported; use pd.Series(s).groupby(...)" + ) + + setattr(self, name, reject) diff --git a/microdf/tests/test_binary_weight_alignment.py b/microdf/tests/test_binary_weight_alignment.py index 15b16696..2e6018fd 100644 --- a/microdf/tests/test_binary_weight_alignment.py +++ b/microdf/tests/test_binary_weight_alignment.py @@ -61,7 +61,10 @@ def test_binary_operators_align_weights_with_labels(method, weighted_other): def test_named_binary_methods_use_calling_series_weights(method, permuted): source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9], name="x") other = MicroSeries( - [2, 1], index=["a", "b"] if permuted else ["b", "a"], weights=[5, 7], name="x" + [2, 1], + index=["a", "b"] if permuted else ["b", "a"], + weights=[9, 1] if permuted else [1, 9], + name="x", ) expected = getattr(pd.Series(source), method)(pd.Series(other)) @@ -77,7 +80,7 @@ def test_named_binary_methods_use_calling_series_weights(method, permuted): @pytest.mark.parametrize("method", [f"__{op}__" for op in COMPARISONS]) def test_comparison_operators_keep_calling_series_weights_and_pandas_errors(method): source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9]) - other = MicroSeries([20, 10], index=source.index, weights=[5, 7]) + other = MicroSeries([20, 10], index=source.index, weights=[1, 9]) expected = getattr(pd.Series(source), method)(pd.Series(other)) assert_weighted_result(getattr(source, method)(other), expected, source) @@ -100,7 +103,7 @@ def test_scalar_and_array_binary_operands_keep_weights(method, operand): @pytest.mark.parametrize("method", ["__add__", "__rsub__", "add", "rsub", "lt"]) def test_matching_duplicate_indexes_keep_positional_weights(method): source = MicroSeries([10, 20, 30], index=["a", "a", "b"], weights=[1, 9, 3]) - other = MicroSeries([2, 1, 4], index=source.index, weights=[5, 7, 11]) + other = MicroSeries([2, 1, 4], index=source.index, weights=[1, 9, 3]) expected = getattr(pd.Series(source), method)(pd.Series(other)) assert_weighted_result(getattr(source, method)(other), expected, source) @@ -108,7 +111,7 @@ def test_matching_duplicate_indexes_keep_positional_weights(method): @pytest.mark.parametrize("method", ["__divmod__", "__rdivmod__", "divmod", "rdivmod"]) def test_divmod_results_keep_calling_series_weights(method): source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9]) - other = MicroSeries([3, 4], index=["a", "b"], weights=[5, 7]) + other = MicroSeries([3, 4], index=["a", "b"], weights=[9, 1]) expected = getattr(pd.Series(source), method)(pd.Series(other)) result = getattr(source, method)(other) assert isinstance(result, tuple) @@ -193,3 +196,17 @@ def test_plain_series_left_expressions_preserve_weighted_dispatch(operation, ind np.testing.assert_array_equal(source.weights, [1, 9]) source.weights.iloc[1] = 200 assert result.weights.iloc[0] == 100 + + +@pytest.mark.parametrize( + "method", + ARITHMETIC + + [f"r{op}" for op in ARITHMETIC] + + COMPARISONS + + ["div", "rdiv", "divmod", "rdivmod"], +) +def test_named_methods_reject_conflicting_weights(method): + source = MicroSeries([10, 20], index=["b", "a"], weights=[1, 9]) + other = MicroSeries([2, 1], index=["a", "b"], weights=[5, 7]) + with pytest.raises(ValueError, match="weights"): + getattr(source, method)(other) diff --git a/microdf/tests/test_fail_closed.py b/microdf/tests/test_fail_closed.py new file mode 100644 index 00000000..14456eee --- /dev/null +++ b/microdf/tests/test_fail_closed.py @@ -0,0 +1,381 @@ +"""Regression cases for issue #333, also used to build the support matrix.""" + +import operator +from pathlib import Path + +import numpy as np +import pandas as pd +import pytest + +import microdf as mdf + + +def survey(): + return mdf.MicroDataFrame({"x": [10.0, 100.0], "g": ["a", "a"]}, weights=[9, 1]) + + +# Each supported spelling has a numerical expectation, not only a type check. +WEIGHTED_CASES = [ + ("groupby column mean", lambda d: d.groupby("g").x.mean().iloc[0], 19), + ("groupby dict agg", lambda d: d.groupby("g").agg({"x": "mean"}).iloc[0, 0], 19), + ("groupby named agg", lambda d: d.groupby("g").agg(m=("x", "mean")).iloc[0, 0], 19), + ( + "groupby callable agg", + lambda d: d.groupby("g").agg({"x": lambda s: s.mean()}).iloc[0, 0], + 19, + ), + ( + "SeriesGroupBy callable agg", + lambda d: d.groupby("g").x.agg(lambda s: s.mean()).iloc[0], + 19, + ), + ( + "callable pivot_table", + lambda d: d.pivot_table(index="g", values="x", aggfunc=lambda s: s.mean()).iloc[ + 0, 0 + ], + 19, + ), + ("row apply", lambda d: d.apply(lambda r: r.x, axis=1).mean(), 19), + ("frame numeric mean", lambda d: d.mean(numeric_only=True)["x"], 19), + ("frame numeric median", lambda d: d.median(numeric_only=True)["x"], 10), + ( + "groupby numeric mean", + lambda d: d.groupby("g").mean(numeric_only=True).loc["a", "x"], + 19, + ), + ( + "groupby numeric median", + lambda d: d.groupby("g").median(numeric_only=True).loc["a", "x"], + 10, + ), + ("frame constructor", lambda d: mdf.MicroDataFrame(d).x.mean(), 19), + ("series constructor", lambda d: mdf.MicroSeries(d.x).mean(), 19), + ("pd.cut preserves weights", lambda d: pd.cut(d.x, 2, labels=False).sum(), 1), + ("pd.qcut preserves weights", lambda d: pd.qcut(d.x, 2, labels=False).sum(), 1), + ( + "pd.to_numeric preserves weights", + lambda d: pd.to_numeric(d.x.astype(str)).mean(), + 19, + ), + ( + "series explode", + lambda d: ( + mdf.MicroSeries([[10.0, 10.0], [100.0]], weights=d.weights).explode().sum() + ), + 280, + ), + ("np.average", lambda d: np.average(d.x), 19), + ("np.mean", lambda d: np.mean(d.x), 19), + ("np.median", lambda d: np.median(d.x), 10), + ("weighted value_counts", lambda d: d.x.value_counts().loc[10.0], 9), + ("weighted mode", lambda d: d.x.mode().iloc[0], 10), + ("numeric frame dropna", lambda d: d[["x"]].dropna().x.mean(), 19), +] + + +@pytest.mark.parametrize( + "label,operation,expected", WEIGHTED_CASES, ids=[c[0] for c in WEIGHTED_CASES] +) +def test_supported_operations(label, operation, expected): + assert operation(survey()) == pytest.approx(expected) + + +REJECTED = [ + "rolling", + "expanding", + "ewm", + "sem", + "skew", + "kurt", + "kurtosis", + "prod", + "product", + "idxmax", + "idxmin", +] + + +@pytest.mark.parametrize("method", REJECTED) +def test_unsupported_series_operations_name_plain_escape(method): + args = (2,) if method in {"rolling", "ewm"} else () + with pytest.raises(NotImplementedError, match=r"pd\.Series\(s\)"): + getattr(survey().x, method)(*args) + + +@pytest.mark.parametrize( + "operation", + [ + operator.add, + operator.sub, + operator.mul, + operator.truediv, + np.add, + np.maximum, + lambda a, b: a.add(b), + lambda a, b: a.rsub(b), + ], +) +@pytest.mark.parametrize("kind", [mdf.MicroSeries, mdf.MicroDataFrame]) +def test_conflicting_weights_raise_in_both_orders(operation, kind): + a = kind([10.0, 100.0], weights=[9, 1]) + b = kind([1.0, 2.0], weights=[1, 9]) + for left, right in [(a, b), (b, a)]: + with pytest.raises(ValueError, match="weights"): + operation(left, right) + + +@pytest.mark.parametrize("kind", [mdf.MicroSeries, mdf.MicroDataFrame]) +def test_constructor_aligns_and_copies_source_weights(kind): + source = kind([10.0, 100.0], index=["b", "a"], weights=[9, 1]) + result = kind(data=source, index=["a", "b"]) + assert result.weights.tolist() == [1, 9] + result.weights.iloc[0] = 100 + assert source.weights.tolist() == [9, 1] + assert kind(source, weights=[2, 3]).weights.tolist() == [2, 3] + + +@pytest.mark.parametrize( + "operation", [lambda d: d.dropna(), lambda d: d.dropna(ignore_index=True)] +) +def test_dropna_tracks_positions_with_duplicate_labels(operation): + d = mdf.MicroDataFrame( + {"x": [10.0, np.nan, 100.0]}, index=[4, 4, 4], weights=[9, 5, 1] + ) + result = operation(d) + assert result.weights.tolist() == [9, 1] + assert result.x.mean() == 19 + + +def test_groupby_multiple_functions_and_named_callables(): + d = survey() + result = d.groupby("g").agg({"x": ["mean", "sum"]}) + assert result.loc["a", ("x", "mean")] == 19 + assert result.loc["a", ("x", "sum")] == 190 + named = d.groupby("g").agg(m=("x", lambda s: s.mean()), total=("x", "sum")) + assert named.loc["a"].tolist() == [19, 190] + assert d.groupby("g").x.aggregate(["mean", "sum"]).loc["a"].tolist() == [19, 190] + + +def test_callable_groups_keep_duplicate_row_weights(): + d = mdf.MicroDataFrame( + {"x": [10.0, 100.0, 30.0], "g": ["a", "a", "b"]}, + index=[2, 2, 2], + weights=[9, 1, 4], + ) + result = d.groupby("g").agg(m=("x", lambda s: s.mean())) + assert result.m.tolist() == [19, 30] + + +def test_weighted_value_counts_and_mode(): + s = mdf.MicroSeries([1, 2, 2, np.nan, 3], weights=[9, 1, 1, 3, 0]) + assert s.value_counts().loc[1] == 9 + assert s.value_counts(normalize=True).loc[2] == pytest.approx(2 / 11) + assert s.value_counts(dropna=False).loc[np.nan] == 3 + assert s.mode().tolist() == [1] + assert type(s.value_counts()) is pd.Series + assert type(s.mode()) is pd.Series + + +def test_numpy_options_cannot_silently_override_weights(): + s = survey().x + assert np.average(s, returned=True) == (19, 10) + with pytest.raises((ValueError, NotImplementedError), match="weights"): + np.average(s, weights=[1, 9]) + with pytest.raises(NotImplementedError): + np.median(s, overwrite_input=True) + + +def support_matrix(): + rows = [ + "# Supported operations", + "", + "Generated from `microdf/tests/test_fail_closed.py` by `uv run python docs/build_support.py`.", + "", + "The regression suite checks these contracts on pandas 2 and 3. Aggregated results carry no row weights; transformations retain independent copies of the input weights.", + "", + "| Operation | Behaviour |", + "|---|---|", + ] + rows += [ + f"| {label} | Weighted result / preserved weights |" + for label, _, _ in WEIGHTED_CASES + ] + rows += [ + f"| `{name}` | Raises; use `pd.Series(s)` for unweighted pandas behaviour |" + for name in REJECTED + ] + rows += [ + "| Arithmetic with conflicting weights | Raises `ValueError` in either operand order |", + "", + "`pd.cut` and `pd.qcut` retain row weights, but choose bin edges using pandas' unweighted rules. Supply explicit bin edges for weighted quantile bins.", + "", + ] + rows += [ + "`pivot_table` grouping keys must name columns; external Series, callable", + "groupers and index-level groupers raise. Weighted `Series.value_counts` and", + "`Series.mode` return plain summary Series; frame and grouped variants raise.", + "", + "`microdf.concat` rejects mixed weighted/plain inputs in either order. Direct", + "`pd.concat` still bypasses microdf when its first input is plain pandas; the", + "regression suite records this upstream dispatch limitation as an expected failure.", + "Use Micro objects for every input to `pd.concat`, or use `microdf.concat`.", + "", + ] + return "\n".join(rows) + + +def test_support_matrix_is_current(): + path = Path(__file__).resolve().parents[2] / "docs" / "support.md" + if not path.exists(): + pytest.skip("support page not installed") + assert path.read_text() == support_matrix() + + +@pytest.mark.parametrize( + "convert", [lambda s: pd.cut(s, 2), lambda s: pd.qcut(s, 2), pd.to_numeric] +) +def test_pandas_conversions_preserve_independent_weights(convert): + source = mdf.MicroSeries([10.0, 100.0], index=[5, 5], weights=[9, 1], name="x") + result = convert(source) + assert isinstance(result, mdf.MicroSeries) + pd.testing.assert_series_equal(result.weights, source.weights) + result.weights.iloc[0] = 25 + assert source.weights.iloc[0] == 9 + + +@pytest.mark.parametrize("weighted_first", [False, True]) +@pytest.mark.parametrize("mapping", [False, True]) +def test_checked_concat_rejects_plain_inputs_in_any_order(weighted_first, mapping): + inputs = [survey(), pd.DataFrame({"x": [1.0]})] + if not weighted_first: + inputs.reverse() + if mapping: + inputs = dict(zip(["a", "b"], inputs)) + with pytest.raises(ValueError, match="weights"): + mdf.concat(inputs) + + +def test_checked_concat_keeps_weighted_rows(): + result = mdf.concat([survey(), survey()], ignore_index=True) + assert result.weights.tolist() == [9, 1, 9, 1] + assert result.x.mean() == 19 + + +@pytest.mark.xfail( + strict=True, + reason="pandas chooses the plain first input's constructor without a subclass dispatch hook; use microdf.concat", +) +def test_plain_first_pandas_concat_dispatch_boundary(): + with pytest.raises(ValueError, match="weights"): + pd.concat([pd.DataFrame({"x": [1.0]}), survey()]) + + +@pytest.mark.parametrize("method", ["sem", "skew", "prod", "idxmax", "idxmin"]) +@pytest.mark.parametrize( + "selection", [lambda d: d.groupby("g"), lambda d: d.groupby("g").x] +) +def test_groupby_unsupported_reductions_raise(method, selection): + with pytest.raises(NotImplementedError, match=r"pd\."): + getattr(selection(survey()), method)() + + +@pytest.mark.parametrize("as_index", [True, False]) +@pytest.mark.parametrize("select", [lambda g: g, lambda g: g[["x"]]]) +def test_groupby_selection_numeric_only_and_result_index(as_index, select): + grouped = select(survey().groupby("g", as_index=as_index)) + for result in [grouped.mean(numeric_only=True), grouped.agg({"x": "mean"})]: + assert result.x.tolist() == [19] + if as_index: + assert result.index.name == "g" + else: + assert result.g.tolist() == ["a"] + + +@pytest.mark.parametrize("kind", [mdf.MicroSeries, mdf.MicroDataFrame]) +def test_equal_weight_arithmetic_retains_weighted_answer(kind): + left = kind([10.0, 100.0], weights=[9, 1]) + right = kind([1.0, 2.0], weights=[9, 1]) + result = left + right + assert result.weights.tolist() == [9, 1] + value = result.mean() + if isinstance(value, pd.Series): + value = value.iloc[0] + assert value == pytest.approx(20.1) + + +def test_covariance_documentation_example(): + pair = mdf.MicroDataFrame( + {"x": [1.0, 3.0, 5.0], "y": [2.0, 5.0, 4.0]}, weights=[1, 2, 1] + ) + expected = np.array([[8 / 3, 4 / 3], [4 / 3, 2]]) + np.testing.assert_allclose(pair.cov(), expected) + np.testing.assert_allclose( + np.cov([[1.0, 3.0, 5.0], [2.0, 5.0, 4.0]], fweights=[1, 2, 1]), expected + ) + replicated = pd.DataFrame(pair).iloc[[0, 1, 1, 2]] + np.testing.assert_allclose(pair.corr(), replicated.corr()) + assert type(pair.cov()) is pd.DataFrame + assert type(pair.corr()) is pd.DataFrame + + +@pytest.mark.parametrize( + "operation", [np.maximum, np.minimum, np.add, lambda a, b: a + b] +) +def test_frame_arithmetic_with_equal_weights_keeps_weights(operation): + left = survey()[["x"]] + right = mdf.MicroDataFrame({"x": [12.0, 80.0]}, weights=[9, 1]) + result = operation(left, right) + expected = operation(pd.DataFrame(left), pd.DataFrame(right)) + pd.testing.assert_frame_equal(pd.DataFrame(result), expected) + assert result.weights.tolist() == [9, 1] + assert result.x.mean() == pytest.approx((expected.x * [9, 1]).sum() / 10) + + +def test_numeric_group_keys_are_excluded_from_reductions(): + d = mdf.MicroDataFrame({"x": [10.0, 100.0], "g": [1, 1]}, weights=[9, 1]) + result = d.groupby("g").mean(numeric_only=True) + assert list(result.columns) == ["x"] + assert result.x.tolist() == [19] + + +@pytest.mark.parametrize("method", REJECTED + ["mode", "value_counts"]) +def test_frame_unsupported_reductions_name_plain_escape(method): + args = (2,) if method in {"rolling", "ewm"} else () + with pytest.raises(NotImplementedError, match=r"pd\.DataFrame\(df\)"): + getattr(survey(), method)(*args) + + +def test_named_aggregation_accepts_keyword_like_output_names(): + result = survey().groupby("g").agg(numeric_only=("x", "mean"), engine=("x", "sum")) + assert result.loc["a"].tolist() == [19, 190] + + +def test_grouped_callables_with_missing_group_keys_and_multiple_keys(): + d = mdf.MicroDataFrame( + {"x": [10.0, 100.0, 30.0], "g": ["a", "a", None], "h": [1, 1, 2]}, + weights=[9, 1, 4], + ) + result = d.groupby(["g", "h"], dropna=False).agg(m=("x", lambda s: s.mean())) + assert result.m.tolist() == [19, 30] + + +def test_dropna_inplace_and_columns(): + d = mdf.MicroDataFrame({"x": [10.0, 100.0], "y": [None, 2.0]}, weights=[9, 1]) + assert d.dropna(axis=1).x.mean() == 19 + assert d.dropna(inplace=True) is None + assert d.weights.tolist() == [1] + assert d.x.mean() == 100 + + +def test_row_apply_expansion_retains_weights(): + result = survey().apply(lambda r: [r.x, 2 * r.x], axis=1, result_type="expand") + assert isinstance(result, mdf.MicroDataFrame) + assert result.mean().tolist() == [19, 38] + + +@pytest.mark.parametrize("keys", [pd.Series(["a", "a"], index=[5, 6]), lambda row: row]) +def test_pivot_rejects_groupers_without_positional_provenance(keys): + d = mdf.MicroDataFrame({"x": [10.0, 100.0]}, index=[5, 6], weights=[9, 1]) + with pytest.raises(NotImplementedError, match="grouping keys must name columns"): + d.pivot_table(values="x", index=keys, aggfunc=lambda s: s.mean()) diff --git a/microdf/tests/test_poverty.py b/microdf/tests/test_poverty.py new file mode 100644 index 00000000..43e0edac --- /dev/null +++ b/microdf/tests/test_poverty.py @@ -0,0 +1,49 @@ +"""Hand-calculated poverty estimators for issue #334.""" + +import pytest +import microdf as mdf + + +@pytest.mark.parametrize( + "method,expected", + [ + ("poverty_rate", 2 / 5), + ("deep_poverty_rate", 2 / 5), + ("poverty_gap", 160), + ("deep_poverty_gap", 60), + ("squared_poverty_gap", 12800), + ], +) +def test_poverty_estimators(method, expected): + # Only the first row contributes: 2 people at income 20, threshold 100. + # The 3 people exactly at threshold are not poor; income 0 has weight 0. + d = mdf.MicroDataFrame( + {"income": [20.0, 100.0, 0.0], "threshold": [100.0, 100.0, 100.0]}, + weights=[2, 3, 0], + ) + assert getattr(d, method)("income", "threshold") == pytest.approx(expected) + assert getattr(d, method)(d.income, d.threshold) == pytest.approx(expected) + + +@pytest.mark.parametrize( + "income,rate,deep_rate,gap,deep_gap,squared", + [ + ([50.0, 100.0, 0.0], 2 / 5, 0, 100, 0, 5000), + ([100.0, 200.0, 0.0], 0, 0, 0, 0, 0), + ], +) +def test_poverty_threshold_boundaries(income, rate, deep_rate, gap, deep_gap, squared): + d = mdf.MicroDataFrame( + {"income": income, "threshold": [100.0] * 3}, weights=[2, 3, 0] + ) + for method, expected in zip( + [ + "poverty_rate", + "deep_poverty_rate", + "poverty_gap", + "deep_poverty_gap", + "squared_poverty_gap", + ], + [rate, deep_rate, gap, deep_gap, squared], + ): + assert getattr(d, method)("income", "threshold") == pytest.approx(expected)