This is an automated email from the ASF dual-hosted git repository.
rusackas pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/superset.git
The following commit(s) were added to refs/heads/master by this push:
new ea79c63e878 fix(post-processing): stop treating gaps as zero for
cumprod, cummin and cummax (#44828)
ea79c63e878 is described below
commit ea79c63e8786e6dad7cc53b12c0cfd0c0fb6aa36
Author: Sepuri Sai Krishna <[email protected]>
AuthorDate: Sat Oct 3 10:52:50 2026 +0530
fix(post-processing): stop treating gaps as zero for cumprod, cummin and
cummax (#44828)
---
superset/utils/pandas_postprocessing/cum.py | 7 +++--
tests/unit_tests/pandas_postprocessing/test_cum.py | 35 ++++++++++++++++++++++
2 files changed, 40 insertions(+), 2 deletions(-)
diff --git a/superset/utils/pandas_postprocessing/cum.py
b/superset/utils/pandas_postprocessing/cum.py
index d3eb969f79f..e0224124384 100644
--- a/superset/utils/pandas_postprocessing/cum.py
+++ b/superset/utils/pandas_postprocessing/cum.py
@@ -46,7 +46,6 @@ def cum(
"""
columns = columns or {}
df_cum = df.loc[:, columns.keys()]
- df_cum = df_cum.fillna(0)
operation = "cum" + operator
if operation not in ALLOWLIST_CUMULATIVE_FUNCTIONS or not hasattr(
df_cum, operation
@@ -54,5 +53,9 @@ def cum(
raise InvalidPostProcessingError(
_("Invalid cumulative operator: %(operator)s", operator=operator)
)
- df_cum = _append_columns(df, getattr(df_cum, operation)(), columns)
+ # Cumulate first, then carry the last cumulative value across gaps. Filling
+ # the gaps with 0 beforehand would be correct only for ``sum``, where 0 is
+ # the additive identity: it zeroes the rest of a ``prod`` series and makes 0
+ # the running ``min``/``max``, a value that need not appear in the data.
+ df_cum = _append_columns(df, getattr(df_cum, operation)().ffill(), columns)
return df_cum
diff --git a/tests/unit_tests/pandas_postprocessing/test_cum.py
b/tests/unit_tests/pandas_postprocessing/test_cum.py
index 25d7fd045f5..1d39ef7ac4c 100644
--- a/tests/unit_tests/pandas_postprocessing/test_cum.py
+++ b/tests/unit_tests/pandas_postprocessing/test_cum.py
@@ -91,6 +91,41 @@ def test_cum_with_gap():
assert series_to_list(post_df["y2"]) == [1.0, 3.0, 3.0, 7.0]
+def test_cum_with_gap_non_additive_operators():
+ """A gap must not be treated as 0 for operators where 0 is not the
identity.
+
+ ``fillna(0)`` is only correct for ``sum``. For ``prod`` a single gap zeroes
+ the rest of the series, and for ``min``/``max`` it makes 0 the running
+ extreme even though 0 never appears in the data. The last cumulative value
+ is carried across the gap instead.
+ """
+ gapped = pd.DataFrame({"y": [2.0, None, 3.0, 4.0]})
+ assert series_to_list(
+ pp.cum(df=gapped, columns={"y": "y"}, operator="prod")["y"]
+ ) == [2.0, 2.0, 6.0, 24.0]
+
+ gapped = pd.DataFrame({"y": [5.0, None, 3.0, 7.0]})
+ assert series_to_list(
+ pp.cum(df=gapped, columns={"y": "y"}, operator="min")["y"]
+ ) == [5.0, 5.0, 3.0, 3.0]
+
+ gapped = pd.DataFrame({"y": [-5.0, None, -3.0, -9.0]})
+ assert series_to_list(
+ pp.cum(df=gapped, columns={"y": "y"}, operator="max")["y"]
+ ) == [-5.0, -5.0, -3.0, -3.0]
+
+
+def test_cum_with_leading_gap():
+ """A leading gap has no cumulative value to carry, so it stays empty.
+
+ Reporting 0 there would put a data point on the chart before the series
+ has any data.
+ """
+ leading = pd.DataFrame({"y": [None, 1.0, 2.0]})
+ post_df = pp.cum(df=leading, columns={"y": "y"}, operator="sum")
+ assert series_to_list(post_df["y"]) == [None, 1.0, 3.0]
+
+
def test_cum_after_pivot_with_single_metric():
pivot_df = pp.pivot(
df=single_metric_df,