aminghadersohi commented on code in PR #43771:
URL: https://github.com/apache/superset/pull/43771#discussion_r4204241814
##########
superset/common/form_data_query_context.py:
##########
@@ -288,71 +427,1311 @@ def _pie_contribution_post_processing(metrics:
list[Any]) -> list[dict[str, Any]
]
-def build_query_context_from_form_data(
- form_data: dict[str, Any],
- datasource: dict[str, Any],
- viz_type: str | None = None,
+def _as_list(value: Any) -> list[Any]:
+ """Return the frontend ``ensureIsArray`` representation of a value."""
+ if value is None:
+ return []
+ return list(value) if isinstance(value, (list, tuple)) else [value]
+
+
+def _label(value: Any, *, metric: bool = False) -> str:
+ """Resolve a frontend-compatible query-field label."""
+ try:
+ return get_metric_name(value) if metric else get_column_name(value)
+ except (AttributeError, KeyError, TypeError, ValueError):
+ if isinstance(value, Mapping):
+ return str(
+ value.get("label")
+ or value.get("column_name")
+ or value.get("sqlExpression")
+ or value
+ )
+ return str(value)
+
+
+def _deduplicate_fields(values: list[Any], *, metric: bool = False) ->
list[Any]:
+ """Deduplicate query fields by their frontend-visible label."""
+ result: list[Any] = []
+ labels: set[str] = set()
+ for value in values:
+ if value is None or value == "":
+ continue
+ label = _label(value, metric=metric)
+ if label in labels:
+ continue
+ labels.add(label)
+ result.append(value)
+ return result
+
+
+def retain_mixed_timeseries_secondary_form_data(
+ form_data: Mapping[str, Any],
) -> dict[str, Any]:
- """
- Build a query-context payload (the JSON shape
``ChartDataQueryContextSchema``
- loads) from a chart's form data and datasource reference.
+ """Mirror ``retainFormDataSuffix(formData, '_b')`` exactly.
- :param form_data: The chart's saved ``params`` parsed to a dict.
- :param datasource: ``{"id": <int>, "type": "table"}`` datasource reference.
- :param viz_type: The chart's viz type, used for viz-specific handling.
- :returns: A single-query query-context dict.
+ Suffixed values are installed first, including falsey values, and shared
+ unsuffixed controls fill only keys that query B did not explicitly set.
"""
- columns, metrics = _columns_and_metrics(form_data, viz_type)
-
- # SIMPLE adhoc filters (+ legacy top-level ``filters``) become query
filters;
- # free-form SQL predicates go into ``extras``. Only ``WHERE``-clause SIMPLE
- # filters are applied (matching the chart), so the export never filters on
a
- # ``HAVING`` clause the chart itself ignores.
- filters = adhoc_filters_to_query_filters(
- form_data.get("adhoc_filters", []), where_only=True
- )
- for flt in form_data.get("filters") or []:
- if isinstance(flt, dict) and flt.get("col") is not None:
- filters.append(flt)
+ secondary: dict[str, Any] = {}
+ for key, value in form_data.items():
+ if key.endswith("_b"):
+ secondary[key[:-2]] = value
+ for key, value in form_data.items():
+ if not key.endswith("_b") and key not in secondary:
+ secondary[key] = value
+ secondary_filter_keys = {
+ "adhoc_filters": "adhoc_filters_b",
+ "extra_filters": "extra_filters_b",
+ "filters": "filters_b",
+ "having": "having_b",
+ "where": "where_b",
+ }
+ if any(suffixed in form_data for suffixed in
secondary_filter_keys.values()):
+ # The frontend exposes adhoc_filters_b, while saved/server payloads can
+ # carry equivalent legacy aliases. Treat the family atomically: an
+ # explicit clear in any B alias must not be repopulated by query A's
+ # differently named filter representation.
+ for primary, suffixed in secondary_filter_keys.items():
+ if suffixed not in form_data:
+ secondary.pop(primary, None)
+ return secondary
- extras = freeform_where_having(form_data)
- if form_data.get("time_grain_sqla"):
- extras["time_grain_sqla"] = form_data["time_grain_sqla"]
- # Prefer the modern ``time_range``; fall back to the legacy
``since``/``until``
- # pair (older charts store the range that way) before defaulting to no
filter.
- time_range = form_data.get("time_range")
- if not time_range and (form_data.get("since") or form_data.get("until")):
- time_range = f"{form_data.get('since') or ''} :
{form_data.get('until') or ''}"
- time_range = time_range or "No filter"
+def _base_query_object( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ row_limit: int | None,
+ order_desc: bool | None,
+ filters_prepared: bool,
+) -> dict[str, Any]:
+ """Build the shared frontend-equivalent portion of a QueryObject."""
+ columns, metrics, orderby = query_fields_from_form_data(form_data)
query: dict[str, Any] = {
"columns": columns,
"metrics": metrics,
- "orderby": orderby_from_form_data(form_data, metrics, viz_type),
- "filters": filters,
- "time_range": time_range,
}
+ if orderby:
+ query["orderby"] = orderby
+
+ if filters_prepared:
+ query["filters"] = list(form_data.get("filters") or [])
+ for clause in ("where", "having"):
+ if form_data.get(clause):
+ query[clause] = form_data[clause]
+ if form_data.get("extras"):
+ query["extras"] = dict(form_data["extras"])
+ else:
+ filters = adhoc_filters_to_query_filters(
+ form_data.get("adhoc_filters", []), where_only=True
+ )
+ filters.extend(
+ filter_
+ for filter_ in form_data.get("filters") or []
+ if isinstance(filter_, dict) and filter_.get("col") is not None
+ )
+ query["filters"] = filters
+ if extras := freeform_where_having(form_data):
+ query["extras"] = extras
+
+ extras = dict(query.get("extras") or {})
+ if form_data.get("time_grain_sqla") is not None:
+ extras["time_grain_sqla"] = form_data["time_grain_sqla"]
if extras:
query["extras"] = extras
- if viz_type == "pie" and (
- post_processing := _pie_contribution_post_processing(metrics)
+
+ effective_limit = row_limit if row_limit is not None else
form_data.get("row_limit")
+ if effective_limit is not None:
+ query["row_limit"] = effective_limit
+ if form_data.get("row_offset") is not None:
+ query["row_offset"] = form_data["row_offset"]
+ if order_desc is not None:
+ query["order_desc"] = order_desc
+ elif "order_desc" in form_data and form_data["order_desc"] is not None:
+ query["order_desc"] = form_data["order_desc"]
+
+ time_range = form_data.get("time_range")
+ if not time_range and (form_data.get("since") or form_data.get("until")):
+ time_range = f"{form_data.get('since') or ''} :
{form_data.get('until') or ''}"
+ if time_range:
+ query["time_range"] = time_range
+ for key in ("since", "until", "annotation_layers", "url_params",
"custom_params"):
+ if form_data.get(key) is not None:
+ query[key] = form_data[key]
+
+ granularity = form_data.get("granularity") or
form_data.get("granularity_sqla")
+ if granularity:
+ query["granularity"] = granularity
+ series_limit = form_data.get("series_limit", form_data.get("limit"))
+ if series_limit is not None:
+ query["series_limit"] = series_limit
+ series_limit_metric = form_data.get("series_limit_metric")
+ if series_limit_metric is None:
+ series_limit_metric = form_data.get("timeseries_limit_metric")
+ if series_limit_metric is not None:
+ query["series_limit_metric"] = series_limit_metric
+ if form_data.get("group_others_when_limit_reached") is not None:
+ query["group_others_when_limit_reached"] = form_data[
+ "group_others_when_limit_reached"
+ ]
+ return query
+
+
+def _temporalized_columns(form_data: dict[str, Any], columns: list[Any]) ->
list[Any]:
+ """Apply the pivot BASE_AXIS temporal-column contract."""
+ time_grain = form_data.get("time_grain_sqla")
+ temporal_lookup = form_data.get("temporal_columns_lookup") or {}
+ result: list[Any] = []
+ for column in columns:
+ if (
+ isinstance(column, str)
+ and time_grain
+ and (
+ temporal_lookup.get(column)
+ or form_data.get("granularity_sqla") == column
+ )
+ ):
+ result.append(
+ {
+ "timeGrain": time_grain,
+ "columnType": "BASE_AXIS",
+ "sqlExpression": column,
+ "label": column,
+ "expressionType": "SQL",
+ }
+ )
+ else:
+ result.append(column)
+ return result
+
+
+def _box_temporalized_columns(
+ form_data: dict[str, Any], columns: list[Any]
+) -> list[Any]:
+ """Convert only physical columns confirmed temporal by Box Plot
metadata."""
+ time_grain = form_data.get("time_grain_sqla")
+ temporal_lookup = form_data.get("temporal_columns_lookup")
+ if not time_grain or not isinstance(temporal_lookup, Mapping):
+ return columns
+ return [
+ {
+ "timeGrain": time_grain,
+ "columnType": "BASE_AXIS",
+ "sqlExpression": column,
+ "label": column,
+ "expressionType": "SQL",
+ }
+ if isinstance(column, str) and temporal_lookup.get(column) is True
+ else column
+ for column in columns
+ ]
+
+
+def _table_temporalized_columns(
+ form_data: dict[str, Any], columns: list[Any]
+) -> list[Any]:
+ """Promote the first temporal table group-by to the frontend BASE_AXIS.
+
+ Table's builder treats only physical columns named in
+ ``temporal_columns_lookup`` as temporal and moves the first match to the
+ front. Later temporal dimensions remain ordinary group-bys.
+ """
+ time_grain = form_data.get("time_grain_sqla")
+ temporal_lookup = form_data.get("temporal_columns_lookup") or {}
+ if not time_grain or not isinstance(temporal_lookup, Mapping):
+ return columns
+
+ temporal_column: dict[str, Any] | None = None
+ remaining: list[Any] = []
+ for column in columns:
+ if (
+ temporal_column is None
+ and isinstance(column, str)
+ and temporal_lookup.get(column)
+ ):
+ temporal_column = {
+ "timeGrain": time_grain,
+ "columnType": "BASE_AXIS",
+ "sqlExpression": column,
+ "label": column,
+ "expressionType": "SQL",
+ "isColumnReference": True,
+ }
+ else:
+ remaining.append(column)
+ return [temporal_column, *remaining] if temporal_column else columns
+
+
+def _histogram_query(form_data: dict[str, Any], query: dict[str, Any]) -> None:
+ groupby = _as_list(form_data.get("groupby"))
+ column = form_data.get("column")
+ query["columns"] = [*groupby, *([column] if column is not None else [])]
+ query["post_processing"] = [
+ {
+ "operation": "histogram",
+ "options": {
+ "column": _label(column),
+ "groupby": [_label(value) for value in groupby],
+ "bins": int(form_data.get("bins", 5)),
+ "cumulative": form_data.get("cumulative", False),
+ "normalize": form_data.get("normalize", False),
+ },
+ }
+ ]
+ if any(
+ isinstance(filter_, dict) and filter_.get("clause") == "HAVING"
+ for filter_ in form_data.get("adhoc_filters") or []
+ ):
+ query["metrics"] = [
+ {
+ "expressionType": "SQL",
+ "sqlExpression": "COUNT(*)",
+ "label": "COUNT(*)",
+ }
+ ]
+ else:
+ query["metrics"] = []
+
+
+def _box_plot_query(form_data: dict[str, Any], query: dict[str, Any]) -> None:
+ distributed = _as_list(form_data.get("columns"))
+ if not distributed and form_data.get("granularity_sqla"):
+ distributed = [form_data["granularity_sqla"]]
+ groupby = _as_list(form_data.get("groupby"))
+ query["columns"] = [*_box_temporalized_columns(form_data, distributed),
*groupby]
+ query["series_columns"] = groupby
+ whisker = form_data.get("whiskerOptions")
+ if not whisker:
+ query["post_processing"] = []
+ return
+ whisker_type = "tukey"
+ percentiles: list[int] | None = None
+ if whisker == "Min/max (no outliers)":
+ whisker_type = "min/max"
+ elif isinstance(whisker, str) and whisker.endswith(" percentiles"):
+ low, high = whisker.removesuffix(" percentiles").split("/", 1)
+ whisker_type = "percentile"
+ percentiles = [int(low), int(high)]
+ query["post_processing"] = [
+ {
+ "operation": "boxplot",
+ "options": {
+ "whisker_type": whisker_type,
+ "percentiles": percentiles,
+ "groupby": [_label(value) for value in groupby],
+ "metrics": [_label(value, metric=True) for value in
query["metrics"]],
+ },
+ }
+ ]
+
+
+_PIVOT_ADDITIVE_AGGREGATES = frozenset({"SUM", "COUNT", "MIN", "MAX"})
+
+
+def _all_metrics_additive(metrics: list[Any]) -> bool:
+ """Mirror Pivot's conservative additive-metric fast-path."""
+ return bool(metrics) and all(
+ isinstance(metric, Mapping)
+ and metric.get("expressionType") == "SIMPLE"
+ and metric.get("aggregate") in _PIVOT_ADDITIVE_AGGREGATES
+ for metric in metrics
+ )
+
+
+def _pivot_grouping_sets(
+ form_data: dict[str, Any], rows: list[Any], columns: list[Any]
+) -> list[list[str]]:
+ """Enumerate the rollup levels requested by Pivot's frontend builder."""
+ row_prefixes = [[], *(rows[: index + 1] for index in range(len(rows)))]
+ column_prefixes = [
+ [],
+ *(columns[: index + 1] for index in range(len(columns))),
+ ]
+ show_values_as = form_data.get("showValuesAs")
+ needs_rows_collapsed = show_values_as in {"percent_col", "percent_total"}
+ needs_columns_collapsed = show_values_as in {"percent_row",
"percent_total"}
+
+ def row_prefix_needed(prefix: list[Any]) -> bool:
+ if len(prefix) == len(rows):
+ return True
+ if not prefix:
+ return bool(form_data.get("colTotals")) or needs_rows_collapsed
+ return bool(form_data.get("rowSubTotals"))
+
+ def column_prefix_needed(prefix: list[Any]) -> bool:
+ if len(prefix) == len(columns):
+ return True
+ if not prefix:
+ return bool(form_data.get("rowTotals")) or needs_columns_collapsed
+ return bool(form_data.get("colSubTotals"))
+
+ levels = [
+ (row_prefix, column_prefix)
+ for row_prefix in row_prefixes
+ if row_prefix_needed(row_prefix)
+ for column_prefix in column_prefixes
+ if column_prefix_needed(column_prefix)
+ ]
+ if form_data.get("combineMetric"):
+ metrics_layout = form_data.get("metricsLayout")
+
+ def forced_denominator(level: tuple[list[Any], list[Any]]) -> bool:
+ row_prefix, column_prefix = level
+ return (needs_rows_collapsed and not row_prefix) or (
+ needs_columns_collapsed and not column_prefix
+ )
+
+ if metrics_layout == "ROWS":
+ levels = [
+ level
+ for level in levels
+ if len(level[0]) == len(rows) or forced_denominator(level)
+ ]
+ else:
+ levels = [
+ level
+ for level in levels
+ if len(level[1]) == len(columns) or forced_denominator(level)
+ ]
+
+ return [
+ [_label(value) for value in _deduplicate_fields([*row_prefix,
*column_prefix])]
+ for row_prefix, column_prefix in levels
+ ]
+
+
+def _pivot_query(form_data: dict[str, Any], query: dict[str, Any]) -> None:
+ rows = _as_list(form_data.get("groupbyRows"))
+ columns = _as_list(form_data.get("groupbyColumns"))
+ if form_data.get("transposePivot"):
+ rows, columns = columns, rows
+ query["columns"] = _temporalized_columns(
+ form_data, _deduplicate_fields([*rows, *columns])
+ )
+ metric = query.get("series_limit_metric") or next(
+ iter(query.get("metrics") or []), None
+ )
+ query["orderby"] = (
+ [[metric, not bool(query.get("order_desc", True))]]
+ if metric is not None
+ else []
+ )
+ if not _all_metrics_additive(query.get("metrics") or []):
+ query["grouping_sets"] = _pivot_grouping_sets(form_data, rows, columns)
+
+
+def _waterfall_query(form_data: dict[str, Any], query: dict[str, Any]) -> None:
+ x_axis = form_data.get("x_axis") or form_data.get("granularity_sqla")
+ columns = [*_as_list(x_axis), *_as_list(form_data.get("groupby"))]
+ query["columns"] = _deduplicate_fields(columns)
+ query["orderby"] = [[column, True] for column in query["columns"]]
+
+
+def _gantt_query(form_data: dict[str, Any], query: dict[str, Any]) -> None:
+ groupby = _as_list(form_data.get("series"))
+ orderby = query_fields_from_form_data(form_data)[2]
+ columns = [
+ form_data.get("start_time"),
+ form_data.get("end_time"),
+ form_data.get("y_axis"),
+ *groupby,
+ *_as_list(form_data.get("tooltip_columns")),
+ *(entry[0] for entry in orderby if entry),
+ ]
+ query["columns"] = _deduplicate_fields(columns)
+ query["metrics"] = _as_list(form_data.get("tooltip_metrics"))
+ query["orderby"] = orderby
+ query["series_columns"] = groupby
+
+
+def _normalize_query_orderby(query: dict[str, Any]) -> None:
+ """Mirror ``normalizeOrderBy`` while retaining limit-direction controls."""
+ orderby = query.get("orderby")
+ if (
+ isinstance(orderby, list)
+ and orderby
+ and isinstance(orderby[0], (list, tuple))
+ and len(orderby[0]) == 2
+ and orderby[0][0]
+ and isinstance(orderby[0][1], bool)
+ ):
+ return
+ metric = (
+ query.get("series_limit_metric")
+ or query.get("legacy_order_by")
+ or next(iter(query.get("metrics") or []), None)
+ )
+ if metric is None:
+ query.pop("orderby", None)
+ return
+ query["orderby"] = [[metric, not bool(query.get("order_desc", True))]]
+
+
+_TIME_COMPARISON_TYPES = frozenset({"values", "difference", "percentage",
"ratio"})
+
+
+def _metric_offset_map(
+ form_data: dict[str, Any],
+ metric_labels: list[str],
+ offsets: list[Any] | None = None,
+) -> dict[str, str]:
+ """Return the frontend time-comparison metric label map."""
+ if form_data.get("comparison_type") not in _TIME_COMPARISON_TYPES:
+ return {}
+ return {
+ f"{metric}__{offset}": metric
+ for metric in metric_labels
+ for offset in (
+ offsets if offsets is not None else
_as_list(form_data.get("time_compare"))
+ )
+ }
+
+
+def _table_time_offsets(form_data: dict[str, Any]) -> list[Any]:
+ """Resolve Table custom/inherited shifts like its frontend query
adapter."""
+ raw_offsets = _as_list(form_data.get("time_compare"))
+ offsets = [offset for offset in raw_offsets if offset not in {"custom",
"inherit"}]
+ if "custom" in raw_offsets and form_data.get("start_date_offset") is not
None:
+ offsets.append(form_data["start_date_offset"])
+ extra_form_data = form_data.get("extra_form_data")
+ if isinstance(extra_form_data, Mapping) and
extra_form_data.get("time_compare"):
+ inherited = extra_form_data["time_compare"]
+ if inherited not in offsets:
+ offsets = [inherited]
+ return offsets
+
+
+def _x_axis_column(form_data: Mapping[str, Any]) -> Any | None:
+ """Return a supported x-axis column, excluding legacy granularity.
+
+ ``column_name`` mappings are retained for old server/native payloads. Big
+ Number uses the stricter frontend predicate below.
+ """
+ x_axis = form_data.get("x_axis")
+ if isinstance(x_axis, str):
+ return x_axis if x_axis else None
+ if isinstance(x_axis, Mapping):
+ if isinstance(column_name := x_axis.get("column_name"), str) and
column_name:
+ return column_name
+ # Frontend SQL adhoc columns remain objects in the QueryObject.
+ return x_axis if x_axis else None
+ return None
+
+
+def _frontend_x_axis_column(form_data: Mapping[str, Any]) -> Any | None:
+ """Mirror ``isQueryFormColumn`` for physical and SQL adhoc columns."""
+ x_axis = form_data.get("x_axis")
+ if isinstance(x_axis, str):
+ return x_axis if x_axis else None
+ if (
+ isinstance(x_axis, Mapping)
+ and "sqlExpression" in x_axis
+ and "label" in x_axis
+ and x_axis.get("expressionType") in {None, "SQL"}
+ ):
+ return x_axis
+ return None
+
+
+def normalize_time_column(
+ form_data: Mapping[str, Any], query: dict[str, Any]
+) -> dict[str, Any]:
+ """Apply the final shared frontend ``normalizeTimeColumn`` mutator."""
+ x_axis = _frontend_x_axis_column(form_data)
+ columns = query.get("columns")
+ if x_axis is None or not isinstance(columns, list):
+ return query
+
+ axis_index: int | None = None
+ for index, column in enumerate(columns):
+ if isinstance(x_axis, str) and isinstance(column, str) and column ==
x_axis:
+ axis_index = index
+ break
+ if (
+ isinstance(x_axis, Mapping)
+ and isinstance(column, Mapping)
+ and column.get("sqlExpression") == x_axis.get("sqlExpression")
+ ):
+ axis_index = index
+ break
+ if axis_index is None:
+ return query
+
+ normalized = dict(query)
+ normalized_columns = list(columns)
+ grain = (query.get("extras") or {}).get("time_grain_sqla")
+ if isinstance(columns[axis_index], Mapping):
+ normalized_axis = {
+ "columnType": "BASE_AXIS",
+ **({"timeGrain": grain} if grain is not None else {}),
+ **columns[axis_index],
+ }
+ else:
+ normalized_axis = {
+ "columnType": "BASE_AXIS",
+ "sqlExpression": x_axis,
+ "label": x_axis,
+ "expressionType": "SQL",
+ "isColumnReference": True,
+ **({"timeGrain": grain} if grain is not None else {}),
+ }
+ normalized_columns[axis_index] = normalized_axis
+ normalized["columns"] = normalized_columns
+ normalized.pop("is_timeseries", None)
+ return normalized
+
+
+def _finalize_query_objects(
+ form_data: Mapping[str, Any], queries: list[dict[str, Any]]
+) -> list[dict[str, Any]]:
+ """Run shared query-context mutators after every visualization adapter."""
+ return [normalize_time_column(form_data, query) for query in queries]
+
+
+def _x_axis_label(
+ form_data: Mapping[str, Any], *, frontend_strict: bool = False
+) -> str | None:
+ """Mirror getXAxisColumn/getXAxisLabel for explicit and legacy axes."""
+ explicit = (
+ _frontend_x_axis_column(form_data)
+ if frontend_strict
+ else _x_axis_column(form_data)
+ )
+ if explicit:
+ return _label(explicit)
+ if form_data.get("granularity_sqla"):
+ return DTTM_ALIAS
+ return None
+
+
+def _rename_operator(
+ form_data: dict[str, Any],
+ query: dict[str, Any],
+ *,
+ x_axis_label: str | None,
+) -> dict[str, Any] | None:
+ """Mirror the ECharts ``renameOperator`` for Timeseries and Mixed
charts."""
+ metrics = list(query.get("metrics") or [])
+ metric_labels = [_label(metric, metric=True) for metric in metrics]
+ series_columns = query.get("series_columns")
+ columns = _as_list(
+ series_columns if series_columns is not None else query.get("columns")
+ )
+ time_offsets = _as_list(form_data.get("time_compare"))
+ offset_map = _metric_offset_map(form_data, metric_labels)
+ is_time_comparison = bool(offset_map)
+ truncate_metric = form_data.get("truncate_metric")
+
+ should_rename = (
+ bool(metrics)
+ and bool(x_axis_label)
+ and (
+ is_time_comparison
+ or (
+ (bool(columns) or len(time_offsets) > 1)
+ and "truncate_metric" in form_data
+ and bool(truncate_metric)
+ )
+ )
+ )
+ if not should_rename:
+ return None
+
+ renamed: dict[str, str | None] = {}
+ comparison_type = form_data.get("comparison_type")
+ if is_time_comparison:
+ for metric_with_offset, metric_only in offset_map.items():
+ offset_label = next(
+ (
+ str(offset)
+ for offset in time_offsets
+ if metric_with_offset.endswith(f"__{offset}")
+ ),
+ None,
+ )
+ source = (
+ metric_with_offset
+ if comparison_type == "values"
+ else f"{comparison_type}__{metric_only}__{metric_with_offset}"
+ )
+ renamed[source] = (
+ f"{metric_only}, {offset_label}" if len(metrics) > 1 else
offset_label
+ )
+
+ if (
+ comparison_type not in {"difference", "percentage", "ratio"}
+ and len(metrics) == 1
+ and not renamed
):
+ renamed[metric_labels[0]] = None
+ if not renamed:
+ return None
+ return {
+ "operation": "rename",
+ "options": {"columns": renamed, "level": 0, "inplace": True},
+ }
+
+
+def _timeseries_post_processing( # noqa: C901
+ form_data: dict[str, Any],
+ query: dict[str, Any],
+ *,
+ x_axis_label: str | None,
+ groupby: list[Any],
+ mixed: bool,
+ sort_metric: Any = None,
+) -> tuple[list[dict[str, Any]], list[Any]]:
+ """Build the Timeseries/Mixed operator pipeline in frontend order."""
+ metric_labels = [_label(value, metric=True) for value in
query.get("metrics") or []]
+ sort_metric_label = (
+ _label(sort_metric, metric=True) if sort_metric is not None else None
+ )
+ offset_map = _metric_offset_map(form_data, metric_labels)
+ time_offsets = _as_list(form_data.get("time_compare")) if offset_map else
[]
+ post_processing: list[dict[str, Any]] = []
+
+ if x_axis_label and metric_labels:
+ aggregate_labels = (
+ [*offset_map.values(), *offset_map] if offset_map else
list(metric_labels)
+ )
+ if not offset_map and sort_metric_label is not None:
+ aggregate_labels.append(sort_metric_label)
+ post_processing.append(
+ {
+ "operation": "pivot",
+ "options": {
+ "index": [x_axis_label],
+ "columns": [_label(value) for value in groupby],
+ "aggregates": {
+ label: {"operator": "mean"} for label in
aggregate_labels
+ },
+ "drop_missing_columns": not form_data.get(
+ "show_empty_columns", False
+ ),
+ },
+ }
+ )
+
+ if form_data.get("resample_method") and form_data.get("resample_rule"):
+ zero_fill = form_data["resample_method"] == "zerofill"
+ post_processing.append(
+ {
+ "operation": "resample",
+ "options": {
+ "method": "asfreq" if zero_fill else
form_data["resample_method"],
+ "rule": form_data["resample_rule"],
+ "fill_value": 0 if zero_fill else None,
+ **(
+ {"fill_time_range": True}
+ if form_data.get("resample_fill_time_range")
+ else {}
+ ),
+ },
+ }
+ )
+
+ rolling_labels = (
+ [*offset_map.values(), *offset_map] if offset_map else metric_labels
+ )
+ columns_map = {label: label for label in rolling_labels}
+ rolling_type = form_data.get("rolling_type")
+ if rolling_type == "cumsum":
+ post_processing.append(
+ {
+ "operation": "cum",
+ "options": {"operator": "sum", "columns": columns_map},
+ }
+ )
+ elif rolling_type in {"sum", "mean", "std"}:
+ post_processing.append(
+ {
+ "operation": "rolling",
+ "options": {
+ "rolling_type": rolling_type,
+ "window": int(form_data.get("rolling_periods") or 1),
+ "min_periods": int(form_data.get("min_periods") or 0),
+ "columns": columns_map,
+ },
+ }
+ )
+
+ comparison_type = form_data.get("comparison_type")
+ if offset_map and comparison_type != "values":
+ post_processing.append(
+ {
+ "operation": "compare",
+ "options": {
+ "source_columns": list(offset_map.values()),
+ "compare_columns": list(offset_map),
+ "compare_type": comparison_type,
+ "drop_original_columns": True,
+ },
+ }
+ )
+
+ if not mixed and form_data.get("contributionMode"):
+ post_processing.append(
+ {
+ "operation": "contribution",
+ "options": {
+ "orientation": form_data["contributionMode"],
+ "time_shifts": time_offsets,
+ },
+ }
+ )
+
+ if rename := _rename_operator(form_data, query, x_axis_label=x_axis_label):
+ post_processing.append(rename)
+
+ if not mixed:
+ sortable = {
+ x_axis_label or "",
+ *metric_labels,
+ sort_metric_label or "",
+ }
+ if (
+ "x_axis_sort" in form_data
+ and "x_axis_sort_asc" in form_data
+ and form_data.get("x_axis_sort") in sortable
+ and not groupby
+ ):
+ options: dict[str, Any] = {"ascending":
form_data.get("x_axis_sort_asc")}
+ if form_data.get("x_axis_sort") == x_axis_label:
+ options["is_sort_index"] = True
+ else:
+ options["by"] = form_data.get("x_axis_sort")
+ post_processing.append({"operation": "sort", "options": options})
+
+ post_processing.append({"operation": "flatten"})
+ if not mixed and form_data.get("forecastEnabled") and x_axis_label:
+ x_axis = _x_axis_column(form_data)
+ axis_grain = x_axis.get("timeGrain") if isinstance(x_axis, Mapping)
else None
+ time_grain = (
+ axis_grain
+ or (query.get("extras") or {}).get("time_grain_sqla")
+ or form_data.get("time_grain_sqla")
+ or "P1D"
+ )
+ post_processing.append(
+ {
+ "operation": "prophet",
+ "options": {
+ "time_grain": time_grain,
+ "periods": int(form_data.get("forecastPeriods") or 0),
+ "confidence_interval": float(
+ form_data.get("forecastInterval") or 0
+ ),
+ "yearly_seasonality":
form_data.get("forecastSeasonalityYearly"),
+ "weekly_seasonality":
form_data.get("forecastSeasonalityWeekly"),
+ "daily_seasonality":
form_data.get("forecastSeasonalityDaily"),
+ "index": x_axis_label,
+ },
+ }
+ )
+ return post_processing, time_offsets
+
+
+def _timeseries_query(form_data: dict[str, Any], query: dict[str, Any]) ->
None:
+ groupby = _as_list(form_data.get("groupby"))
+ x_axis = _x_axis_column(form_data)
+ x_axis_label = _x_axis_label(form_data)
+ query["columns"] = _deduplicate_fields([*_as_list(x_axis), *groupby])
+ query["series_columns"] = groupby
+ if not x_axis:
+ query["is_timeseries"] = True
+
+ # Timeseries includes its sort-only metric in the SELECT when no series is
+ # present. This lets the post-processing sort operator use a metric not
+ # otherwise displayed.
+ sort_metric = form_data.get("timeseries_limit_metric")
+ if isinstance(sort_metric, list):
+ sort_metric = next(iter(sort_metric), None)
+ if (
+ not groupby
+ and sort_metric is not None
+ and _label(sort_metric, metric=True) == form_data.get("x_axis_sort")
+ and _label(sort_metric, metric=True)
+ not in {_label(metric, metric=True) for metric in query.get("metrics")
or []}
+ ):
+ extra_metric = sort_metric
+ else:
+ extra_metric = None
+ _normalize_query_orderby(query)
+ post_processing, time_offsets = _timeseries_post_processing(
+ form_data,
+ query,
+ x_axis_label=x_axis_label,
+ groupby=groupby,
+ mixed=form_data.get("viz_type") == "mixed_timeseries",
+ sort_metric=extra_metric,
+ )
+ if extra_metric is not None:
+ query.setdefault("metrics", []).append(extra_metric)
+ query["post_processing"] = post_processing
+ query["time_offsets"] = time_offsets
+ if form_data.get("viz_type") != "mixed_timeseries":
+ query["time_compare_full_range"] = bool(
+ time_offsets and form_data.get("time_compare_full_range")
+ )
+
+
+def _big_number_queries(
+ form_data: dict[str, Any], query: dict[str, Any]
+) -> list[dict[str, Any]]:
+ """Mirror Big Number with Trendline's one/two-query contract."""
+ # Saved/native Big Number payloads can carry the temporal binding as a
+ # ``{"column_name": ...}`` mapping, which the strict frontend predicate
does
+ # not recognize; keep grouping by it rather than falling back to a total.
+ frontend_x_axis = _frontend_x_axis_column(form_data)
+ explicit_x_axis = frontend_x_axis or _x_axis_column(form_data)
+ time_column = _as_list(explicit_x_axis)
+ x_axis_label = (
+ _label(explicit_x_axis)
+ if explicit_x_axis
+ else _x_axis_label(form_data, frontend_strict=True)
+ )
+ query["columns"] = time_column
+ if time_column and frontend_x_axis is None:
+ # A native ``{"column_name": ...}`` axis groups by its temporal column
+ # but is not rewritten by normalize_time_column, so drop the legacy
+ # granularity binding here rather than bucketing the same dimension
+ # twice.
+ query.pop("granularity", None)
Review Comment:
Fixed in 0fc667fbcf76f893be11a254f0ec6ad3c73b81c1.
Saved physical `column_name` references are normalized before the
time-column mutator, and Big Number no longer removes the granularity/grain
binding. The saved mapping selects a BASE_AXIS `ds` with `timeGrain=P1M` while
retaining `granularity=ds` and the requested time range, matching the
string-axis contract.
`test_big_number_saved_axis_keeps_monthly_range_binding` covers the reported
mapping and a SQL adhoc axis; the mapping case failed before and passes after.
The native-axis pivot and actual preview query-contract tests also pass.
Validation: 1,931 passed, 3 skipped across the 14 touched pytest modules;
pre-commit passed on all 83 branch-changed files, including mypy.
##########
superset/mcp_service/chart/chart_utils.py:
##########
@@ -1694,6 +1772,792 @@ def map_bubble_config(config: BubbleChartConfig) ->
Dict[str, Any]:
return form_data
+def map_sunburst_config(config: SunburstChartConfig) -> Dict[str, Any]:
+ """Map typed Sunburst config to the ECharts ``sunburst_v2`` form_data.
+
+ The frontend control panel stores hierarchy levels under ``columns`` and
+ metrics under singular ``metric`` / ``secondary_metric`` keys. Its
+ buildQuery adds primary-metric descending ordering when ``sort_by_metric``
+ is enabled; server-side query builders mirror that transform separately.
+ """
+ form_data: Dict[str, Any] = {
+ "viz_type": "sunburst_v2",
+ "columns": [dimension.name for dimension in config.hierarchy],
+ "metric": create_metric_object(config.metric),
+ "sort_by_metric": config.sort_by_metric,
+ "row_limit": config.row_limit,
+ "show_labels": config.show_labels,
+ "show_labels_threshold": config.show_labels_threshold,
+ "show_total": config.show_total,
+ "show_null_values": config.show_null_values,
+ "label_type": config.label_type,
+ "number_format": config.number_format,
+ "date_format": config.date_format,
+ }
+ if config.secondary_metric is not None:
+ form_data["secondary_metric"] =
create_metric_object(config.secondary_metric)
+ if config.color_scheme is not None:
+ form_data["color_scheme"] = config.color_scheme
+ if config.linear_color_scheme is not None:
+ form_data["linear_color_scheme"] = config.linear_color_scheme
+ if config.time_range is not None:
+ form_data["time_range"] = config.time_range
+ if config.temporal_column is not None:
+ form_data["granularity_sqla"] = config.temporal_column
+ if config.time_grain is not None:
+ form_data["time_grain_sqla"] = config.time_grain
+
+ _copy_sunburst_native_envelope(form_data, config)
+
+ add_currency_format(form_data, config.currency_format)
+ _add_adhoc_filters(form_data, config.filters)
+ return form_data
+
+
+# Sunburst fields with explicit omission/clear semantics. Mapper defaults must
+# not overwrite same-viz state when the typed field was omitted, while explicit
+# clears must also beat the shared preservation registry on cross-viz updates.
+# Required query roles (hierarchy and metric) are deliberately absent: a full
+# replacement always updates them.
+_SUNBURST_UPDATE_FIELD_KEYS: dict[str, str] = {
+ "time_range": "time_range",
+ "time_grain": "time_grain_sqla",
+ "temporal_column": "granularity_sqla",
+ "sort_by_metric": "sort_by_metric",
+ "row_limit": "row_limit",
+ "color_scheme": "color_scheme",
+ "linear_color_scheme": "linear_color_scheme",
+ "show_labels": "show_labels",
+ "show_labels_threshold": "show_labels_threshold",
+ "show_total": "show_total",
+ "show_null_values": "show_null_values",
+ "label_type": "label_type",
+ "number_format": "number_format",
+ "date_format": "date_format",
+ "currency_format": "currency_format",
+ "extra_form_data": "extra_form_data",
+ "url_params": "url_params",
+ "standardized_form_data": "standardizedFormData",
+}
+
+
+# Presentation controls emitted sparsely by chart mappers need three-way update
+# semantics: omitted preserves saved native state, an explicit value replaces
+# it, and explicit ``None``/``False`` clears a truthy saved value when the
mapper
+# has no canonical false/null representation. Required query roles are absent:
+# a replacement config owns those through the plugin contract. Optional
grouping
+# and ordering retain saved state unless their modeled fields are supplied.
+# Paths below also cover nested axis/legend models so an omitted nested
property
+# is not mistaken for an explicit clear of the whole control.
+_MODELED_UPDATE_CONTROL_PATHS: dict[str, dict[str, tuple[tuple[str, ...],
...]]] = {
+ "GaugeChartConfig": {
+ key: ((key,),)
+ for key in (
+ "sort_by_metric",
+ "row_limit",
+ "min_val",
+ "max_val",
+ "color_scheme",
+ "font_size",
+ "number_format",
+ "currency_format",
+ "value_formatter",
+ "start_angle",
+ "end_angle",
+ "show_pointer",
+ "animation",
+ "show_axis_tick",
+ "show_split_line",
+ "split_number",
+ "show_progress",
+ "overlap",
+ "round_cap",
+ "intervals",
+ "interval_color_indices",
+ "time_range",
+ "granularity_sqla",
+ )
+ },
+ "PieChartConfig": {
+ "color_scheme": (("color_scheme",),),
+ "show_labels": (("show_labels",),),
+ "show_legend": (("show_legend",),),
+ "legendOrientation": (("legend_orientation",),),
+ "label_type": (("label_type",),),
+ "number_format": (("number_format",),),
+ "date_format": (("date_format",),),
+ "sort_by_metric": (("sort_by_metric",),),
+ "row_limit": (("row_limit",),),
+ "donut": (("donut",),),
+ "show_total": (("show_total",),),
+ "labels_outside": (("labels_outside",),),
+ "outerRadius": (("outer_radius",),),
+ "innerRadius": (("inner_radius",),),
+ "currency_format": (("currency_format",),),
+ },
+ "TableChartConfig": {
+ "order_by_cols": (("sort_by",),),
+ "row_limit": (("row_limit",),),
+ "color_scheme": (("color_scheme",),),
+ "column_config": (("column_config",),),
+ },
+ "XYChartConfig": {
+ "groupby": (("group_by",),),
+ "row_limit": (("row_limit",),),
+ "series_limit": (("series_limit",),),
+ "stack": (("stacked",),),
+ "orientation": (("orientation",),),
+ "x_axis_title": (("x_axis", "title"),),
+ "x_axis_format": (("x_axis", "format"),),
+ "y_axis_title": (("y_axis", "title"),),
+ "y_axis_format": (("y_axis", "format"),),
+ "y_axis_scale": (("y_axis", "scale"),),
+ "show_legend": (("legend", "show"),),
+ "legendOrientation": (("legend", "position"), ("legend_orientation",)),
+ "x_axis_time_format": (("x_axis_time_format",),),
+ "show_value": (("show_value",),),
+ "currency_format": (("currency_format",),),
+ "color_scheme": (("color_scheme",),),
+ },
+ "HistogramChartConfig": {
+ "bins": (("bins",),),
+ "normalize": (("normalize",),),
+ "cumulative": (("cumulative",),),
+ "row_limit": (("row_limit",),),
+ },
+ "BoxPlotChartConfig": {
+ "whiskerOptions": (
+ ("whisker_type",),
+ ("percentile_low",),
+ ("percentile_high",),
+ ),
+ "row_limit": (("row_limit",),),
+ "number_format": (("number_format",),),
+ "date_format": (("date_format",),),
+ },
+ "GanttChartConfig": {
+ "tooltip_columns": (("tooltip_columns",),),
+ "tooltip_metrics": (("tooltip_metrics",),),
+ "order_by_cols": (("order_by",),),
+ "row_limit": (("row_limit",),),
+ },
+ "WaterfallChartConfig": {
+ "show_total": (("show_total",),),
+ "show_legend": (("show_legend",),),
+ "increase_label": (("increase_label",),),
+ "decrease_label": (("decrease_label",),),
+ "total_label": (("total_label",),),
+ "x_axis_time_format": (("x_axis_time_format",),),
+ "y_axis_format": (("y_axis_format",),),
+ "currency_format": (("currency_format",),),
+ "row_limit": (("row_limit",),),
+ },
+ "BigNumberChartConfig": {
+ "subheader": (("subheader",),),
+ "y_axis_format": (("y_axis_format",),),
+ "time_format": (("time_format",),),
+ "currency_format": (("currency_format",),),
+ "color_scheme": (("color_scheme",),),
+ "start_y_axis_at_zero": (("start_y_axis_at_zero",),),
+ "compare_lag": (("compare_lag",),),
+ "aggregation": (("aggregation",),),
+ },
+ "HandlebarsChartConfig": {
+ "row_limit": (("row_limit",),),
+ "order_desc": (("order_desc",),),
+ "styleTemplate": (("style_template",),),
+ },
+ "PivotTableChartConfig": {
+ "aggregateFunction": (("aggregate_function",),),
+ "rowTotals": (("show_row_totals",),),
+ "colTotals": (("show_column_totals",),),
+ "transposePivot": (("transpose",),),
+ "combineMetric": (("combine_metric",),),
+ "valueFormat": (("value_format",),),
+ "date_format": (("date_format",),),
+ "currency_format": (("currency_format",),),
+ "row_limit": (("row_limit",),),
+ },
+ "InteractivePivotChartConfig": {
+ "order_desc": (("sort_descending",),),
+ "row_limit": (("row_limit",),),
+ "rowGroupCounts": (("show_row_group_counts",),),
+ "rowTotals": (("show_row_totals",),),
+ "colTotals": (("show_column_totals",),),
+ "colSubTotals": (("show_column_subtotals",),),
+ "valueFormat": (("value_format",),),
+ "date_format": (("date_format",),),
+ "currency_format": (("currency_format",),),
+ "colOrder": (("column_sort",),),
+ "allow_render_html": (("allow_render_html",),),
+ "expand_pivot_groups": (("expand_pivot_groups",),),
+ "time_compare": (("comparison_period",),),
+ "comparison_type": (("comparison_type",),),
+ },
+ "MixedTimeseriesChartConfig": {
+ "seriesType": (("primary_kind",),),
+ "area": (("primary_kind",),),
+ "seriesTypeB": (("secondary_kind",),),
+ "areaB": (("secondary_kind",),),
+ "show_legend": (("show_legend",),),
+ "legendOrientation": (("legend_orientation",),),
+ "show_value": (("show_value",),),
+ "color_scheme": (("color_scheme",),),
+ "currency_format": (("currency_format",),),
+ "currency_format_secondary": (("currency_format_secondary",),),
+ "xAxisTitle": (("x_axis", "title"),),
+ "x_axis_time_format": (("x_axis", "format"),),
+ "yAxisTitle": (("y_axis", "title"),),
+ "y_axis_format": (("y_axis", "format"),),
+ "logAxis": (("y_axis", "scale"),),
+ "yAxisTitleSecondary": (("y_axis_secondary", "title"),),
+ "y_axis_format_secondary": (("y_axis_secondary", "format"),),
+ "logAxisSecondary": (("y_axis_secondary", "scale"),),
+ "row_limit": (("row_limit",),),
+ },
+}
+
+
+def _model_path_was_set(config: Any, path: tuple[str, ...]) -> bool:
+ """Return whether every component of a Pydantic model path was supplied."""
+ current = config
+ for field_name in path:
+ if field_name not in getattr(current, "model_fields_set", set()):
+ return False
+ current = getattr(current, field_name, None)
+ if current is None:
+ # An explicit null parent clears all of its mapped descendants.
+ return True
+ return True
+
+
+def _apply_modeled_update_semantics(
+ existing_form_data: Mapping[str, Any],
+ new_form_data: Dict[str, Any],
+ config: Any,
+) -> set[str]:
+ """Preserve truly omitted modeled controls and return explicit clears."""
+ explicit_clears: set[str] = set()
+ controls = _MODELED_UPDATE_CONTROL_PATHS.get(type(config).__name__, {})
+ for form_key, paths in controls.items():
+ if any(_model_path_was_set(config, path) for path in paths):
+ if form_key not in new_form_data:
+ explicit_clears.add(form_key)
+ continue
+ if form_key in existing_form_data:
+ new_form_data[form_key] = existing_form_data[form_key]
+ else:
+ new_form_data.pop(form_key, None)
+ return explicit_clears
+
+
+_TEMPORAL_FORM_DATA_KEYS = frozenset(
+ {
+ "granularity",
+ "granularity_sqla",
+ "since",
+ "time_grain",
+ "time_grain_sqla",
+ "time_range",
+ "until",
+ }
+)
+
+
+def _is_temporal_filter(filter_: Any) -> bool:
+ """Return whether a native, adhoc, or legacy filter carries a time
range."""
+ return isinstance(filter_, dict) and (
+ filter_.get("operator") == FilterOperator.TEMPORAL_RANGE.value
+ or filter_.get("op") == FilterOperator.TEMPORAL_RANGE.value
+ or filter_.get("col") in {"__time_col", "__time_grain", "__time_range"}
+ )
+
+
+def _without_temporal_filters(value: Any) -> Any:
+ """Copy a filter list without temporal predicates, preserving other
shapes."""
+ if not isinstance(value, list):
+ return value
+ return [filter_ for filter_ in value if not _is_temporal_filter(filter_)]
+
+
+def _scrub_temporal_form_data(form_data: Mapping[str, Any]) -> Dict[str, Any]:
+ """Remove every source capable of reconstructing explicitly cleared time
state."""
+ scrubbed = dict(form_data)
+ for key in _TEMPORAL_FORM_DATA_KEYS:
+ scrubbed.pop(key, None)
+ scrubbed.pop(MCP_DASHBOARD_TIME_FILTER_SUBJECT, None)
+
+ for key in ("adhoc_filters", "extra_filters", "filters"):
+ if key in scrubbed:
+ scrubbed[key] = _without_temporal_filters(scrubbed[key])
+
+ extra_form_data = scrubbed.get("extra_form_data")
+ if isinstance(extra_form_data, dict):
+ cleaned_extra = dict(extra_form_data)
+ for key in _TEMPORAL_FORM_DATA_KEYS:
+ cleaned_extra.pop(key, None)
+ for key in ("adhoc_filters", "extra_filters", "filters"):
+ if key in cleaned_extra:
+ cleaned_extra[key] =
_without_temporal_filters(cleaned_extra[key])
+ scrubbed["extra_form_data"] = cleaned_extra
+ elif extra_form_data is None:
+ scrubbed.pop("extra_form_data", None)
+ return scrubbed
+
+
+# One bounded registry owns state that may survive a form-data replacement.
+# Query roles and plugin-specific controls are deliberately absent. This keeps
+# cross-viz transitions preview/save-safe without chart-by-chart allowlists
that
+# can drift as new plugins are registered.
+FORM_DATA_UPDATE_PRESERVE_KEYS: dict[str, frozenset[str]] = {
+ "envelope": frozenset(
+ {
+ "dashboardId",
+ "dashboards",
+ "datasource",
+ "extra_form_data",
+ "slice_id",
+ "slice_name",
+ "standardizedFormData",
+ "url_params",
+ }
+ ),
+ "presentation": frozenset(
+ {
+ "color_scheme",
+ "currency_format",
+ "date_format",
+ "legendOrientation",
+ "linear_color_scheme",
+ "number_format",
+ "show_legend",
+ }
+ ),
+ "filters": frozenset({"adhoc_filters", "extra_filters", "filters"}),
+ "time": frozenset(
+ {
+ "granularity_sqla",
+ "since",
+ "time_grain_sqla",
+ "time_range",
+ "until",
+ }
+ ),
+}
+_FORM_DATA_UPDATE_PRESERVE_KEYS = frozenset().union(
+ *FORM_DATA_UPDATE_PRESERVE_KEYS.values()
+)
+
+
+_SAVED_PREDICATE_FORM_DATA_KEYS = frozenset(
+ {"adhoc_filters", "extra_filters", "filters", "having", "where"}
+)
+
+
+def _merge_preserved_adhoc_filters(
+ existing_form_data: Mapping[str, Any],
+ new_form_data: Mapping[str, Any],
+ *,
+ drop_existing_temporal: bool,
+) -> list[Any] | None:
+ """Merge omitted structured filters while removing stale time bindings."""
+ previous = existing_form_data.get("adhoc_filters")
+ generated = new_form_data.get("adhoc_filters")
+ if not isinstance(previous, list):
+ return list(generated) if isinstance(generated, list) else None
+
+ previous_binding =
existing_form_data.get(MCP_DASHBOARD_TIME_FILTER_SUBJECT)
+ new_binding = new_form_data.get(MCP_DASHBOARD_TIME_FILTER_SUBJECT)
+ merged: list[Any] = []
+ for filter_ in previous:
+ is_temporal = (
+ isinstance(filter_, dict)
+ and filter_.get("operator") == FilterOperator.TEMPORAL_RANGE.value
+ )
+ stale_generated_binding = (
+ is_temporal
+ and previous_binding
+ and previous_binding != new_binding
+ and filter_.get("subject") == previous_binding
+ and filter_.get("comparator") == NO_TIME_RANGE
+ )
+ if (drop_existing_temporal and is_temporal) or stale_generated_binding:
+ continue
+ merged.append(filter_)
+
+ for filter_ in generated if isinstance(generated, list) else []:
+ if isinstance(filter_, dict):
+ same_filter = any(
+ isinstance(previous_filter, dict)
+ and previous_filter.get("clause") == filter_.get("clause")
+ and previous_filter.get("expressionType")
+ == filter_.get("expressionType")
+ and previous_filter.get("subject") == filter_.get("subject")
+ and previous_filter.get("operator") == filter_.get("operator")
+ for previous_filter in merged
+ )
+ if same_filter:
+ continue
+ elif filter_ in merged:
+ continue
+ merged.append(filter_)
+ return merged
+
+
+def _merge_allowlisted_form_data(
+ existing_form_data: Mapping[str, Any],
+ new_form_data: Mapping[str, Any],
+) -> Dict[str, Any]:
+ """Start from mapped target state and add only registry-approved
omissions."""
+ merged = dict(new_form_data)
+ for key in _FORM_DATA_UPDATE_PRESERVE_KEYS:
+ if key not in merged and key in existing_form_data:
+ merged[key] = existing_form_data[key]
+ return merged
+
+
+def merge_form_data_for_update(
+ existing_form_data: Dict[str, Any],
+ new_form_data: Dict[str, Any],
+ config: Any,
+ *,
+ dataset_rebind: bool = False,
+) -> Dict[str, Any]:
+ """Merge mapped updates without leaking query roles across visualizations.
+
+ Same-viz updates retain native controls outside the simplified MCP schema
by
+ starting from saved form data. Cross-viz updates remain bounded by the
+ shared preservation registry. Explicit clears are applied last.
+
+ A dataset rebind prunes every dataset-bound role from the saved state and
+ then merges as a same-dataset update, unless the owning plugin declares a
+ strict rebind contract (``strict_dataset_rebind``), in which case its
+ ``merge_update_form_data`` hook receives ``dataset_rebind=True``. Plugins
+ that declare ``owns_update_merge`` merge same-viz updates themselves;
+ every other update takes the shared overlay and then the plugin's
+ ``finalize_update_form_data`` hook.
+ """
+ from superset.mcp_service.chart.registry import plugin_for_viz_type
+
+ plugin = plugin_for_viz_type(new_form_data.get("viz_type"))
+ if dataset_rebind and not (plugin is not None and
plugin.strict_dataset_rebind):
+ existing_form_data = scrub_dataset_bound_form_data(
+ existing_form_data,
+ target_viz_type=new_form_data.get("viz_type"),
+ )
+ dataset_rebind = False
+
+ same_viz = existing_form_data.get("viz_type") ==
new_form_data.get("viz_type")
+ if same_viz and plugin is not None and (dataset_rebind or
plugin.owns_update_merge):
+ plugin_merged = plugin.merge_update_form_data(
+ existing_form_data,
+ new_form_data,
+ config,
+ dataset_rebind=dataset_rebind,
+ )
+ if plugin_merged is not None:
+ return plugin_merged
+ if dataset_rebind:
+ # A strict rebind never inherits saved state the plugin did not merge.
+ return dict(new_form_data)
+
+ merged = overlay_update_form_data(existing_form_data, new_form_data,
config)
+ if plugin is not None:
+ merged = plugin.finalize_update_form_data(
+ existing_form_data, new_form_data, merged, config
+ )
+ return merged
+
+
+def overlay_update_form_data(
+ existing_form_data: Dict[str, Any],
+ new_form_data: Dict[str, Any],
+ config: Any,
+) -> Dict[str, Any]:
+ """Overlay a same-dataset update on the saved state with shared
semantics."""
+ same_viz = existing_form_data.get("viz_type") ==
new_form_data.get("viz_type")
+ explicit_control_clears = (
+ _apply_modeled_update_semantics(existing_form_data, new_form_data,
config)
+ if same_viz
+ else set()
+ )
+ if same_viz:
+ from superset.mcp_service.chart.registry import (
+ query_role_keys_for_viz_type,
+ )
+
+ # Strip every target-owned query role first, then overlay the mapper's
+ # complete replacement. This removes mutually exclusive aliases (for
+ # example Pie ``metrics`` vs ``metric`` and raw vs aggregate table
+ # roles) without dropping unmodeled native presentation controls.
+ query_role_keys = query_role_keys_for_viz_type(
+ str(new_form_data.get("viz_type"))
+ )
+ merged = {
+ key: value
+ for key, value in existing_form_data.items()
+ if key not in query_role_keys
Review Comment:
Fixed in 0fc667fbcf76f893be11a254f0ec6ad3c73b81c1.
Both Mixed grouping roles participate in omitted-control restoration, and
the secondary-state lookup uses `group_by_secondary` for `groupby_b`. Explicit
secondary grouping clears emit `[]`, preventing accidental inheritance from
query A. XY preserves omitted ranking metrics; supplying either ranking-metric
spelling replaces the saved aliases, and explicit null clears them. Cross-viz
updates do not inherit these roles.
`test_mixed_update_preserves_only_omitted_grouping` and
`test_xy_update_keeps_omitted_series_ranking_metric` check the resulting
queries, including omission, independent clears, replacement, both ranking
aliases, and cross-viz behavior. The reported omission cases failed before and
pass after. Explicit-ranking/default-precedence and metric-validation tests
also pass.
Validation: 1,931 passed, 3 skipped across the 14 touched pytest modules;
pre-commit passed on all 83 branch-changed files, including mypy.
##########
superset/mcp_service/chart/preview_utils.py:
##########
@@ -1543,3 +1764,52 @@ def _generate_vega_lite_preview_from_data( # noqa: C901
data_url=None,
supports_streaming=False,
)
+
+
+def generate_xy_vega_lite_preview(
Review Comment:
Fixed in 0fc667fbcf76f893be11a254f0ec6ad3c73b81c1.
Mixed Timeseries has a Vega-Lite preview hook using the wide-result adapter,
so the renderer folds all flattened grouped series instead of selecting one
numeric column.
`test_grouped_timeseries_preview_keeps_all_pivoted_series` executes the real
Mixed query post-processing pipeline and checks the rendered fold/color/value
encodings for one or two grouping dimensions, with metric truncation enabled
and disabled. All four Mixed wide-result cases failed before and pass after.
Long-form results retain their original grouping/color encoding.
Validation: 1,931 passed, 3 skipped across the 14 touched pytest modules;
pre-commit passed on all 83 branch-changed files, including mypy.
##########
superset/mcp_service/chart/query_result.py:
##########
@@ -15,67 +15,1523 @@
# specific language governing permissions and limitations
# under the License.
-"""Helpers for interpreting ChartDataCommand result envelopes."""
+"""Canonicalize and validate ``ChartDataCommand`` result envelopes."""
+import base64
import math
+import time as system_time
+from bisect import bisect_right
from collections.abc import Mapping
+from dataclasses import dataclass
+from datetime import date, datetime, time, timedelta, timezone
from decimal import Decimal
+from enum import Enum
from numbers import Real
from typing import Any, cast
+from uuid import UUID
+from zoneinfo import ZoneInfo
+import numpy as np
+import pandas as pd
+import pytz
+from dateutil import tz as dateutil_tz, zoneinfo as dateutil_zoneinfo
+from pydantic import BaseModel
+from pydantic_core import to_json
+
+from superset.common.chart_data import ChartDataResultFormat
+from superset.common.db_query_status import QueryStatus
+from superset.constants import CACHE_DISABLED_TIMEOUT
from superset.mcp_service.chart.schemas import ChartError
+from superset.mcp_service.utils.serialization import BINARY_PREFIX
+from superset.utils.core import (
+ ExtraFiltersReasonType,
+ ExtraFiltersTimeColumnType,
+ GenericDataType,
+)
FAILED_QUERY_STATUSES = frozenset(
{"error", "failed", "stopped", "timed_out", "cancelled", "canceled"}
)
+# These are aggregate envelope limits, not per-query allowances. In particular,
+# splitting a result across the maximum number of queries must not multiply the
+# permitted rows, nodes, or encoded bytes.
+MAX_QUERY_RESULTS = 32
+MAX_QUERY_RESULT_ROWS_PER_QUERY = 50_000
+MAX_QUERY_RESULT_ROWS = 100_000
+MAX_QUERY_RESULT_COLUMNS = 4_096
+MAX_QUERY_RESULT_VALUES = 2_500_000
+MAX_QUERY_RESULT_VALUE_BYTES = 16 * 1024 * 1024
+MAX_QUERY_RESULT_METADATA_BYTES = 1024 * 1024
+MAX_QUERY_RESULT_METADATA_ITEMS = 32_768
+MAX_RESULT_VALUE_ITEMS = 4_096
+MAX_RESULT_VALUE_DEPTH = 32
+MAX_RESULT_STRING_LENGTH = 65_536
+MAX_RESULT_KEY_LENGTH = 4_096
+MAX_RESULT_INTEGER_BITS = 4_096
+MAX_RESULT_INTEGER_DIGITS = 1_234
+MAX_RESULT_DECIMAL_DIGITS = 1_024
+MAX_RESULT_DECIMAL_MAGNITUDE = 4_096
+MAX_RESULT_DECIMAL_STORAGE = 2_048
+MAX_QUERY_RESULT_ROWCOUNT = 2**63 - 1
+MAX_QUERY_RESULT_CACHE_TIMEOUT = 2**31 - 1
+MAX_QUERY_RESULT_TIMESTAMP_LENGTH = 64
+
+_ERROR_KEYS = ("error", "errors", "error_message", "message", "detail")
+_MAX_ERROR_TEXT_BYTES = 2_000
+_TRUSTED_TIMEZONE_TYPES = (timezone, ZoneInfo)
+_SAFE_RESULT_ENUM_TYPES = frozenset(
+ {
+ ChartDataResultFormat,
+ QueryStatus,
+ ExtraFiltersReasonType,
+ ExtraFiltersTimeColumnType,
+ GenericDataType,
+ }
+)
+_RESULT_FORMAT_VALUES = frozenset(
+ object.__getattribute__(member, "_value_") for member in
ChartDataResultFormat
+)
+_COLTYPE_VALUES = frozenset(
+ object.__getattribute__(member, "_value_") for member in GenericDataType
+)
+_NUMPY_INTEGER_TYPES = frozenset(
+ type(value)
+ for value in (
+ np.int8(0),
+ np.int16(0),
+ np.int32(0),
+ np.int64(0),
+ np.uint8(0),
+ np.uint16(0),
+ np.uint32(0),
+ np.uint64(0),
+ )
+)
+_NUMPY_FLOAT_TYPES = frozenset(
+ type(value)
+ for value in (np.float16(0), np.float32(0), np.float64(0),
np.longdouble(0))
+)
+_PANDAS_NAT_TYPE = type(pd.NaT)
+_PANDAS_NA_TYPE = type(pd.NA)
+_PANDAS_PERIOD_TYPE = type(pd.Period("2000-01", freq="M"))
+_PANDAS_INTERVAL_TYPE = type(pd.Interval(0, 1))
+_DATEUTIL_FIXED_TIMEZONE_TYPES = frozenset(
+ {type(dateutil_tz.tzoffset(None, 0)), type(dateutil_tz.tzutc())}
+)
+_DATEUTIL_NAMED_TIMEZONE_TYPES = frozenset(
+ {dateutil_tz.tzfile, dateutil_zoneinfo.tzfile}
+)
+_DATEUTIL_TTINFO_TYPE = type(
+ object.__getattribute__(dateutil_tz.gettz("UTC"),
"__dict__")["_ttinfo_std"]
+)
+_DATEUTIL_LOCAL_TIMEZONE_TYPE = type(dateutil_tz.tzlocal())
+_PYTZ_FIXED_TIMEZONE_TYPES = frozenset({type(pytz.FixedOffset(1))})
+_MAX_DATEUTIL_TRANSITIONS = 4_096
+_MAX_DATEUTIL_TTINFOS = 256
+_MAX_DATEUTIL_TRANSITION_MAGNITUDE = 10**12
+
+
+@dataclass
+class _ResultBudget:
+ """Aggregate counters shared by every query and metadata value."""
+
+ rows: int = 0
+ values: int = 0
+ json_bytes: int = 0
+ metadata_items: int = 0
+ metadata_bytes: int = 0
+
+
+@dataclass(frozen=True)
+class _DateutilTimezoneState:
+ """Hook-free subset of a validated exact dateutil tzfile transition
table."""
+
+ transitions: tuple[int, ...]
+ transition_offsets: tuple[int, ...]
+ standard_offset: int
+ before_offset: int | None
+
-def _query_error_text(value: Any) -> str | None:
- """Convert a bounded query error payload into a useful message."""
- if value is None or value is False:
+def _invalid_result(message: str) -> ChartError:
+ return ChartError(
+ error=f"Chart query returned {message}.",
+ error_type="InvalidQueryResult",
+ )
+
+
+def _invalid_metadata(label: str) -> ChartError:
+ return ChartError(
+ error=f"{label} returned hostile or malformed metadata.",
+ error_type="InvalidQueryResult",
+ )
+
+
+def _safe_enum_value(value: Any, expected: frozenset[type[Any]]) -> Any | None:
+ """Read trusted enum storage without invoking public conversion hooks."""
+ if type(value) not in expected or type(value) not in
_SAFE_RESULT_ENUM_TYPES:
return None
- if isinstance(value, Mapping):
- for key in ("error", "error_message", "message", "detail"):
- if text := _query_error_text(value.get(key)):
- return text
+ return object.__getattribute__(value, "_value_")
+
+
+def _bounded_utf8_length(value: str, maximum: int) -> int | None:
+ """Return the exact UTF-8 size while bounding pre-encoding work."""
+ if str.__len__(value) > maximum:
+ return None
+ try:
+ encoded = str.encode(value, "utf-8", errors="strict")
+ except UnicodeEncodeError:
return None
- if isinstance(value, (list, tuple)):
- parts = [text for item in value if (text := _query_error_text(item))]
- return "; ".join(parts[:3]) or None
- text = str(value)
- return text[:2000] if text else None
+ size = bytes.__len__(encoded)
+ return size if size <= maximum else None
-def _failure_for_query_payload(
- payload: Mapping[str, Any], label: str
-) -> ChartError | None:
- """Extract one failure from a top-level or per-query payload."""
+def _json_string_size(value: str, maximum: int) -> int | None:
+ """Return compact UTF-8 JSON string size without serializing the value."""
+ raw_size = _bounded_utf8_length(value, maximum)
+ if raw_size is None:
+ return None
+ escaped_size = raw_size + 2
+ for character in value:
+ codepoint = ord(character)
+ if character in {'"', "\\", "\b", "\t", "\n", "\f", "\r"}:
+ escaped_size += 1
+ elif codepoint < 0x20:
+ escaped_size += 5
+ return escaped_size
+
+
+def _integer_json_size(value: int) -> int:
+ """Return exact decimal JSON size without rendering the bounded integer."""
+ magnitude = -value if value < 0 else value
+ if magnitude == 0:
+ digits = 1
+ else:
+ bits = int.bit_length(magnitude)
+ digits = ((bits - 1) * 30103) // 100000 + 1
+ if magnitude >= 10**digits:
+ digits += 1
+ return digits + (value < 0)
+
+
+def _container_json_syntax_size(item_count: int, *, mapping: bool) -> int:
+ """Return braces/brackets plus compact separators and mapping colons."""
+ if item_count == 0:
+ return 2
+ return 2 + item_count - 1 + (item_count if mapping else 0)
+
+
+def _normalized_scalar_json_size(value: Any) -> int:
+ """Return exact compact JSON size for a normalized scalar."""
+ value_type = type(value)
+ if value is None:
+ return 4
+ if value_type is bool:
+ return 4 if value else 5
+ if value_type is str:
+ size = _json_string_size(value, MAX_RESULT_STRING_LENGTH)
+ assert size is not None
+ return size
+ if value_type is int:
+ return _integer_json_size(value)
+ if value_type is float:
+ return len(float.__repr__(value))
+ if value_type is Decimal:
+ # Pydantic serializes Decimal values as JSON strings so their exact
+ # finite value survives the wire projection without binary rounding.
+ text = Decimal.__str__(value)
+ size = _json_string_size(text, MAX_RESULT_STRING_LENGTH)
+ assert size is not None
+ return size
+ raise AssertionError("result scalar was not normalized")
+
+
+def _pydantic_scalar_json_size(value: Any) -> int:
+ """Return the scalar size emitted by Pydantic's JSON serializer."""
+ if type(value) is float:
+ # pydantic-core uses the shortest exponent (``1e-7``), while Python's
+ # repr retains a leading zero (``1e-07``).
+ return len(to_json(value))
+ return _normalized_scalar_json_size(value)
+
+
+def _charge_json_bytes(
+ budget: _ResultBudget, size: int, *, metadata: bool = False
+) -> str | None:
+ budget.json_bytes += size
+ if budget.json_bytes > MAX_QUERY_RESULT_VALUE_BYTES:
+ return "too many aggregate JSON bytes"
+ if metadata:
+ budget.metadata_bytes += size
+ if budget.metadata_bytes > MAX_QUERY_RESULT_METADATA_BYTES:
+ return "too many aggregate metadata JSON bytes"
+ return None
+
+
+def _charge_value(budget: _ResultBudget, *, metadata: bool = False) -> str |
None:
+ budget.values += 1
+ if budget.values > MAX_QUERY_RESULT_VALUES:
+ return "too many aggregate values"
+ if metadata:
+ budget.metadata_items += 1
+ if budget.metadata_items > MAX_QUERY_RESULT_METADATA_ITEMS:
+ return "too many aggregate metadata values"
+ return None
+
+
+def _charge_text(
+ value: str,
+ budget: _ResultBudget,
+ *,
+ key: bool = False,
+ metadata: bool = False,
+) -> str | None:
+ maximum = (
+ MAX_RESULT_KEY_LENGTH
+ if key
+ else MAX_QUERY_RESULT_METADATA_BYTES
+ if metadata
+ else MAX_RESULT_STRING_LENGTH
+ )
+ size = _json_string_size(value, maximum)
+ if size is None:
+ return "an invalid or oversized object key" if key else "invalid text
data"
+ return _charge_json_bytes(budget, size, metadata=metadata)
+
+
+def _integer_failure(value: int) -> str | None:
+ bits = int.bit_length(value)
+ if bits > MAX_RESULT_INTEGER_BITS:
+ return "an oversized integer"
+ digits = 1 if bits == 0 else ((bits - 1) * 30103) // 100000 + 1
+ if digits > MAX_RESULT_INTEGER_DIGITS:
+ return "an oversized integer"
+ return None
+
+
+def _decimal_failure(value: Decimal) -> str | None:
+ if Decimal.__sizeof__(value) > MAX_RESULT_DECIMAL_STORAGE:
+ return "an oversized Decimal"
+ if not Decimal.is_finite(value):
+ return "a non-finite Decimal"
+ parts = Decimal.as_tuple(value)
+ if tuple.__len__(parts.digits) > MAX_RESULT_DECIMAL_DIGITS:
+ return "an oversized Decimal"
+ exponent = parts.exponent
+ if type(exponent) is not int or abs(exponent) >
MAX_RESULT_DECIMAL_MAGNITUDE:
+ return "an oversized Decimal"
+ return None
+
+
+def _type_mro(value_type: type[Any]) -> tuple[type[Any], ...]:
+ """Read a concrete type's MRO without consulting metaclass overrides."""
+ try:
+ mro = type.__getattribute__(value_type, "__mro__")
+ except (AttributeError, TypeError): # pragma: no cover - defensive
metaclass
+ return ()
+ return mro if type(mro) is tuple else ()
+
+
+def _timezone_name_without_hooks(tzinfo: Any) -> str | None: # noqa: C901
+ """Read common pytz/dateutil zone state without dispatching timezone
hooks."""
+ value_mro = _type_mro(type(tzinfo))
+ if any(base is pytz.tzinfo.BaseTzInfo for base in value_mro):
+ for base in value_mro:
+ try:
+ namespace = type.__getattribute__(base, "__dict__")
+ except (AttributeError, TypeError): # pragma: no cover
+ continue
+ zone = namespace.get("zone")
+ if type(zone) is str and _bounded_utf8_length(zone, 256) is not
None:
+ try:
+ canonical = pytz.timezone(zone)
+ except (KeyError, ValueError):
+ return None
+ # Generated pytz types are trusted; arbitrary subclasses that
+ # inherit their internal fields are not.
+ return zone if type(canonical) is type(tzinfo) else None
+
+ if type(tzinfo) not in _DATEUTIL_NAMED_TIMEZONE_TYPES:
+ return None
+
+ try:
+ namespace = object.__getattribute__(tzinfo, "__dict__")
+ except (AttributeError, TypeError):
+ return None
+ if type(namespace) is not dict:
+ return None
+ filename = dict.get(namespace, "_filename")
+ if type(filename) is not str or _bounded_utf8_length(filename, 4_096) is
None:
+ return None
+ marker = "/zoneinfo/"
+ if (offset := str.find(filename, marker)) >= 0:
+ name = str.__getitem__(filename, slice(offset + len(marker), None))
+ elif not str.startswith(filename, "/") and str.find(filename, "\\") < 0:
+ name = filename
+ else:
+ return None
+ parts = str.split(name, "/")
+ if not parts or any(part in {"", ".", ".."} for part in parts):
+ return None
+ return name if _bounded_utf8_length(name, 256) is not None else None
+
+
+def _object_namespace(value: Any) -> dict[str, Any] | None:
+ """Read exact instance storage without descriptor dispatch."""
+ try:
+ namespace = object.__getattribute__(value, "__dict__")
+ except (AttributeError, TypeError):
+ return None
+ return namespace if type(namespace) is dict else None
+
+
+def _dateutil_ttinfo_offset_without_hooks(value: Any) -> int | None:
+ """Validate one exact dateutil transition record and return its offset."""
+ if type(value) is not _DATEUTIL_TTINFO_TYPE:
+ return None
+ try:
+ offset = object.__getattribute__(value, "offset")
+ delta = object.__getattribute__(value, "delta")
+ isdst = object.__getattribute__(value, "isdst")
+ abbreviation = object.__getattribute__(value, "abbr")
+ is_standard = object.__getattribute__(value, "isstd")
+ is_gmt = object.__getattribute__(value, "isgmt")
+ dst_offset = object.__getattribute__(value, "dstoffset")
+ except (AttributeError, TypeError):
+ return None
+ if type(offset) is not int or not -86_400 < offset < 86_400:
+ return None
+ if type(delta) is not timedelta or delta != timedelta(seconds=offset):
+ return None
+ if type(isdst) is not int or isdst not in {0, 1}:
+ return None
+ if abbreviation is not None and (
+ type(abbreviation) is not str or _bounded_utf8_length(abbreviation,
256) is None
+ ):
+ return None
+ if type(is_standard) is not bool or type(is_gmt) is not bool:
+ return None
+ if type(dst_offset) is not timedelta:
+ return None
+ if not -timedelta(days=1) < dst_offset < timedelta(days=1):
+ return None
+ return offset
+
+
+def _dateutil_named_state_without_hooks( # noqa: C901
+ tzinfo: Any,
+) -> _DateutilTimezoneState | None:
+ """Validate bounded exact dateutil tzfile state without timezone hooks."""
+ if type(tzinfo) not in _DATEUTIL_NAMED_TIMEZONE_TYPES:
+ return None
+ if _timezone_name_without_hooks(tzinfo) is None:
+ return None
+ namespace = _object_namespace(tzinfo)
+ if namespace is None:
+ return None
+ transitions = dict.get(namespace, "_trans_list")
+ utc_transitions = dict.get(namespace, "_trans_list_utc")
+ transition_info = dict.get(namespace, "_trans_idx")
+ info_list = dict.get(namespace, "_ttinfo_list")
+ standard_info = dict.get(namespace, "_ttinfo_std")
+ before_info = dict.get(namespace, "_ttinfo_before")
+ first_info = dict.get(namespace, "_ttinfo_first")
+ if (
+ type(transitions) is not tuple
+ or type(utc_transitions) is not tuple
+ or type(transition_info) is not tuple
+ or type(info_list) is not list
+ or tuple.__len__(transitions) > _MAX_DATEUTIL_TRANSITIONS
+ or tuple.__len__(utc_transitions) != tuple.__len__(transitions)
+ or tuple.__len__(transition_info) != tuple.__len__(transitions)
+ or list.__len__(info_list) == 0
+ or list.__len__(info_list) > _MAX_DATEUTIL_TTINFOS
+ ):
+ return None
+
+ previous_transition: int | None = None
+ previous_utc_transition: int | None = None
+ for index in range(tuple.__len__(transitions)):
+ transition = tuple.__getitem__(transitions, index)
+ utc_transition = tuple.__getitem__(utc_transitions, index)
+ if (
+ type(transition) is not int
+ or type(utc_transition) is not int
+ or abs(transition) > _MAX_DATEUTIL_TRANSITION_MAGNITUDE
+ or abs(utc_transition) > _MAX_DATEUTIL_TRANSITION_MAGNITUDE
+ or (previous_transition is not None and transition <=
previous_transition)
+ or (
+ previous_utc_transition is not None
+ and utc_transition <= previous_utc_transition
+ )
+ ):
+ return None
+ previous_transition = transition
+ previous_utc_transition = utc_transition
+
+ known_info_ids: set[int] = set()
+ for index in range(list.__len__(info_list)):
+ info = list.__getitem__(info_list, index)
+ if _dateutil_ttinfo_offset_without_hooks(info) is None:
+ return None
+ known_info_ids.add(id(info))
+ for info in (standard_info, before_info, first_info):
+ if info is not None and id(info) not in known_info_ids:
+ return None
+ for index in range(tuple.__len__(transition_info)):
+ if id(tuple.__getitem__(transition_info, index)) not in known_info_ids:
+ return None
+ if not transitions:
+ if (
+ standard_info is not list.__getitem__(info_list, 0)
+ or first_info is not standard_info
+ or before_info is not None
+ ):
+ return None
+ else:
+ expected_standard = None
+ expected_dst = None
+ for index in range(tuple.__len__(transition_info) - 1, -1, -1):
+ info = tuple.__getitem__(transition_info, index)
+ is_dst = object.__getattribute__(info, "isdst")
+ if expected_standard is None and not is_dst:
+ expected_standard = info
+ elif expected_dst is None and is_dst:
+ expected_dst = info
+ if expected_standard is not None and expected_dst is not None:
+ break
+ if expected_standard is None:
+ expected_standard = expected_dst
+ expected_before = None
+ for index in range(list.__len__(info_list)):
+ info = list.__getitem__(info_list, index)
+ if not object.__getattribute__(info, "isdst"):
+ expected_before = info
+ break
+ if expected_before is None:
+ expected_before = list.__getitem__(info_list, 0)
+ if standard_info is not expected_standard or before_info is not
expected_before:
+ return None
+ standard_offset = _dateutil_ttinfo_offset_without_hooks(standard_info)
+ if standard_offset is None:
+ return None
+ before_offset = (
+ _dateutil_ttinfo_offset_without_hooks(before_info)
+ if before_info is not None
+ else None
+ )
+ transition_offsets: list[int] = []
+ previous_offset: int | None = None
+ previous_base_offset: int | None = None
+ previous_is_dst: int | None = None
+ previous_dst_offset = 0
+ for index in range(tuple.__len__(transition_info)):
+ info = tuple.__getitem__(transition_info, index)
+ if id(info) not in known_info_ids:
+ return None
+ offset = _dateutil_ttinfo_offset_without_hooks(info)
+ if offset is None:
+ return None
+ is_dst = object.__getattribute__(info, "isdst")
+ dst_offset_seconds = 0
+ if previous_is_dst is not None and is_dst:
+ if not previous_is_dst:
+ assert previous_offset is not None
+ dst_offset_seconds = offset - previous_offset
+ if not dst_offset_seconds and previous_dst_offset:
+ dst_offset_seconds = previous_dst_offset
+ previous_dst_offset = dst_offset_seconds
+ base_offset = offset - dst_offset_seconds
+ adjustment = base_offset
+ if (
+ previous_base_offset is not None
+ and base_offset != previous_base_offset
+ and is_dst != previous_is_dst
+ ):
+ adjustment = previous_base_offset
+ if (
+ tuple.__getitem__(transitions, index)
+ != tuple.__getitem__(utc_transitions, index) + adjustment
+ ):
+ return None
+ transition_offsets.append(offset)
+ previous_offset = offset
+ previous_base_offset = base_offset
+ previous_is_dst = is_dst
+ if transitions and before_offset is None:
+ return None
+ return _DateutilTimezoneState(
+ transitions=transitions,
+ transition_offsets=tuple(transition_offsets),
+ standard_offset=standard_offset,
+ before_offset=before_offset,
+ )
+
+
+def _dateutil_named_offset_without_hooks( # noqa: C901
+ value: datetime, tzinfo: Any
+) -> timezone | None:
+ """Preserve dateutil's source-selected wall offset from validated state."""
+ state = _dateutil_named_state_without_hooks(tzinfo)
+ if state is None:
+ return None
+ epoch_ordinal = date.toordinal(date(1970, 1, 1))
+ wall_timestamp = (
+ (datetime.toordinal(value) - epoch_ordinal) * 86_400
+ + value.hour * 3_600
+ + value.minute * 60
+ + value.second
+ )
+ transitions = state.transitions
+ selected_offset: int | None
+ if not transitions:
+ selected_offset = state.standard_offset
+ else:
+ index = bisect_right(transitions, wall_timestamp) - 1
+
+ def offset_at(transition_index: int | None) -> int | None:
+ if transition_index is None or transition_index + 1 >=
len(transitions):
+ return state.standard_offset
+ if transition_index < 0:
+ return state.before_offset
+ return state.transition_offsets[transition_index]
+
+ if index > 0:
+ selected_offset = offset_at(index)
+ previous_offset = offset_at(index - 1)
+ if selected_offset is None or previous_offset is None:
+ return None
+ is_ambiguous = wall_timestamp < transitions[index] + (
+ previous_offset - selected_offset
+ )
+ if not value.fold and is_ambiguous:
+ index -= 1
+ selected_offset = offset_at(index)
+ if selected_offset is None:
+ return None
+ try:
+ return timezone(timedelta(seconds=selected_offset))
+ except (OverflowError, ValueError):
+ return None
+
+
+def _pytz_named_offset_without_hooks(tzinfo: Any) -> timezone | None:
+ """Return a localized pytz zone's stored offset without calling hooks."""
+ if _timezone_name_without_hooks(tzinfo) is None:
+ return None
+ namespace = _object_namespace(tzinfo)
+ offset = dict.get(namespace, "_utcoffset") if namespace is not None else
None
+ if offset is None:
+ # Static pytz zones store their fixed offset on the generated class.
+ class_namespace = type.__getattribute__(type(tzinfo), "__dict__")
+ offset = class_namespace.get("_utcoffset")
+ if type(offset) is not timedelta:
+ return None
+ try:
+ return timezone(offset)
+ except ValueError:
+ return None
+
+
+def _dateutil_local_offset_without_hooks(
+ value: datetime, tzinfo: Any
+) -> timezone | None:
+ """Select a dateutil local offset using builtin system-time data."""
+ if type(tzinfo) is not _DATEUTIL_LOCAL_TIMEZONE_TYPE:
+ return None
+ namespace = _object_namespace(tzinfo)
+ if namespace is None:
+ return None
+ standard_offset = dict.get(namespace, "_std_offset")
+ daylight_offset = dict.get(namespace, "_dst_offset")
+ has_daylight = dict.get(namespace, "_hasdst")
+ if (
+ type(standard_offset) is not timedelta
+ or type(daylight_offset) is not timedelta
+ or type(has_daylight) is not bool
+ ):
+ return None
+ selected_offset = standard_offset
+ if has_daylight:
+ epoch = datetime(1970, 1, 1)
+ naive = datetime(
+ value.year,
+ value.month,
+ value.day,
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ )
+ timestamp = (naive - epoch).total_seconds()
+ try:
+ is_daylight = bool(
+ system_time.localtime(timestamp +
system_time.timezone).tm_isdst
+ )
+ daylight_saved = daylight_offset - standard_offset
+ previous_is_daylight = bool(
+ system_time.localtime(
+ timestamp
+ - timedelta.total_seconds(daylight_saved)
+ + system_time.timezone
+ ).tm_isdst
+ )
+ except (OverflowError, OSError, ValueError):
+ return None
+ if not is_daylight and is_daylight != previous_is_daylight:
+ is_daylight = not bool(value.fold)
+ selected_offset = daylight_offset if is_daylight else standard_offset
+ try:
+ return timezone(selected_offset)
+ except ValueError:
+ return None
+
+
+def _canonical_timezone(tzinfo: Any) -> timezone | ZoneInfo | None: # noqa:
C901
+ if any(type(tzinfo) is trusted for trusted in _TRUSTED_TIMEZONE_TYPES):
+ return tzinfo
+ if tzinfo is pytz.UTC:
+ return timezone.utc
+ if type(tzinfo) in _DATEUTIL_FIXED_TIMEZONE_TYPES:
+ try:
+ namespace = object.__getattribute__(tzinfo, "__dict__")
+ except (AttributeError, TypeError):
+ return timezone.utc if type(tzinfo) is type(dateutil_tz.tzutc())
else None
+ if type(namespace) is not dict:
+ return None
+ offset = dict.get(namespace, "_offset")
+ if type(offset) is not timedelta:
+ return timezone.utc if type(tzinfo) is type(dateutil_tz.tzutc())
else None
+ if abs(offset) >= timedelta(days=1):
+ return None
+ return timezone(offset)
+ if type(tzinfo) in _PYTZ_FIXED_TIMEZONE_TYPES:
+ try:
+ namespace = object.__getattribute__(tzinfo, "__dict__")
+ except (AttributeError, TypeError):
+ return None
+ if type(namespace) is not dict:
+ return None
+ minutes = dict.get(namespace, "_minutes")
+ if type(minutes) is not int or not -1_440 < minutes < 1_440:
+ return None
+ return timezone(timedelta(minutes=minutes))
+ return None
+
+
+def _timestamp_offset_without_hooks(value: pd.Timestamp) -> timezone | None:
+ """Recover a timestamp's stored wall-clock offset without timezone
hooks."""
+ multipliers = {"s": 1_000_000_000, "ms": 1_000_000, "us": 1_000, "ns": 1}
+ multiplier = multipliers.get(value.unit)
+ if multiplier is None:
+ return None
+ try:
+ instant_ns = int(value.asm8.view("i8")) * multiplier
+ epoch_ordinal = date.toordinal(date(1970, 1, 1))
+ wall_ns = (
+ (
+ (datetime.toordinal(value) - epoch_ordinal) * 86_400
+ + value.hour * 3600
+ + value.minute * 60
+ + value.second
+ )
+ * 1_000_000_000
+ + value.microsecond * 1000
+ + value.nanosecond
+ )
+ offset_ns = wall_ns - instant_ns
+ if offset_ns % 1000:
+ return None
+ return timezone(timedelta(microseconds=offset_ns // 1000))
+ except (OverflowError, TypeError, ValueError):
+ return None
+
+
+def _canonical_datetime(value: datetime) -> tuple[str | None, str | None]:
+ """Serialize an exact datetime through trusted timezone state only."""
+ tzinfo = value.tzinfo
+ canonical_value = value
+ if tzinfo is not None and not any(
+ type(tzinfo) is trusted for trusted in _TRUSTED_TIMEZONE_TYPES
+ ):
+ canonical_tz = (
+ _pytz_named_offset_without_hooks(tzinfo)
+ or _dateutil_local_offset_without_hooks(value, tzinfo)
+ or _dateutil_named_offset_without_hooks(value, tzinfo)
+ or _canonical_timezone(tzinfo)
+ )
+ if canonical_tz is None:
+ return None, "a datetime with an unsupported timezone"
+ canonical_value = datetime(
+ value.year,
+ value.month,
+ value.day,
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ tzinfo=canonical_tz,
+ fold=value.fold,
+ )
+ try:
+ return datetime.isoformat(canonical_value), None
+ except (OverflowError, TypeError, ValueError):
+ return None, "an invalid datetime"
+
+
+def _canonical_time(value: time) -> tuple[str | None, str | None]:
+ """Serialize an exact time through trusted timezone state only."""
+ tzinfo = value.tzinfo
+ canonical_value = value
+ if tzinfo is not None and not any(
+ type(tzinfo) is trusted for trusted in _TRUSTED_TIMEZONE_TYPES
+ ):
+ if _dateutil_named_state_without_hooks(tzinfo) is not None:
+ canonical_value = time(
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ fold=value.fold,
+ )
+ else:
+ canonical_tz = _canonical_timezone(tzinfo)
+ if canonical_tz is None and type(tzinfo) is
_DATEUTIL_LOCAL_TIMEZONE_TYPE:
+ namespace = _object_namespace(tzinfo)
+ if namespace is not None and dict.get(namespace, "_hasdst") is
False:
+ offset = dict.get(namespace, "_std_offset")
+ if type(offset) is timedelta:
+ try:
+ canonical_tz = timezone(offset)
+ except ValueError:
+ canonical_tz = None
+ if canonical_tz is None:
+ return None, "a time with an unsupported timezone"
+ canonical_value = time(
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ tzinfo=canonical_tz,
+ fold=value.fold,
+ )
+ try:
+ return time.isoformat(canonical_value), None
+ except (OverflowError, TypeError, ValueError):
+ return None, "an invalid time"
+
+
+def _canonical_timestamp(value: pd.Timestamp) -> tuple[str | None, str | None]:
+ """Preserve a trusted timestamp's instant, offset, nanoseconds, and
fold."""
+ try:
+ tzinfo = value.tzinfo
+ if tzinfo is not None and not any(
+ type(tzinfo) is trusted for trusted in _TRUSTED_TIMEZONE_TYPES
+ ):
+ supported_timezone = (
+ _canonical_timezone(tzinfo) is not None
+ or _pytz_named_offset_without_hooks(tzinfo) is not None
+ or type(tzinfo) is _DATEUTIL_LOCAL_TIMEZONE_TYPE
+ or _dateutil_named_state_without_hooks(tzinfo) is not None
+ )
+ if not supported_timezone:
+ return None, "a timestamp with an unsupported timezone"
+ canonical_tz = _timestamp_offset_without_hooks(value)
+ if canonical_tz is None:
+ return None, "an invalid timestamp"
+ raw_value = value.asm8.view("i8")
+ value = pd.Timestamp(raw_value, unit=value.unit,
tz="UTC").tz_convert(
+ canonical_tz
+ )
+ return pd.Timestamp.isoformat(value), None
+ except (KeyError, OverflowError, TypeError, ValueError):
+ return None, "an invalid timestamp"
+
+
+def _canonical_binary(
+ value: bytes | bytearray | memoryview,
+) -> tuple[str | None, str | None]:
+ """Render exact binary cells as UTF-8 text or ``base64:``-prefixed text.
+
+ Matches the MCP response serializer: valid UTF-8 is returned as text and
+ anything else is losslessly base64-encoded. The raw size is bounded before
+ copying so the rendered string stays within the per-cell text budget.
+ """
+ size = value.nbytes if type(value) is memoryview else len(value)
+ if size > MAX_RESULT_STRING_LENGTH:
+ return None, "an oversized binary value"
+ try:
+ raw = bytes(value)
+ except (BufferError, TypeError, ValueError):
+ return None, "an invalid binary value"
+ try:
+ text = raw.decode("utf-8")
+ except UnicodeDecodeError:
+ text = BINARY_PREFIX + base64.b64encode(raw).decode("ascii")
+ if _bounded_utf8_length(text, MAX_RESULT_STRING_LENGTH) is None:
+ return None, "an oversized binary value"
+ return text, None
+
+
+def _normalize_scalar(value: Any) -> tuple[Any, str | None]: # noqa: C901
+ """Convert one exact trusted producer scalar to a JSON-safe scalar."""
+ value_type = type(value)
+ if value is None or value_type is bool or value_type is str:
+ return value, None
+ if value_type is int:
+ return value, _integer_failure(value)
+ if value_type is float:
+ if math.isnan(value):
+ return None, None
+ return (value, None) if math.isfinite(value) else (None, "a non-finite
number")
+ if value_type is Decimal:
+ return value, _decimal_failure(value)
+ if value_type is datetime:
+ return _canonical_datetime(value)
+ if value_type is time:
+ return _canonical_time(value)
+ if value_type is date:
+ return date.isoformat(value), None
+ if value_type is timedelta:
+ try:
+ return pd.Timedelta(value).isoformat(), None
+ except (OverflowError, TypeError, ValueError):
+ return None, "an invalid duration"
+ if value_type is UUID:
+ return UUID.__str__(value), None
+ if value_type is bytes or value_type is bytearray or value_type is
memoryview:
+ return _canonical_binary(value)
+
+ if value_type is _PANDAS_NAT_TYPE or value_type is _PANDAS_NA_TYPE:
+ return None, None
+ if value_type is pd.Timestamp:
+ return _canonical_timestamp(value)
+ if value_type is pd.Timedelta:
+ if pd.isna(value):
+ return None, None
+ try:
+ return pd.Timedelta.isoformat(value), None
+ except (OverflowError, TypeError, ValueError):
+ return None, "an invalid pandas duration"
+ if value_type is _PANDAS_PERIOD_TYPE or value_type is
_PANDAS_INTERVAL_TYPE:
+ # These concrete immutable pandas extension scalars are trusted. Exact
+ # type checks deliberately exclude subclasses with conversion hooks.
+ try:
+ normalized_text = str(value)
+ except (OverflowError, TypeError, ValueError):
+ return None, "an invalid pandas scalar"
+ if _bounded_utf8_length(normalized_text, MAX_RESULT_STRING_LENGTH) is
None:
+ return None, "an oversized pandas scalar"
+ return normalized_text, None
+ if value_type in _NUMPY_INTEGER_TYPES:
+ normalized = int(value)
+ return normalized, _integer_failure(normalized)
+ if value_type in _NUMPY_FLOAT_TYPES:
+ normalized_float = float(value)
+ if math.isnan(normalized_float):
+ return None, None
+ return (
+ (normalized_float, None)
+ if math.isfinite(normalized_float)
+ else (None, "a non-finite NumPy number")
+ )
+ if value_type is np.bool_:
+ return bool(value), None
+ if value_type is np.str_:
+ return str(value), None
+ if value_type is np.datetime64:
+ if np.isnat(value):
+ return None, None
+ try:
+ return _canonical_timestamp(pd.Timestamp(value))
+ except (OverflowError, TypeError, ValueError):
+ return None, "an invalid NumPy timestamp"
+ if value_type is np.timedelta64:
+ if np.isnat(value):
+ return None, None
+ try:
+ return pd.Timedelta(value).isoformat(), None
+ except (OverflowError, TypeError, ValueError):
+ return None, "an invalid NumPy duration"
+ return None, "an unsupported or subclassed value"
+
+
+def _normalize_value( # noqa: C901
+ value: Any,
+ budget: _ResultBudget,
+ *,
+ enum_types: frozenset[type[Any]] = frozenset(),
+ metadata: bool = False,
+) -> tuple[Any, str | None]:
+ """Iteratively normalize one bounded exact-container value tree."""
+ stack: list[
+ tuple[Any, list[Any] | dict[str, Any] | None, int | str | None, int,
bool]
+ ] = [(value, None, None, 0, False)]
+ active_containers: set[int] = set()
+ root = value
+
+ while stack:
+ item, parent, slot, depth, leaving = stack.pop()
+ if leaving:
+ active_containers.remove(id(item))
+ continue
+ if depth > MAX_RESULT_VALUE_DEPTH:
+ return None, "excessively nested data"
+ if reason := _charge_value(budget, metadata=metadata):
+ return None, reason
+
+ if type(item) is list or type(item) is tuple or type(item) is
np.ndarray:
+ identity = id(item)
+ if identity in active_containers:
+ return None, "cyclic containers"
+ active_containers.add(identity)
+ if type(item) is np.ndarray:
+ if item.ndim == 0:
+ return None, "a non-sequence array"
+ width = item.shape[0]
+ elif type(item) is tuple:
+ width = tuple.__len__(item)
+ else:
+ width = list.__len__(cast(list[Any], item))
+ if width > MAX_RESULT_VALUE_ITEMS:
+ return None, "an oversized array"
+ if reason := _charge_json_bytes(
+ budget,
+ _container_json_syntax_size(width, mapping=False),
+ metadata=metadata,
+ ):
+ return None, reason
+ # Convert only exact producer sequences, after checking their size.
+ # Traverse children through the same depth/node/byte budgets, and
+ # track the original container so object-array cycles stay bounded.
+ if type(item) is tuple:
+ normalized_array = [
+ tuple.__getitem__(item, index) for index in range(width)
+ ]
+ elif type(item) is np.ndarray:
+ normalized_array = [
+ np.ndarray.__getitem__(item, index) for index in
range(width)
+ ]
+ else:
+ normalized_array = cast(list[Any], item)
+ if normalized_array is not item:
+ if parent is None:
+ root = normalized_array
+ elif type(parent) is list:
+ assert type(slot) is int
+ list.__setitem__(parent, slot, normalized_array)
+ else:
+ assert type(parent) is dict
+ assert type(slot) is str
+ dict.__setitem__(parent, slot, normalized_array)
+ stack.append((item, None, None, depth, True))
+ stack.extend(
+ (
+ list.__getitem__(normalized_array, index),
+ normalized_array,
+ index,
+ depth + 1,
+ False,
+ )
+ for index in range(width - 1, -1, -1)
+ )
+ continue
+
+ if type(item) is dict:
+ identity = id(item)
+ if identity in active_containers:
+ return None, "cyclic containers"
+ active_containers.add(identity)
+ width = dict.__len__(item)
+ if width > MAX_RESULT_VALUE_ITEMS:
+ return None, "an oversized object"
+ if reason := _charge_json_bytes(
+ budget,
+ _container_json_syntax_size(width, mapping=True),
+ metadata=metadata,
+ ):
+ return None, reason
+ children: list[tuple[Any, dict[str, Any], str, int, bool]] = []
+ for key, child in dict.items(item):
+ if type(key) is not str:
Review Comment:
Fixed in 0fc667fbcf76f893be11a254f0ec6ad3c73b81c1.
Nested row maps stringify exact built-in keys within the existing key/text
budgets, and non-finite Python/NumPy floating-point values become null
recursively. Integer bounds, container/depth/node/byte limits, cycle checks,
and rejection of arbitrary objects/subclasses remain enforced. Metadata retains
its strict wire contract.
`test_nested_warehouse_values_keep_serializer_coercions` covers integer
keys, nested maps/arrays, key collisions, and non-finite floats; its six cases
failed before and pass after. Additional tests verify oversized/hostile keys
are still rejected and exercise the reported map/array cells through the actual
`get_chart_data`, `query_dataset`, and built-in `get_table` entry points.
Validation: 1,931 passed, 3 skipped across the 14 touched pytest modules;
pre-commit passed on all 83 branch-changed files, including mypy.
##########
UPDATING.md:
##########
@@ -24,6 +24,10 @@ assists people when migrating to a new version.
## Next
+- MCP `update_chart` requires a complete `config` when changing `dataset_id`
Review Comment:
Fixed in 0fc667fbcf76f893be11a254f0ec6ad3c73b81c1.
UPDATING.md explicitly documents the mandatory result caps before response
serialization and CSV/XLSX export, including the 50,000-row per-query limit and
aggregate/container/text budgets. It states that raising `SQL_MAX_ROW` or
raising/disabling `MCP_RESPONSE_SIZE_CONFIG` cannot change these fixed limits,
and explains that oversized saved Table exports return `InvalidQueryResult` and
require lower limits, filtering, narrower selection, or aggregation.
This is a documentation-only correction to the cap behavior: the
unconditional validator calls before export and the constants in
`query_result.py` were verified, and the existing per-query/aggregate row-cap
tests pass. Runtime caps were not changed.
Validation: 1,931 passed, 3 skipped across the 14 touched pytest modules;
pre-commit passed on all 83 branch-changed files, including mypy.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]