sadpandajoe commented on code in PR #43770:
URL: https://github.com/apache/superset/pull/43770#discussion_r4191487503
##########
superset/mcp_service/chart/compile.py:
##########
@@ -209,6 +240,22 @@ def _adhoc_filter_column_valid(
WHERE filters must reference a physical column; HAVING filters may also
reference a saved metric because Superset resolves metric names there.
"""
+ column_names = [item["name"] for item in dataset_context.available_columns]
+ metric_names = [metric["name"] for metric in
dataset_context.available_metrics]
+ if column in column_names or (clause == "HAVING" and column in
metric_names):
+ return True
+ try:
+ if resolve_dataset_column(column, dataset_context) is not None:
+ return True
+ except ValueError:
Review Comment:
`AmbiguousDatasetReferenceError` subclasses `ValueError`, so this `except
ValueError` swallows it before the caller's `except
AmbiguousDatasetReferenceError` can run. With physical columns `Score` and
`SCORE` and a preserved WHERE filter on `score`, the user now gets the generic
`CHART_VALIDATION_FAILED` "not in dataset" error instead of
`AMBIGUOUS_DATASET_REFERENCE` listing both candidates. The earlier
`test_preserved_filter_ambiguity_is_actionable` that asserted this was removed
in the same change, so nothing catches it. Could the ambiguity be re-raised
here (and that regression restored)? The statements after the `return` on the
previous lines are also unreachable now.
##########
superset/mcp_service/chart/chart_helpers.py:
##########
@@ -809,10 +1738,560 @@ def build_mixed_timeseries_secondary(
return qd
-# Deck.gl viz types that conditionally set is_timeseries from time_grain_sqla
-_DECK_TIMESERIES_VIZ_TYPES: frozenset[str] = frozenset(
- {"deck_arc", "deck_path", "deck_polygon", "deck_scatter",
"deck_screengrid"}
-)
+def build_histogram_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Histogram buildQuery, including its histogram post-processing."""
+ column = form_data.get("column")
+ histogram_groupby = _as_list(form_data.get("groupby"))
+ query = build_single_query_dict(
+ form_data,
+ [*histogram_groupby, column] if column is not None else
histogram_groupby,
+ [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ having_filter = bool(form_data.get("having")) or any(
+ isinstance(filter_, dict) and filter_.get("clause") == "HAVING"
+ for filter_ in form_data.get("adhoc_filters") or []
+ )
+ if having_filter:
+ query["metrics"] = [
+ {
+ "expressionType": "SQL",
+ "sqlExpression": "COUNT(*)",
+ "label": "COUNT(*)",
+ }
+ ]
+ bins = form_data.get("bins", 5)
+ try:
+ parsed_bins = float(bins)
+ parsed_bins = int(parsed_bins) if parsed_bins.is_integer() else
parsed_bins
+ except (TypeError, ValueError):
+ parsed_bins = 5
+ query["post_processing"] = [
+ {
+ "operation": "histogram",
+ "options": {
+ "column": _column_label(column),
+ "groupby": [
+ label
+ for item in histogram_groupby
+ if (label := _column_label(item))
+ ],
+ "bins": parsed_bins,
+ "cumulative": bool(form_data.get("cumulative")),
+ "normalize": bool(form_data.get("normalize")),
+ },
+ }
+ ]
+ return [query]
+
+
+def build_box_plot_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Box Plot buildQuery, including its boxplot post-processing."""
+ distribute = _as_list(form_data.get("columns"))
+ if not distribute and form_data.get("granularity_sqla"):
+ distribute = [form_data["granularity_sqla"]]
+ box_groupby = _as_list(form_data.get("groupby"))
+ query = build_single_query_dict(
+ form_data,
+ [
+ *(_temporal_column(column, form_data) for column in distribute),
+ *box_groupby,
+ ],
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ query["series_columns"] = box_groupby
+ if whisker := form_data.get("whiskerOptions"):
+ whisker_type = "tukey"
+ percentiles: list[int] | None = None
+ if whisker == "Min/max (no outliers)":
+ whisker_type = "min/max"
+ elif match := re.fullmatch(r"(\d{1,3})/(\d{1,3}) percentiles",
str(whisker)):
+ whisker_type = "percentile"
+ percentiles = [int(match.group(1)), int(match.group(2))]
+ elif whisker != "Tukey":
+ raise ValueError(f"Unsupported whisker type: {whisker}")
+ query["post_processing"] = [
+ {
+ "operation": "boxplot",
+ "options": {
+ "whisker_type": whisker_type,
+ "percentiles": percentiles,
+ "groupby": [
+ label
+ for column in box_groupby
+ if (label := _column_label(column))
+ ],
+ "metrics": [
+ label
+ for metric in query["metrics"]
+ if (label := _metric_label(metric))
+ ],
+ },
+ }
+ ]
+ return [query]
+
+
+def build_pivot_table_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Pivot Table buildQuery, including subtotal grouping sets."""
+ rows = _as_list(form_data.get("groupbyRows"))
+ pivot_columns = _as_list(form_data.get("groupbyColumns"))
+ if form_data.get("transposePivot"):
+ rows, pivot_columns = pivot_columns, rows
+ columns = _dedupe_query_fields([*rows, *pivot_columns], _column_label)
+ query = build_single_query_dict(
+ form_data,
+ [_temporal_column(column, form_data) for column in columns],
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ sort_metric = query.get("series_limit_metric")
+ if sort_metric is None and query["metrics"]:
+ sort_metric = query["metrics"][0]
+ if sort_metric is not None:
+ query["orderby"] = [[sort_metric, not query.get("order_desc", True)]]
+ if grouping_sets := _pivot_grouping_sets(form_data, rows, pivot_columns):
+ query["grouping_sets"] = grouping_sets
+ return [query]
+
+
+def build_pie_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ contribution: bool,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Pie/Sunburst buildQuery; Pie adds a contribution operator."""
+ metric = form_data.get("metric")
+ query = build_single_query_dict(
+ form_data,
+ _as_list(form_data.get("groupby")),
+ [metric] if metric is not None else [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ if form_data.get("sort_by_metric") and metric is not None:
+ query["orderby"] = [[metric, False]]
+ if contribution and (label := _metric_label(metric)):
+ query["post_processing"] = [
+ {
+ "operation": "contribution",
+ "options": {
+ "columns": [label],
+ "rename_columns": [f"{label}__contribution"],
+ },
+ }
+ ]
+ return [query]
+
+
+def _positive_int(value: Any) -> int:
+ """Coerce a stored limit (int, numeric string, or empty) to a positive int
or 0."""
+ try:
+ coerced = int(value)
+ except (TypeError, ValueError):
+ return 0
+ return coerced if coerced > 0 else 0
+
+
+def build_table_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Table buildQuery: percent metrics, comparisons, totals,
paging."""
+ raw_mode = form_data.get("query_mode") == "raw" or (
+ form_data.get("query_mode") not in {"raw", "aggregate"}
+ and bool(form_data.get("all_columns"))
+ )
+ # Native extractQueryFields excludes empty-string column references.
+ table_columns = [
+ column
+ for column in _as_list(
+ form_data.get("all_columns") if raw_mode else
form_data.get("groupby")
+ )
+ if column != ""
+ ]
+ table_metrics = [] if raw_mode else _as_list(form_data.get("metrics"))
+ percent_metrics = [] if raw_mode else
_as_list(form_data.get("percent_metrics"))
+ query_metrics = _dedupe_query_fields(
+ [*table_metrics, *percent_metrics], _metric_label
+ )
+ table_orderby = _parse_orderby(form_data.get("order_by_cols"))
+ if not raw_mode:
+ sort_metrics = _as_list(form_data.get("timeseries_limit_metric"))
+ if sort_metrics:
+ table_orderby = [[sort_metrics[0], not form_data.get("order_desc",
False)]]
+ elif table_metrics:
+ table_orderby = [[table_metrics[0], False]]
+ query = build_single_query_dict(
+ form_data,
+ table_columns,
+ query_metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=table_orderby,
+ )
+ if not raw_mode:
+ # Table selects one temporal axis and places it before the other roles.
+ for index, column in enumerate(table_columns):
+ temporal_column = _temporal_column(column, form_data)
+ if temporal_column is not column:
+ query["columns"] = [
+ temporal_column,
+ *table_columns[:index],
+ *table_columns[index + 1 :],
+ ]
+ break
+ # Native comparisons use ordinary metrics, before percentage-only metrics
+ # are added to the selected query and contribution operator.
+ has_time_comparison = _time_comparison(form_data, table_metrics)
+ offsets = _table_time_offsets(form_data, {**query, "metrics":
table_metrics})
+ query["time_offsets"] = offsets
+ post_processing: list[dict[str, Any]] = []
+ contribution: dict[str, Any] | None = None
+ if percent_metrics:
+ labels: list[str] = []
+ for metric in percent_metrics:
+ if label := _metric_label(metric):
+ candidates = [label]
+ if has_time_comparison:
+ candidates.extend(f"{label}__{offset}" for offset in
offsets)
+ for candidate in candidates:
+ if candidate not in labels:
+ labels.append(candidate)
+ contribution = {
+ "operation": "contribution",
+ "options": {
+ "columns": labels,
+ "rename_columns": [f"%{label}" for label in labels],
+ },
+ }
+ post_processing.append(contribution)
+ if has_time_comparison and offsets and form_data.get("comparison_type") !=
"values":
+ source: list[str] = []
+ shifted: list[str] = []
+ for metric in table_metrics:
+ if label := _metric_label(metric):
+ for offset in offsets:
+ source.append(label)
+ shifted.append(f"{label}__{offset}")
+ post_processing.append(
+ {
+ "operation": "compare",
+ "options": {
+ "source_columns": source,
+ "compare_columns": shifted,
+ "compare_type": form_data.get("comparison_type"),
+ "drop_original_columns": True,
+ },
+ }
+ )
+ query["post_processing"] = post_processing
+
+ # ``query["row_limit"]`` is the normalized caller limit (explicit request
+ # limit or the saved row_limit, which may be stored as a string); page
+ # sizing narrows it but never replaces it.
+ configured_limit = _positive_int(query.get("row_limit"))
+ if form_data.get("server_pagination"):
+ if page_size := _positive_int(form_data.get("server_page_length")):
+ query["row_limit"] = (
+ min(page_size, configured_limit) if configured_limit else
page_size
+ )
+ query["row_offset"] = 0
+
+ extra_queries: list[dict[str, Any]] = []
+ if form_data.get("percent_metric_calculation") == "all_records" and
percent_metrics:
+ extra_queries.append(
+ {
+ **query,
+ "columns": [],
+ "metrics": percent_metrics,
+ "post_processing": [],
+ "row_limit": 0,
+ "row_offset": 0,
+ "orderby": [],
+ "is_timeseries": False,
+ }
+ )
+ if query_metrics and form_data.get("show_totals") and not raw_mode:
+ totals = {
+ **query,
+ "columns": [],
+ "metrics": _table_totals_metrics(
+ query_metrics, form_data.get("totals_aggregate")
+ ),
+ "row_limit": 0,
+ "row_offset": 0,
+ "post_processing": [contribution] if contribution else [],
+ }
+ totals.pop("orderby", None)
+ totals.pop("order_desc", None)
+ extra_queries.append(totals)
+ if form_data.get("server_pagination"):
+ rowcount = {
+ **query,
+ "time_offsets": [],
+ "row_limit": configured_limit or 0,
+ "row_offset": 0,
+ "post_processing": [],
+ "is_rowcount": True,
+ }
+ return [query, rowcount, *extra_queries]
+ return [query, *extra_queries]
+
+
+def build_gantt_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Gantt buildQuery with its interval columns and series."""
+ (
+ gantt_columns,
+ gantt_metrics,
+ gantt_orderby,
+ gantt_groupby,
+ ) = resolve_gantt_query_fields(form_data)
+ query = build_single_query_dict(
+ form_data,
+ gantt_columns,
+ gantt_metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=gantt_orderby,
+ )
+ query["series_columns"] = gantt_groupby
+ return [query]
+
+
+def build_interactive_pivot_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Interactive Pivot Table buildQuery."""
+ interactive_columns = [
+ _temporal_column(column, form_data)
+ for column in _as_list(form_data.get("groupby"))
+ ]
+ query = build_single_query_dict(
+ form_data,
+ interactive_columns,
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ _normalize_orderby(query)
+ return [query]
+
+
+def build_big_number_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ trendline: bool,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Big Number (with or without trendline) buildQuery."""
+ metric = form_data.get("metric")
+ if metric is None:
+ plural_metrics = _as_list(form_data.get("metrics"))
+ metric = plural_metrics[0] if plural_metrics else None
+ columns = _resolve_big_number_query_columns(form_data) if trendline else []
+ query = build_single_query_dict(
+ form_data,
+ columns,
+ [metric] if metric is not None else [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ if trendline:
+ # Big Number has no series dimension. Its frontend pivot receives
+ # the common base QueryObject (whose columns are empty), not the
+ # final QueryObject after the explicit x-axis is added. Preserve
+ # that distinction instead of falling back to the final columns.
+ query["series_columns"] = []
+ if not form_data.get("x_axis"):
+ query["is_timeseries"] = True
+ query["post_processing"] = _timeseries_post_processing(form_data,
query)
+ if form_data.get("aggregation") == "raw":
+ return [
+ query,
+ {
+ **query,
+ "columns": [],
+ "is_timeseries": False,
+ "post_processing": [],
+ },
+ ]
+ return [query]
+
+
+def build_waterfall_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Waterfall buildQuery with raw-axis ordering."""
+ metrics, groupby = resolve_metrics_and_groupby(form_data)
+ # normalizeTimeColumn runs after Waterfall's buildQuery callback. It
+ # wraps only the final x-axis column; orderby deliberately retains the
+ # raw control value produced inside the callback.
+ raw_axis = form_data.get("x_axis") or form_data.get("granularity_sqla")
+ query_axis = (
+ _normalized_x_axis_query_field(form_data)
+ if form_data.get("x_axis")
+ else raw_axis
+ )
+ waterfall_columns = ([query_axis] if query_axis else []) + groupby
+ raw_ordering_columns = ([raw_axis] if raw_axis else []) + groupby
+ query = build_single_query_dict(
+ form_data,
+ waterfall_columns,
+ metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=None,
+ )
+ query["orderby"] = [[column, True] for column in raw_ordering_columns]
+ if form_data.get("x_axis"):
+ query.pop("is_timeseries", None)
+ return [query]
+
+
+def build_mixed_timeseries_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Mixed Timeseries buildQuery for both independent layers."""
+ from superset.utils.core import split_adhoc_filters_into_base_filters
+
+ queries: list[dict[str, Any]] = []
+ x_axis = _normalized_x_axis_query_field(form_data)
+ for secondary in (False, True):
+ layer = _mixed_layer_form_data(form_data, secondary=secondary)
+ if secondary and form_data.get("adhoc_filters_b") is not None:
+ for key in ("filters", "where", "having"):
+ layer.pop(key, None)
+ secondary_filters = list(form_data.get("adhoc_filters_b") or [])
+ # Request-level filters (extra_form_data / extra_filters) are not
+ # suffixed, so they apply to both layers on top of the layer's own.
+ secondary_filters.extend(
+ filter_
+ for filter_ in form_data.get("adhoc_filters") or []
+ if isinstance(filter_, dict)
+ and filter_.get("isExtra")
+ and filter_ not in secondary_filters
+ )
+ layer["adhoc_filters"] = secondary_filters
+ split_adhoc_filters_into_base_filters(layer, engine)
+ layer_metrics = _timeseries_base_metrics(layer)
+ layer_groupby = _as_list(layer.get("groupby"))
+ columns = [*(_as_list(x_axis) if x_axis else []), *layer_groupby]
+ query = build_single_query_dict(
+ layer,
+ _dedupe_query_fields(columns, _column_label),
+ layer_metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=layer.get("orderby"),
+ )
+ query["series_columns"] = layer_groupby
+ if not x_axis:
+ query["is_timeseries"] = True
+ comparison = _time_comparison(layer, layer_metrics)
+ query["time_offsets"] = (
+ _as_list(layer.get("time_compare")) if comparison else []
+ )
+ query["post_processing"] = _timeseries_post_processing(
+ layer,
+ query,
+ operator_metrics=layer_metrics,
+ )
+ _normalize_orderby(query)
+ queries.append(query)
+ return queries
+
+
+def build_timeseries_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render ECharts Timeseries buildQuery and its post-processing."""
+ x_axis = _normalized_x_axis_query_field(form_data)
+ timeseries_metrics = _timeseries_base_metrics(form_data)
+ extra_metrics = _timeseries_extra_metrics(form_data)
+ timeseries_groupby = _as_list(form_data.get("groupby"))
+ columns = [*(_as_list(x_axis) if x_axis else []), *timeseries_groupby]
+ query = build_single_query_dict(
+ form_data,
+ _dedupe_query_fields(columns, _column_label),
+ [*timeseries_metrics, *extra_metrics],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ query["series_columns"] = timeseries_groupby
+ if not x_axis:
+ query["is_timeseries"] = True
+ comparison = _time_comparison(form_data, timeseries_metrics)
+ query["time_offsets"] = (
+ _as_list(form_data.get("time_compare")) if comparison else []
+ )
+ query["time_compare_full_range"] = bool(
+ query["time_offsets"] and form_data.get("time_compare_full_range")
+ )
+ query["post_processing"] = _timeseries_post_processing(
Review Comment:
This pivot post-processing turns a grouped timeseries into wide columns such
as `SUM(revenue), EU` and `SUM(revenue), US`, but the generic Vega-Lite
renderer in `preview_utils.py` still looks for the long-form `region` column
plus one metric column. For a preview of a line chart with `x=event_date`,
`SUM(revenue)` and `groupby=["region"]`, `_resolve_y_metric_column` falls back
to the first numeric column and no color encoding is emitted, so only one
series is drawn and the preview still reports success. Could the preview path
unpivot these rows (or the XY plugin supply a matching preview adapter), with a
grouped-series test?
##########
superset/mcp_service/chart/chart_helpers.py:
##########
@@ -809,10 +1738,560 @@ def build_mixed_timeseries_secondary(
return qd
-# Deck.gl viz types that conditionally set is_timeseries from time_grain_sqla
-_DECK_TIMESERIES_VIZ_TYPES: frozenset[str] = frozenset(
- {"deck_arc", "deck_path", "deck_polygon", "deck_scatter",
"deck_screengrid"}
-)
+def build_histogram_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Histogram buildQuery, including its histogram post-processing."""
+ column = form_data.get("column")
+ histogram_groupby = _as_list(form_data.get("groupby"))
+ query = build_single_query_dict(
+ form_data,
+ [*histogram_groupby, column] if column is not None else
histogram_groupby,
+ [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ having_filter = bool(form_data.get("having")) or any(
+ isinstance(filter_, dict) and filter_.get("clause") == "HAVING"
+ for filter_ in form_data.get("adhoc_filters") or []
+ )
+ if having_filter:
+ query["metrics"] = [
+ {
+ "expressionType": "SQL",
+ "sqlExpression": "COUNT(*)",
+ "label": "COUNT(*)",
+ }
+ ]
+ bins = form_data.get("bins", 5)
+ try:
+ parsed_bins = float(bins)
+ parsed_bins = int(parsed_bins) if parsed_bins.is_integer() else
parsed_bins
+ except (TypeError, ValueError):
+ parsed_bins = 5
+ query["post_processing"] = [
+ {
+ "operation": "histogram",
+ "options": {
+ "column": _column_label(column),
+ "groupby": [
+ label
+ for item in histogram_groupby
+ if (label := _column_label(item))
+ ],
+ "bins": parsed_bins,
+ "cumulative": bool(form_data.get("cumulative")),
+ "normalize": bool(form_data.get("normalize")),
+ },
+ }
+ ]
+ return [query]
+
+
+def build_box_plot_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Box Plot buildQuery, including its boxplot post-processing."""
+ distribute = _as_list(form_data.get("columns"))
+ if not distribute and form_data.get("granularity_sqla"):
+ distribute = [form_data["granularity_sqla"]]
+ box_groupby = _as_list(form_data.get("groupby"))
+ query = build_single_query_dict(
+ form_data,
+ [
+ *(_temporal_column(column, form_data) for column in distribute),
+ *box_groupby,
+ ],
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ query["series_columns"] = box_groupby
+ if whisker := form_data.get("whiskerOptions"):
+ whisker_type = "tukey"
+ percentiles: list[int] | None = None
+ if whisker == "Min/max (no outliers)":
+ whisker_type = "min/max"
+ elif match := re.fullmatch(r"(\d{1,3})/(\d{1,3}) percentiles",
str(whisker)):
+ whisker_type = "percentile"
+ percentiles = [int(match.group(1)), int(match.group(2))]
+ elif whisker != "Tukey":
+ raise ValueError(f"Unsupported whisker type: {whisker}")
+ query["post_processing"] = [
+ {
+ "operation": "boxplot",
+ "options": {
+ "whisker_type": whisker_type,
+ "percentiles": percentiles,
+ "groupby": [
+ label
+ for column in box_groupby
+ if (label := _column_label(column))
+ ],
+ "metrics": [
+ label
+ for metric in query["metrics"]
+ if (label := _metric_label(metric))
+ ],
+ },
+ }
+ ]
+ return [query]
+
+
+def build_pivot_table_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Pivot Table buildQuery, including subtotal grouping sets."""
+ rows = _as_list(form_data.get("groupbyRows"))
+ pivot_columns = _as_list(form_data.get("groupbyColumns"))
+ if form_data.get("transposePivot"):
+ rows, pivot_columns = pivot_columns, rows
+ columns = _dedupe_query_fields([*rows, *pivot_columns], _column_label)
+ query = build_single_query_dict(
+ form_data,
+ [_temporal_column(column, form_data) for column in columns],
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ sort_metric = query.get("series_limit_metric")
+ if sort_metric is None and query["metrics"]:
+ sort_metric = query["metrics"][0]
+ if sort_metric is not None:
+ query["orderby"] = [[sort_metric, not query.get("order_desc", True)]]
+ if grouping_sets := _pivot_grouping_sets(form_data, rows, pivot_columns):
+ query["grouping_sets"] = grouping_sets
+ return [query]
+
+
+def build_pie_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ contribution: bool,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Pie/Sunburst buildQuery; Pie adds a contribution operator."""
+ metric = form_data.get("metric")
+ query = build_single_query_dict(
+ form_data,
+ _as_list(form_data.get("groupby")),
+ [metric] if metric is not None else [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ if form_data.get("sort_by_metric") and metric is not None:
+ query["orderby"] = [[metric, False]]
+ if contribution and (label := _metric_label(metric)):
+ query["post_processing"] = [
+ {
+ "operation": "contribution",
+ "options": {
+ "columns": [label],
+ "rename_columns": [f"{label}__contribution"],
+ },
+ }
+ ]
+ return [query]
+
+
+def _positive_int(value: Any) -> int:
+ """Coerce a stored limit (int, numeric string, or empty) to a positive int
or 0."""
+ try:
+ coerced = int(value)
+ except (TypeError, ValueError):
+ return 0
+ return coerced if coerced > 0 else 0
+
+
+def build_table_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Table buildQuery: percent metrics, comparisons, totals,
paging."""
+ raw_mode = form_data.get("query_mode") == "raw" or (
+ form_data.get("query_mode") not in {"raw", "aggregate"}
+ and bool(form_data.get("all_columns"))
+ )
+ # Native extractQueryFields excludes empty-string column references.
+ table_columns = [
+ column
+ for column in _as_list(
+ form_data.get("all_columns") if raw_mode else
form_data.get("groupby")
+ )
+ if column != ""
+ ]
+ table_metrics = [] if raw_mode else _as_list(form_data.get("metrics"))
+ percent_metrics = [] if raw_mode else
_as_list(form_data.get("percent_metrics"))
+ query_metrics = _dedupe_query_fields(
+ [*table_metrics, *percent_metrics], _metric_label
+ )
+ table_orderby = _parse_orderby(form_data.get("order_by_cols"))
+ if not raw_mode:
+ sort_metrics = _as_list(form_data.get("timeseries_limit_metric"))
+ if sort_metrics:
+ table_orderby = [[sort_metrics[0], not form_data.get("order_desc",
False)]]
+ elif table_metrics:
+ table_orderby = [[table_metrics[0], False]]
+ query = build_single_query_dict(
+ form_data,
+ table_columns,
+ query_metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=table_orderby,
+ )
+ if not raw_mode:
+ # Table selects one temporal axis and places it before the other roles.
+ for index, column in enumerate(table_columns):
+ temporal_column = _temporal_column(column, form_data)
+ if temporal_column is not column:
+ query["columns"] = [
+ temporal_column,
+ *table_columns[:index],
+ *table_columns[index + 1 :],
+ ]
+ break
+ # Native comparisons use ordinary metrics, before percentage-only metrics
+ # are added to the selected query and contribution operator.
+ has_time_comparison = _time_comparison(form_data, table_metrics)
+ offsets = _table_time_offsets(form_data, {**query, "metrics":
table_metrics})
+ query["time_offsets"] = offsets
+ post_processing: list[dict[str, Any]] = []
+ contribution: dict[str, Any] | None = None
+ if percent_metrics:
+ labels: list[str] = []
+ for metric in percent_metrics:
+ if label := _metric_label(metric):
+ candidates = [label]
+ if has_time_comparison:
+ candidates.extend(f"{label}__{offset}" for offset in
offsets)
+ for candidate in candidates:
+ if candidate not in labels:
+ labels.append(candidate)
+ contribution = {
+ "operation": "contribution",
+ "options": {
+ "columns": labels,
+ "rename_columns": [f"%{label}" for label in labels],
+ },
+ }
+ post_processing.append(contribution)
+ if has_time_comparison and offsets and form_data.get("comparison_type") !=
"values":
+ source: list[str] = []
+ shifted: list[str] = []
+ for metric in table_metrics:
+ if label := _metric_label(metric):
+ for offset in offsets:
+ source.append(label)
+ shifted.append(f"{label}__{offset}")
+ post_processing.append(
+ {
+ "operation": "compare",
+ "options": {
+ "source_columns": source,
+ "compare_columns": shifted,
+ "compare_type": form_data.get("comparison_type"),
+ "drop_original_columns": True,
+ },
+ }
+ )
+ query["post_processing"] = post_processing
+
+ # ``query["row_limit"]`` is the normalized caller limit (explicit request
+ # limit or the saved row_limit, which may be stored as a string); page
+ # sizing narrows it but never replaces it.
+ configured_limit = _positive_int(query.get("row_limit"))
+ if form_data.get("server_pagination"):
+ if page_size := _positive_int(form_data.get("server_page_length")):
+ query["row_limit"] = (
+ min(page_size, configured_limit) if configured_limit else
page_size
+ )
+ query["row_offset"] = 0
+
+ extra_queries: list[dict[str, Any]] = []
+ if form_data.get("percent_metric_calculation") == "all_records" and
percent_metrics:
+ extra_queries.append(
+ {
+ **query,
+ "columns": [],
+ "metrics": percent_metrics,
+ "post_processing": [],
+ "row_limit": 0,
+ "row_offset": 0,
+ "orderby": [],
+ "is_timeseries": False,
+ }
+ )
+ if query_metrics and form_data.get("show_totals") and not raw_mode:
+ totals = {
+ **query,
+ "columns": [],
+ "metrics": _table_totals_metrics(
+ query_metrics, form_data.get("totals_aggregate")
+ ),
+ "row_limit": 0,
+ "row_offset": 0,
+ "post_processing": [contribution] if contribution else [],
+ }
+ totals.pop("orderby", None)
+ totals.pop("order_desc", None)
+ extra_queries.append(totals)
+ if form_data.get("server_pagination"):
+ rowcount = {
+ **query,
+ "time_offsets": [],
+ "row_limit": configured_limit or 0,
+ "row_offset": 0,
+ "post_processing": [],
+ "is_rowcount": True,
+ }
+ return [query, rowcount, *extra_queries]
+ return [query, *extra_queries]
+
+
+def build_gantt_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Gantt buildQuery with its interval columns and series."""
+ (
+ gantt_columns,
+ gantt_metrics,
+ gantt_orderby,
+ gantt_groupby,
+ ) = resolve_gantt_query_fields(form_data)
+ query = build_single_query_dict(
+ form_data,
+ gantt_columns,
+ gantt_metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=gantt_orderby,
+ )
+ query["series_columns"] = gantt_groupby
+ return [query]
+
+
+def build_interactive_pivot_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Interactive Pivot Table buildQuery."""
+ interactive_columns = [
+ _temporal_column(column, form_data)
+ for column in _as_list(form_data.get("groupby"))
+ ]
+ query = build_single_query_dict(
+ form_data,
+ interactive_columns,
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ _normalize_orderby(query)
+ return [query]
+
+
+def build_big_number_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ trendline: bool,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Big Number (with or without trendline) buildQuery."""
+ metric = form_data.get("metric")
+ if metric is None:
+ plural_metrics = _as_list(form_data.get("metrics"))
+ metric = plural_metrics[0] if plural_metrics else None
+ columns = _resolve_big_number_query_columns(form_data) if trendline else []
+ query = build_single_query_dict(
+ form_data,
+ columns,
+ [metric] if metric is not None else [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ if trendline:
+ # Big Number has no series dimension. Its frontend pivot receives
+ # the common base QueryObject (whose columns are empty), not the
+ # final QueryObject after the explicit x-axis is added. Preserve
+ # that distinction instead of falling back to the final columns.
+ query["series_columns"] = []
+ if not form_data.get("x_axis"):
+ query["is_timeseries"] = True
+ query["post_processing"] = _timeseries_post_processing(form_data,
query)
+ if form_data.get("aggregation") == "raw":
+ return [
+ query,
+ {
+ **query,
+ "columns": [],
+ "is_timeseries": False,
+ "post_processing": [],
+ },
+ ]
+ return [query]
+
+
+def build_waterfall_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Waterfall buildQuery with raw-axis ordering."""
+ metrics, groupby = resolve_metrics_and_groupby(form_data)
+ # normalizeTimeColumn runs after Waterfall's buildQuery callback. It
+ # wraps only the final x-axis column; orderby deliberately retains the
+ # raw control value produced inside the callback.
+ raw_axis = form_data.get("x_axis") or form_data.get("granularity_sqla")
+ query_axis = (
+ _normalized_x_axis_query_field(form_data)
+ if form_data.get("x_axis")
+ else raw_axis
+ )
+ waterfall_columns = ([query_axis] if query_axis else []) + groupby
+ raw_ordering_columns = ([raw_axis] if raw_axis else []) + groupby
+ query = build_single_query_dict(
+ form_data,
+ waterfall_columns,
+ metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=None,
+ )
+ query["orderby"] = [[column, True] for column in raw_ordering_columns]
+ if form_data.get("x_axis"):
+ query.pop("is_timeseries", None)
+ return [query]
+
+
+def build_mixed_timeseries_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Mixed Timeseries buildQuery for both independent layers."""
+ from superset.utils.core import split_adhoc_filters_into_base_filters
+
+ queries: list[dict[str, Any]] = []
+ x_axis = _normalized_x_axis_query_field(form_data)
+ for secondary in (False, True):
+ layer = _mixed_layer_form_data(form_data, secondary=secondary)
+ if secondary and form_data.get("adhoc_filters_b") is not None:
+ for key in ("filters", "where", "having"):
+ layer.pop(key, None)
+ secondary_filters = list(form_data.get("adhoc_filters_b") or [])
+ # Request-level filters (extra_form_data / extra_filters) are not
+ # suffixed, so they apply to both layers on top of the layer's own.
+ secondary_filters.extend(
+ filter_
+ for filter_ in form_data.get("adhoc_filters") or []
+ if isinstance(filter_, dict)
+ and filter_.get("isExtra")
+ and filter_ not in secondary_filters
+ )
+ layer["adhoc_filters"] = secondary_filters
+ split_adhoc_filters_into_base_filters(layer, engine)
+ layer_metrics = _timeseries_base_metrics(layer)
+ layer_groupby = _as_list(layer.get("groupby"))
+ columns = [*(_as_list(x_axis) if x_axis else []), *layer_groupby]
+ query = build_single_query_dict(
+ layer,
+ _dedupe_query_fields(columns, _column_label),
+ layer_metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
Review Comment:
Passing the shared `order_desc` here overrides the secondary layer's own
`order_desc_b` that `_mixed_layer_form_data` just folded into `layer`. Callers
such as `get_chart_preview` and the cached-form-data path in `get_chart_data`
always pass `form_data.get("order_desc", True)`, so a Mixed chart with
`order_desc=False`, `order_desc_b=True`, `limit_b=1` and
`timeseries_limit_metric_b="costs"` picks the lowest-cost series for layer B
instead of the highest, which changes which data comes back rather than just
its order. Could the secondary layer prefer its own resolved `order_desc` when
a layer-specific value exists, with a regression that passes the primary flag
explicitly?
##########
superset/mcp_service/chart/query_result.py:
##########
@@ -18,90 +18,2010 @@
"""Helpers for interpreting ChartDataCommand result envelopes."""
import math
-from collections.abc import Mapping
-from decimal import Decimal
+import re
+import time as system_time
+from bisect import bisect_right
+from collections.abc import Mapping, Sequence
+from dataclasses import dataclass
+from datetime import date, datetime, time, timedelta, timezone
+from decimal import Decimal, InvalidOperation
+from enum import Enum
from numbers import Real
+from types import MappingProxyType
from typing import Any, cast
+from uuid import UUID
+from zoneinfo import ZoneInfo, ZoneInfoNotFoundError
+
+import numpy as np
+import pandas as pd
+import pytz
+from dateutil import tz as dateutil_tz
+from dateutil.tz.tz import _ttinfo as dateutil_ttinfo
+from dateutil.zoneinfo import tzfile as dateutil_zoneinfo_tzfile
+from pydantic import BaseModel
+from pydantic_core import to_json
from superset.mcp_service.chart.schemas import ChartError
+from superset.mcp_service.utils.serialization import decode_binary
+from superset.utils.core import GenericDataType
+from superset.utils.dates import datetime_to_epoch, EPOCH
FAILED_QUERY_STATUSES = frozenset(
{"error", "failed", "stopped", "timed_out", "cancelled", "canceled"}
)
+_ERROR_KEYS = ("error", "error_message", "message", "detail")
+_MAX_ERROR_DEPTH = 32
+_MAX_ERROR_ITEMS = 256
+_MAX_SEQUENCE_ITEMS = 64
+_MAX_ERROR_PARTS = 3
+_MAX_ERROR_BYTES = 2000
+_MAX_INTEGER_DIGITS = 1000
+_MAX_QUERY_COUNT = 64
+_MAX_QUERY_COLUMNS = 4096
+_MAX_COLUMN_NAME_BYTES = 4096
+_MAX_ROW_CONTAINER_DEPTH = 32
+_MAX_ROW_CONTAINER_ITEMS = 4096
+_MAX_CACHE_STRING_BYTES = 4096
+_MAX_RESULT_ROW_COUNT = (1 << 63) - 1
-def _query_error_text(value: Any) -> str | None:
- """Convert a bounded query error payload into a useful message."""
- if value is None or value is False:
+# Chart results are routinely much larger than an MCP response should return,
but
+# legitimate exports and high-cardinality chart queries still need useful room.
+# Each query may return Superset's configured 50k ROW_LIMIT. The aggregate row
+# budget admits both legs of Big Number raw/trend and Mixed Timeseries results
+# at that limit, while the value budget admits twenty scalar columns on both
+# legs (plus their row containers). The complete compact JSON projection is
+# capped at 16 MiB, including scalar tokens, escaping, keys, and syntax.
Metadata
+# profiling has a separate row-by-column work budget in ``response_utils`` so
+# wide sparse results cannot turn bounded validation into an unbounded scan.
+# Individual source-result cell strings are capped at 64 KiB and object keys at
+# 4 KiB. Derived strings in a final Pydantic response have no per-cell cap; the
+# complete compact response remains subject to the 16 MiB aggregate budget.
+# Query metadata has its own 1 MiB aggregate budget so SQL and cache metadata
+# cannot consume the row-data allowance. Row-shaped indexnames use the row-data
+# work budget while retaining the metadata byte budget. Integer/Decimal bounds
+# prevent later hashing, uniqueness, and JSON conversion from allocating by
magnitude.
+MAX_QUERY_RESULT_ROWS = 50_000
+MAX_QUERY_RESULT_TOTAL_ROWS = 2 * MAX_QUERY_RESULT_ROWS
+MAX_QUERY_RESULT_VALUES = 2_500_000
+MAX_QUERY_RESULT_VALUE_BYTES = 16 * 1024 * 1024
+MAX_QUERY_RESULT_METADATA_BYTES = 1024 * 1024
+MAX_QUERY_RESULT_METADATA_ITEMS = 32_768
+MAX_QUERY_RESULT_WORK = MAX_QUERY_RESULT_VALUES +
MAX_QUERY_RESULT_METADATA_ITEMS
+MAX_QUERY_RESULT_STRING_BYTES = 64 * 1024
+MAX_QUERY_RESULT_KEY_BYTES = 4096
+MAX_QUERY_RESULT_INTEGER_BITS = 4096
+MAX_QUERY_RESULT_INTEGER_DIGITS = 1234
+MAX_QUERY_RESULT_DECIMAL_DIGITS = 1024
+MAX_QUERY_RESULT_DECIMAL_EXPONENT = 4096
+MAX_QUERY_RESULT_DECIMAL_STORAGE = 2048
+_BUILTIN_SCALAR_TYPES = (str, bytes, bytearray, memoryview, int, float, bool)
+_SCALAR_BASE_TYPES = (*_BUILTIN_SCALAR_TYPES, Enum)
+_SUPPORTED_COLTYPES = frozenset(GenericDataType)
+_TRUSTED_TZINFO_TYPES = (timezone, ZoneInfo)
+_DATEUTIL_TZFILE_TYPE = dateutil_tz.tzfile
+_DATEUTIL_TZOFFSET_TYPE = type(dateutil_tz.tzoffset(None, 0))
+_DATEUTIL_TZUTC_TYPE = type(dateutil_tz.UTC)
+_DATEUTIL_TZLOCAL_TYPE = type(dateutil_tz.tzlocal())
+_MAX_DATEUTIL_TRANSITIONS = 4096
+_PYTZ_FIXED_OFFSET_TYPE = type(pytz.FixedOffset(1))
+_PYTZ_UTC_TYPE = type(pytz.UTC)
+_PYTZ_NAMED_BASE_TYPES = (pytz.tzinfo.DstTzInfo, pytz.tzinfo.StaticTzInfo)
+_NUMPY_INTEGER_TYPES = frozenset(
+ type(value)
+ for value in (
+ np.int8(0),
+ np.int16(0),
+ np.int32(0),
+ np.int64(0),
+ np.uint8(0),
+ np.uint16(0),
+ np.uint32(0),
+ np.uint64(0),
+ )
+)
+_NUMPY_FLOAT_TYPES = frozenset(
+ type(value)
+ for value in (np.float16(0), np.float32(0), np.float64(0),
np.longdouble(0))
+)
+_NUMPY_EXTENDED_FLOAT_TYPES = frozenset(
+ type_
+ for type_ in _NUMPY_FLOAT_TYPES
+ if np.finfo(type_).nmant > np.finfo(np.float64).nmant
+)
+_PANDAS_NAT_TYPE = type(pd.NaT)
+_PANDAS_NA_TYPE = type(pd.NA)
+_PANDAS_PERIOD_TYPE = type(pd.Period("2000-01", freq="M"))
+_PANDAS_INTERVAL_TYPE = type(pd.Interval(0, 1))
+_EXCEL_MIN_DATE = date(1900, 1, 1)
+# Leave room for XLSX readers' millisecond rounding at the final serial day.
+_EXCEL_MAX_TIME = time(23, 59, 59, 999000)
+_EXCEL_MAX_DATETIME = datetime.combine(date.max, _EXCEL_MAX_TIME)
+# Serial day 2958465 is 9999-12-31 in Excel's 1900 date system.
+_EXCEL_MAX_DURATION = timedelta(
+ days=2958465, hours=23, minutes=59, seconds=59, milliseconds=999
+)
+
+
+@dataclass(frozen=True)
+class _ErrorText:
+ """Bounded error extraction outcome."""
+
+ text: str | None = None
+ malformed: str | None = None
+
+
+@dataclass
+class _ResultBudget:
+ """Aggregate work counters shared across all queries in one result."""
+
+ rows: int = 0
+ values: int = 0
+ json_bytes: int = 0
+ metadata_items: int = 0
+ metadata_bytes: int = 0
+
+
+def _truncate_utf8(value: str, max_bytes: int) -> str:
+ """Return bounded, replacement-decoded UTF-8 text.
+
+ Encoding even the non-truncated path is intentional: Python strings may
+ contain unpaired surrogates, while MCP/JSON responses must always be valid
+ UTF-8. Slicing by characters before encoding also prevents an
+ attacker-sized string from being encoded in full.
+ """
+ if max_bytes <= 0:
+ return ""
+ candidate = value[:max_bytes]
+ encoded = candidate.encode("utf-8", errors="replace")
+ if len(encoded) <= max_bytes and len(candidate) == len(value):
+ return encoded.decode("utf-8", errors="replace")
+ suffix = "... [truncated]"
+ suffix_bytes = suffix.encode()
+ if max_bytes <= len(suffix_bytes):
+ return suffix_bytes[:max_bytes].decode("ascii")
+ content_limit = max(0, max_bytes - len(suffix_bytes))
+ content = encoded[:content_limit].decode("utf-8", errors="ignore")
+ return content + suffix
+
+
+def _type_descriptor(value: Any, max_bytes: int) -> str | None:
+ """Describe an unsupported value without consulting its implementation."""
+ if max_bytes <= 0:
return None
- if isinstance(value, Mapping):
- for key in ("error", "error_message", "message", "detail"):
- if text := _query_error_text(value.get(key)):
- return text
+ value_type = type(value)
+ try:
+ type_name = type.__getattribute__(value_type, "__name__")
+ except (AttributeError, TypeError): # pragma: no cover - defensive
metaclass
+ type_name = "unknown"
+ if type(type_name) is not str:
+ type_name = "unknown"
+ bounded_name = _truncate_utf8(type_name, max_bytes)
+ return _truncate_utf8(f"<{bounded_name} object>", max_bytes)
+
+
+def _type_mro(value_type: type[Any]) -> tuple[type[Any], ...]:
+ """Read a concrete type's MRO without consulting its metaclass
overrides."""
+ try:
+ mro = type.__getattribute__(value_type, "__mro__")
+ except (AttributeError, TypeError): # pragma: no cover - all normal types
have MRO
+ return ()
+ return mro if type(mro) is tuple else ()
+
+
+def _mro_contains(
+ value_mro: tuple[type[Any], ...], base_types: tuple[type[Any], ...]
+) -> bool:
+ """Return whether an MRO contains a base, using identity-only
comparisons."""
+ return any(
+ base is expected_base for base in value_mro for expected_base in
base_types
+ )
+
+
+def _safe_scalar_text(value: Any, max_bytes: int) -> str | None: # noqa: C901
+ """Render a bounded scalar without invoking attacker-controlled string
code."""
+ value_type = type(value)
+ if _mro_contains(_type_mro(value_type), (Enum,)):
+ try:
+ enum_value = object.__getattribute__(value, "_value_")
+ except Exception:
+ return _type_descriptor(value, max_bytes)
+ if not any(
+ type(enum_value) is scalar_type for scalar_type in
_BUILTIN_SCALAR_TYPES
+ ):
+ return _type_descriptor(value, max_bytes)
+ return _safe_scalar_text(enum_value, max_bytes)
+ if value is None or value is False:
return None
- if isinstance(value, (list, tuple)):
- parts = [text for item in value if (text := _query_error_text(item))]
- return "; ".join(parts[:3]) or None
- text = str(value)
- return text[:2000] if text else None
+ if value_type is str:
+ return _truncate_utf8(value, max_bytes) if value else None
+ if value_type is bytes or value_type is bytearray or value_type is
memoryview:
+ try:
+ view = memoryview(value).cast("B")
+ sample = view[: max(0, max_bytes)].tobytes()
+ text = sample.decode("utf-8", errors="replace")
+ if len(view) > len(sample):
+ text += "... [truncated]"
+ return _truncate_utf8(text, max_bytes) if text else None
+ except (TypeError, ValueError):
+ return _type_descriptor(value, max_bytes)
+ if value_type is int:
+ digits = (
+ 1 if value == 0 else int((abs(value).bit_length() - 1) *
math.log10(2)) + 1
+ )
+ if digits > _MAX_INTEGER_DIGITS:
+ sign = "negative " if value < 0 else ""
+ return _truncate_utf8(
+ f"<{sign}integer with approximately {digits} decimal digits>",
+ max_bytes,
+ )
+ return _truncate_utf8(str(value), max_bytes)
+ if value_type is bool or value_type is float:
+ return _truncate_utf8(str(value), max_bytes)
+ return _type_descriptor(value, max_bytes)
+
+
+def _query_error_text(value: Any) -> _ErrorText: # noqa: C901
+ """Iteratively extract actionable text from an untrusted error payload.
+
+ Chart backends and engine adapters can return arbitrary nested error
shapes.
+ Depth, visited-item, sequence-width, and output-byte limits keep validation
+ deterministic even for cycles, repeated containers, and adversarial values.
+ """
+ stack: list[tuple[Any, int]] = [(value, 0)]
+ seen: set[int] = set()
+ parts: list[str] = []
+ visited = 0
+ used_bytes = 0
+
+ while stack and len(parts) < _MAX_ERROR_PARTS:
+ item, depth = stack.pop()
+ visited += 1
+ if visited > _MAX_ERROR_ITEMS:
+ return _ErrorText(malformed="error payload exceeds the item limit")
+ if depth > _MAX_ERROR_DEPTH:
+ return _ErrorText(malformed="error payload exceeds the depth
limit")
+
+ # ChartDataCommand envelopes cross a JSON boundary. Only exact JSON
+ # containers are trusted here: ABC/isinstance checks can consult a
+ # spoofed ``__class__``, and subclass get/contains/iter/len hooks are
+ # attacker-controlled. Exact dict/list operations below are builtin and
+ # non-overridable.
+ is_mapping = type(item) is dict
+ is_sequence = type(item) is list
+ item_mro = _type_mro(type(item))
+ if not (is_mapping or is_sequence) and (
+ _mro_contains(item_mro, (dict, list, Mapping, Sequence))
+ and not _mro_contains(item_mro, _SCALAR_BASE_TYPES)
+ ):
+ return _ErrorText(
+ malformed="error payload contains an unsupported container
type"
+ )
+ if is_mapping or is_sequence:
+ identity = id(item)
+ if identity in seen:
+ return _ErrorText(
+ malformed="error payload contains repeated or cyclic
containers"
+ )
+ seen.add(identity)
+
+ if is_mapping:
+ children: list[Any] = []
+ for key in _ERROR_KEYS:
+ if dict.__contains__(item, key):
+ children.append(dict.__getitem__(item, key))
+ if not children:
+ if dict.__len__(item):
+ return _ErrorText(
+ malformed=(
+ "error payload object has no recognized message
field"
+ )
+ )
+ stack.extend((child, depth + 1) for child in reversed(children))
+ continue
+
+ if is_sequence:
+ width = list.__len__(item)
+ if width > _MAX_SEQUENCE_ITEMS:
+ return _ErrorText(malformed="error payload exceeds the width
limit")
+ children = [list.__getitem__(item, index) for index in
range(width)]
+ stack.extend((child, depth + 1) for child in reversed(children))
+ continue
+
+ remaining = _MAX_ERROR_BYTES - used_bytes - (2 if parts else 0)
+ text = _safe_scalar_text(item, remaining)
+ if text:
+ parts.append(text)
+ used_bytes += len(text.encode("utf-8", errors="replace")) + (
+ 2 if len(parts) > 1 else 0
+ )
+ return _ErrorText(text="; ".join(parts) or None)
-def _failure_for_query_payload(
- payload: Mapping[str, Any], label: str
+
+def _failure_for_query_payload( # noqa: C901
+ payload: dict[str, Any], label: str
) -> ChartError | None:
"""Extract one failure from a top-level or per-query payload."""
+ malformed: str | None = None
for key in ("error", "errors", "error_message"):
- if message := _query_error_text(payload.get(key)):
+ extracted = _query_error_text(dict.get(payload, key))
+ if extracted.malformed:
+ malformed = malformed or extracted.malformed
+ continue
+ if message := extracted.text:
return ChartError(
error=f"{label} failed: {message}", error_type="QueryError"
)
- raw_status = payload.get("status")
- status = str(getattr(raw_status, "value", raw_status) or "")
+ raw_status = dict.get(payload, "status")
+ status = _safe_scalar_text(raw_status, 200) or ""
normalized_status = status.strip().casefold().replace("-", "_").replace("
", "_")
if normalized_status in FAILED_QUERY_STATUSES:
- message = (
- _query_error_text(payload.get("message"))
- or _query_error_text(payload.get("error_message"))
- or normalized_status
- )
+ extracted = _query_error_text(dict.get(payload, "message"))
+ if extracted.malformed:
+ malformed = malformed or extracted.malformed
+ fallback = _query_error_text(dict.get(payload, "error_message"))
+ if fallback.malformed:
+ malformed = malformed or fallback.malformed
+ if malformed and not (extracted.text or fallback.text):
+ return _malformed_result(malformed)
+ message = extracted.text or fallback.text or normalized_status
return ChartError(error=f"{label} failed: {message}",
error_type="QueryError")
- if payload.get("success") is False:
- message = _query_error_text(payload.get("message")) or "request failed"
+ if dict.get(payload, "success") is False:
+ extracted = _query_error_text(dict.get(payload, "message"))
+ if extracted.malformed:
+ malformed = malformed or extracted.malformed
+ if malformed and not extracted.text:
+ return _malformed_result(malformed)
+ message = extracted.text or "request failed"
return ChartError(error=f"{label} failed: {message}",
error_type="QueryError")
if (
raw_status is None
- and "data" not in payload
- and "queries" not in payload
- and (message := _query_error_text(payload.get("message")))
+ and "data" not in dict.keys(payload)
+ and "queries" not in dict.keys(payload)
):
- return ChartError(error=f"{label} failed: {message}",
error_type="QueryError")
+ extracted = _query_error_text(dict.get(payload, "message"))
+ if extracted.malformed:
+ malformed = malformed or extracted.malformed
+ if extracted.text:
+ return ChartError(
+ error=f"{label} failed: {extracted.text}",
error_type="QueryError"
+ )
+ if malformed:
+ return _malformed_result(malformed)
return None
-def query_result_failure(result: Any) -> ChartError | None:
- """Return a structured failure embedded in a ChartDataCommand payload.
+def _malformed_result(message: str) -> ChartError:
+ """Build a stable error for an invalid ChartDataCommand envelope."""
+ return ChartError(
+ error=f"Malformed chart query result: {message}",
+ error_type="MalformedQueryResult",
+ )
- ChartDataCommand can return an HTTP-successful envelope whose top level or
- any query reports a failure. Every query is inspected before callers accept
- data from the result. Successful statuses may carry informational messages,
- so ``message`` alone is not treated as an error.
+
+def bounded_result_row_count(value: Any) -> int | None:
+ """Return one exact bounded row count, rejecting coercive lookalikes."""
+ if value is None:
+ return None
+ if type(value) is int:
+ count = value
+ elif type(value) is float and math.isfinite(value) and value.is_integer():
+ count = int(value)
+ else:
+ raise ValueError("must be a finite non-negative integral number")
+ if count < 0:
+ raise ValueError("must be non-negative")
+ if count > _MAX_RESULT_ROW_COUNT:
+ raise ValueError("exceeds the supported bound")
+ return count
+
+
+def _bounded_utf8_length(value: str, max_bytes: int) -> int | None:
+ """Return an exact UTF-8 size without encoding attacker-sized text."""
+ if str.__len__(value) > max_bytes:
+ return None
+ try:
+ encoded = str.encode(value, "utf-8", errors="strict")
+ except UnicodeEncodeError:
+ return None
+ size = bytes.__len__(encoded)
+ return size if size <= max_bytes else None
+
+
+def _json_string_size(value: str, max_bytes: int) -> int | None:
+ """Return the exact UTF-8 size of a JSON string without serializing it."""
+ raw_size = _bounded_utf8_length(value, max_bytes)
+ if raw_size is None:
+ return None
+ escaped_size = raw_size + 2 # surrounding quotes
+ for character in value:
+ codepoint = ord(character)
+ if character in {'"', "\\"} or character in {"\b", "\t", "\n", "\f",
"\r"}:
+ escaped_size += 1
+ elif codepoint < 0x20:
+ # Other JSON control characters use a six-byte ``\\u00xx`` escape.
+ escaped_size += 5
+ return escaped_size
+
+
+def _integer_json_size(value: int) -> int:
+ """Return an exact integer JSON size without creating its decimal
string."""
+ magnitude = -value if value < 0 else value
+ if magnitude == 0:
+ digits = 1
+ else:
+ bits = int.bit_length(magnitude)
+ # This fixed-point log10(2) estimate is at most one digit low. Refine
it
+ # with one bounded integer comparison rather than rendering the value.
+ digits = ((bits - 1) * 30103) // 100000 + 1
+ if magnitude >= 10**digits:
+ digits += 1
+ return digits + (value < 0)
+
+
+def _container_json_syntax_size(item_count: int, *, mapping: bool) -> int:
+ """Return braces/brackets, separators, and mapping-colon byte cost."""
+ if item_count == 0:
+ return 2
+ return 2 + item_count - 1 + (item_count if mapping else 0)
+
+
+def _trusted_timedelta_text(value: timedelta) -> str:
+ """Render an exact timedelta with Pydantic's stable ISO-8601 spelling."""
+ total_microseconds = (
+ value.days * 86_400 + value.seconds
+ ) * 1_000_000 + value.microseconds
+ sign = "-" if total_microseconds < 0 else ""
+ remaining = abs(total_microseconds)
+ days, remaining = divmod(remaining, 86_400 * 1_000_000)
+ years, days = divmod(days, 365)
+ hours, remaining = divmod(remaining, 3_600 * 1_000_000)
+ minutes, remaining = divmod(remaining, 60 * 1_000_000)
+ seconds, microseconds = divmod(remaining, 1_000_000)
+
+ date_parts = [f"{years}Y" if years else "", f"{days}D" if days else ""]
+ time_parts = [f"{hours}H" if hours else "", f"{minutes}M" if minutes else
""]
+ if microseconds:
+ fraction = f"{microseconds:06d}".rstrip("0")
+ time_parts.append(f"{seconds}.{fraction}S")
+ elif seconds:
+ time_parts.append(f"{seconds}S")
+
+ date_text = "".join(date_parts)
+ time_text = "".join(time_parts)
+ if not date_text and not time_text:
+ time_text = "0S"
+ return f"{sign}P{date_text}{'T' if time_text else ''}{time_text}"
+
+
+def _chart_data_builtin_timedelta_text(value: timedelta) -> str:
+ """Reproduce ``format_timedelta`` without comparison or string hooks."""
+ total_microseconds = (
+ value.days * 86_400 + value.seconds
+ ) * 1_000_000 + value.microseconds
+ sign = "-" if total_microseconds < 0 else ""
+ remaining = abs(total_microseconds)
+ days, remaining = divmod(remaining, 86_400 * 1_000_000)
+ hours, remaining = divmod(remaining, 3_600 * 1_000_000)
+ minutes, remaining = divmod(remaining, 60 * 1_000_000)
+ seconds, microseconds = divmod(remaining, 1_000_000)
+ day_text = f"{days} {'day' if days == 1 else 'days'}, " if days else ""
+ fraction = f".{microseconds:06d}" if microseconds else ""
+ return f"{sign}{day_text}{hours}:{minutes:02d}:{seconds:02d}{fraction}"
+
+
+def _chart_data_pandas_timedelta_text(value: pd.Timedelta) -> str:
+ """Reproduce Chart Data ``format_timedelta`` output from exact fields."""
+ total_nanoseconds = (
+ (
+ object.__getattribute__(value, "days") * 86_400
+ + object.__getattribute__(value, "seconds")
+ )
+ * 1_000_000
+ + object.__getattribute__(value, "microseconds")
+ ) * 1_000 + object.__getattribute__(value, "nanoseconds")
+ sign = "-" if total_nanoseconds < 0 else ""
+ remaining = abs(total_nanoseconds)
+ days, remaining = divmod(remaining, 86_400 * 1_000_000_000)
+ hours, remaining = divmod(remaining, 3_600 * 1_000_000_000)
+ minutes, remaining = divmod(remaining, 60 * 1_000_000_000)
+ seconds, nanoseconds = divmod(remaining, 1_000_000_000)
+ if nanoseconds % 1_000:
+ fraction = f".{nanoseconds:09d}"
+ elif nanoseconds:
+ fraction = f".{nanoseconds // 1_000:06d}"
+ else:
+ fraction = ""
+ return f"{sign}{days} days
{hours:02d}:{minutes:02d}:{seconds:02d}{fraction}"
+
+
+def _normalized_scalar_json_size( # noqa: C901
+ value: Any, *, max_string_bytes: int = MAX_QUERY_RESULT_STRING_BYTES
+) -> int:
+ """Return a conservative encoded size for one normalized exact scalar."""
+ value_type = type(value)
+ if value is None:
+ return 4
+ if value_type is bool:
+ return 4 if value else 5
+ if value_type is str:
+ size = _json_string_size(value, max_string_bytes)
+ assert size is not None # scalar normalization already bounded the
string
+ return size
+ if value_type is int:
+ return _integer_json_size(value)
+ if value_type is float:
+ if not math.isfinite(value):
+ # Raw Gauge exports retain these markers; JSON responses use null.
+ return 4
+ # Exact builtin repr is hook-free, bounded to a shortest-round-trip
+ # spelling, and avoids pessimistically charging 24 bytes for values
+ # such as 0.0 across ordinary large numeric datasets.
+ return len(float.__repr__(value))
+ if value_type is Decimal:
+ # Decimal storage, coefficient digits, and exponent are bounded before
+ # this point. Its canonical spelling is therefore itself bounded, and
+ # Pydantic serializes Decimal values as JSON strings.
+ text = Decimal.__str__(value)
+ size = _json_string_size(text, MAX_QUERY_RESULT_STRING_BYTES)
+ assert size is not None
+ return size
+ if value_type is datetime:
+ return 40
+ if value_type is date:
+ text = date.isoformat(value)
+ elif value_type is time:
+ return 32
+ elif value_type is timedelta:
+ text = _trusted_timedelta_text(value)
+ elif value_type is UUID:
+ text = UUID.__str__(value)
+ else:
+ raise AssertionError(f"unaccounted normalized scalar: {value_type!r}")
+ size = _json_string_size(text, MAX_QUERY_RESULT_STRING_BYTES)
+ assert size is not None
+ return size
+
+
+def _pydantic_scalar_json_size(value: Any) -> int:
+ """Return the exact Pydantic wire size for a normalized scalar.
+
+ Source-result accounting deliberately retains its existing conservative
+ scalar rules. Final response projections, however, must match
+ pydantic-core's JSON number spelling: for example, it emits ``0.00001`` for
+ ``1e-5`` and ``1e-6`` for ``1e-6`` rather than Python's repr spellings.
"""
- if not isinstance(result, Mapping):
+ if type(value) is float:
+ return len(to_json(value))
+ return _normalized_scalar_json_size(value)
+
+
+def _charge_json_bytes(
+ budget: _ResultBudget, size: int, *, metadata: bool = False
+) -> str | None:
+ """Charge aggregate response bytes and the independent metadata
allowance."""
+ budget.json_bytes += size
+ if budget.json_bytes > MAX_QUERY_RESULT_VALUE_BYTES:
+ return "exceeds the total JSON-encoded byte limit"
+ if metadata:
+ budget.metadata_bytes += size
+ if budget.metadata_bytes > MAX_QUERY_RESULT_METADATA_BYTES:
+ return "metadata exceeds the total JSON-encoded byte limit"
+ return None
+
+
+def _integer_failure(value: int) -> str | None:
+ """Validate exact integer magnitude before decimal rendering or hashing."""
+ bits = int.bit_length(value)
+ if bits > MAX_QUERY_RESULT_INTEGER_BITS:
+ return "contains an integer exceeding the bit-length limit"
+ digits = 1 if bits == 0 else ((bits - 1) * 30103) // 100000 + 1
+ if digits > MAX_QUERY_RESULT_INTEGER_DIGITS:
+ return "contains an integer exceeding the digit limit"
+ return None
+
+
+def _decimal_failure(value: Decimal) -> str | None:
+ """Validate exact Decimal storage, finiteness, digits, and exponent."""
+ if Decimal.__sizeof__(value) > MAX_QUERY_RESULT_DECIMAL_STORAGE:
+ return "contains a Decimal exceeding the storage limit"
+ if not Decimal.is_finite(value):
+ return "contains a non-finite Decimal"
+ parts = Decimal.as_tuple(value)
+ if tuple.__len__(parts.digits) > MAX_QUERY_RESULT_DECIMAL_DIGITS:
+ return "contains a Decimal exceeding the digit limit"
+ exponent = parts.exponent
+ if type(exponent) is not int or abs(exponent) >
MAX_QUERY_RESULT_DECIMAL_EXPONENT:
+ return "contains a Decimal exceeding the exponent limit"
+ return None
+
+
+def _exact_object_namespace(value: Any) -> dict[str, Any] | None:
+ """Read an object's concrete storage without descriptor dispatch."""
+ try:
+ namespace = object.__getattribute__(value, "__dict__")
+ except (AttributeError, TypeError):
+ return None
+ return namespace if type(namespace) is dict else None
+
+
+def _dateutil_timezone_name_without_hooks(tzinfo: Any) -> str | None:
+ """Read a dateutil tzfile's IANA name from exact internal storage."""
+ tzinfo_type = type(tzinfo)
+ if tzinfo_type not in {_DATEUTIL_TZFILE_TYPE, dateutil_zoneinfo_tzfile}:
+ return None
+ namespace = _exact_object_namespace(tzinfo)
+ if namespace is None:
+ return None
+ filename = dict.get(namespace, "_filename")
+ if type(filename) is not str or _bounded_utf8_length(filename, 4096) is
None:
+ return None
+ if tzinfo_type is dateutil_zoneinfo_tzfile:
+ name = filename
+ else:
+ marker = "/zoneinfo/"
+ marker_offset = str.find(filename, marker)
+ if marker_offset >= 0:
+ name = str.__getitem__(filename, slice(marker_offset +
len(marker), None))
+ elif not str.startswith(filename, "/") and str.find(filename, "\\") <
0:
+ name = filename
+ else:
+ return None
+ parts = str.split(name, "/")
+ if not parts or any(part in {"", ".", ".."} for part in parts):
+ return None
+ return name if _bounded_utf8_length(name, 256) is not None else None
+
+
+def _dateutil_ttinfo_without_hooks(
+ value: Any,
+) -> tuple[int, timedelta] | None:
+ """Read one exact dateutil transition record without user-hook dispatch."""
+ if type(value) is not dateutil_ttinfo:
+ return None
+ try:
+ offset = object.__getattribute__(value, "offset")
+ delta = object.__getattribute__(value, "delta")
+ except (AttributeError, TypeError):
+ return None
+ if type(offset) is not int or type(delta) is not timedelta:
+ return None
+ try:
+ if delta != timedelta(seconds=offset):
+ return None
+ except OverflowError:
+ return None
+ return offset, delta
+
+
+def _dateutil_named_offset_without_hooks( # noqa: C901
+ value: datetime, tzinfo: Any
+) -> timezone | None:
+ """Recover the offset selected by an exact dateutil named timezone.
+
+ A dateutil tzfile's finite transition table is its wire-semantic source of
+ truth. Reinterpreting its wall time through a system ``ZoneInfo`` database
+ changes negative-DST folds, nonexistent times, and dates after the final
+ transition. This mirrors dateutil's transition selection using only exact
+ builtin containers and its exact trusted transition-record type.
+ """
+ if _dateutil_timezone_name_without_hooks(tzinfo) is None:
+ return None
+ namespace = _exact_object_namespace(tzinfo)
+ if namespace is None:
+ return None
+ transitions = dict.get(namespace, "_trans_list")
+ transition_info = dict.get(namespace, "_trans_idx")
+ standard_info = dict.get(namespace, "_ttinfo_std")
+ before_info = dict.get(namespace, "_ttinfo_before")
+ transition_count = tuple.__len__(transitions) if type(transitions) is
tuple else 0
+ if (
+ type(transitions) is not tuple
+ or type(transition_info) is not tuple
+ or tuple.__len__(transitions) != tuple.__len__(transition_info)
+ or transition_count > _MAX_DATEUTIL_TRANSITIONS
+ or _dateutil_ttinfo_without_hooks(standard_info) is None
+ or (
+ transition_count > 0 and
_dateutil_ttinfo_without_hooks(before_info) is None
+ )
+ ):
+ return None
+
+ previous: int | None = None
+ for transition in transitions:
+ if (
+ type(transition) is not int
+ or int.bit_length(transition) > 63
+ or (previous is not None and transition < previous)
+ ):
+ return None
+ previous = transition
+ if any(_dateutil_ttinfo_without_hooks(info) is None for info in
transition_info):
+ return None
+
+ naive = datetime(
+ value.year,
+ value.month,
+ value.day,
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ )
+ try:
+ timestamp = (naive - EPOCH).total_seconds()
+ except (OverflowError, TypeError, ValueError):
+ return None
+
+ index: int | None = (
+ bisect_right(transitions, timestamp) - 1 if transition_count else None
+ )
+
+ def info_at(selected: int | None) -> Any:
+ if selected is None or selected + 1 >= transition_count:
+ return standard_info
+ if selected < 0:
+ return before_info
+ return tuple.__getitem__(transition_info, selected)
+
+ if index is not None and index != 0:
+ current = _dateutil_ttinfo_without_hooks(info_at(index))
+ prior = _dateutil_ttinfo_without_hooks(info_at(index - 1))
+ if current is None or prior is None:
+ return None
+ offset_delta = prior[0] - current[0]
+ transition = tuple.__getitem__(transitions, index)
+ is_ambiguous = timestamp < transition + offset_delta
+ index -= int(not value.fold and is_ambiguous)
+
+ selected = _dateutil_ttinfo_without_hooks(info_at(index))
+ if selected is None:
+ return None
+ try:
+ return timezone(selected[1])
+ except ValueError:
+ return None
+
+
+def _pytz_timezone_name_without_hooks(tzinfo: Any) -> str | None:
+ """Read and verify one generated pytz named-zone implementation."""
+ value_type = type(tzinfo)
+ if not _mro_contains(_type_mro(value_type), _PYTZ_NAMED_BASE_TYPES):
+ return None
+ try:
+ namespace = type.__getattribute__(value_type, "__dict__")
+ except (AttributeError, TypeError):
+ return None
+ if type(namespace) is not MappingProxyType:
+ return None
+ zone = namespace.get("zone")
+ if type(zone) is not str or _bounded_utf8_length(zone, 256) is None:
+ return None
+ try:
+ canonical = pytz.timezone(zone)
+ except (KeyError, ValueError):
+ return None
+ # A user subclass can inherit pytz's base and spoof ``zone``. Only the
+ # concrete class generated and cached by pytz for that name is trusted.
+ return zone if type(canonical) is value_type else None
+
+
+def _fixed_offset_without_hooks(tzinfo: Any) -> timezone | None:
+ """Reconstruct trusted dateutil/pytz fixed offsets from exact storage."""
+ if type(tzinfo) not in {_DATEUTIL_TZOFFSET_TYPE, _PYTZ_FIXED_OFFSET_TYPE}:
+ return None
+ namespace = _exact_object_namespace(tzinfo)
+ if namespace is None:
+ return None
+ offset = dict.get(namespace, "_offset")
+ if type(offset) is not timedelta:
+ return None
+ try:
+ return timezone(offset)
+ except ValueError:
+ return None
+
+
+def _pytz_named_offset_without_hooks(tzinfo: Any) -> timezone | None:
+ """Return a localized pytz instance's stored offset without its hooks."""
+ if _pytz_timezone_name_without_hooks(tzinfo) is None:
+ return None
+ namespace = _exact_object_namespace(tzinfo)
+ if namespace is None:
+ return None
+ offset = dict.get(namespace, "_utcoffset")
+ if type(offset) is not timedelta:
+ return None
+ try:
+ return timezone(offset)
+ except ValueError:
+ return None
+
+
+def _dateutil_local_offset_without_hooks(
+ value: datetime, tzinfo: Any
+) -> timezone | None:
+ """Select an exact dateutil-local offset using builtin system time data."""
+ if type(tzinfo) is not _DATEUTIL_TZLOCAL_TYPE:
+ return None
+ namespace = _exact_object_namespace(tzinfo)
+ if namespace is None:
+ return None
+ standard_offset = dict.get(namespace, "_std_offset")
+ daylight_offset = dict.get(namespace, "_dst_offset")
+ has_daylight = dict.get(namespace, "_hasdst")
+ if (
+ type(standard_offset) is not timedelta
+ or type(daylight_offset) is not timedelta
+ or type(has_daylight) is not bool
+ ):
+ return None
+ selected_offset = standard_offset
+ if has_daylight:
+ epoch = datetime(1970, 1, 1)
+ naive = datetime(
+ value.year,
+ value.month,
+ value.day,
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ )
+ timestamp = (naive - epoch).total_seconds()
+ try:
+ is_daylight = bool(
+ system_time.localtime(timestamp +
system_time.timezone).tm_isdst
+ )
+ daylight_saved = daylight_offset - standard_offset
+ previous_is_daylight = bool(
+ system_time.localtime(
+ timestamp
+ - timedelta.total_seconds(daylight_saved)
+ + system_time.timezone
+ ).tm_isdst
+ )
+ except (OverflowError, OSError, ValueError):
+ return None
+ is_ambiguous = not is_daylight and is_daylight != previous_is_daylight
+ if is_ambiguous:
+ is_daylight = not bool(value.fold)
+ selected_offset = daylight_offset if is_daylight else standard_offset
+ try:
+ return timezone(selected_offset)
+ except ValueError:
+ return None
+
+
+def _canonical_timezone(tzinfo: Any) -> timezone | ZoneInfo | None:
+ """Return an exact trusted timezone without invoking the source's
methods."""
+ if any(type(tzinfo) is type_ for type_ in _TRUSTED_TZINFO_TYPES):
+ return tzinfo
+ if type(tzinfo) in {_DATEUTIL_TZUTC_TYPE, _PYTZ_UTC_TYPE}:
+ return timezone.utc
+ if fixed_offset := _fixed_offset_without_hooks(tzinfo):
+ return fixed_offset
+ zone_name = _dateutil_timezone_name_without_hooks(
+ tzinfo
+ ) or _pytz_timezone_name_without_hooks(tzinfo)
+ if zone_name:
+ try:
+ return ZoneInfo(zone_name)
+ except (KeyError, ValueError, ZoneInfoNotFoundError):
+ return None
+ return None
+
+
+def _timestamp_offset_without_hooks(value: pd.Timestamp) -> timezone | None:
+ """Recover a timestamp's stored wall-clock offset without timezone
hooks."""
+ unit_multipliers = {"s": 1_000_000_000, "ms": 1_000_000, "us": 1_000,
"ns": 1}
+ multiplier = unit_multipliers.get(value.unit)
+ if multiplier is None:
+ return None
+ try:
+ instant_ns = int(value.asm8.view("i8")) * multiplier
+ epoch_ordinal = date.toordinal(date(1970, 1, 1))
+ wall_ns = (
+ (
+ (datetime.toordinal(value) - epoch_ordinal) * 86_400
+ + value.hour * 3600
+ + value.minute * 60
+ + value.second
+ )
+ * 1_000_000_000
+ + value.microsecond * 1000
+ + value.nanosecond
+ )
+ offset_ns = wall_ns - instant_ns
+ if offset_ns % 1000:
+ return None
+ return timezone(timedelta(microseconds=offset_ns // 1000))
+ except (OverflowError, TypeError, ValueError):
return None
+
+def _trusted_datetime_value(
+ value: datetime,
+) -> tuple[datetime | None, str | None]:
+ """Return an exact datetime rebuilt with only trusted timezone types."""
+ tzinfo = value.tzinfo
+ canonical_value = value
+ if tzinfo is not None and not any(
+ type(tzinfo) is trusted for trusted in _TRUSTED_TZINFO_TYPES
+ ):
+ canonical_tz: timezone | ZoneInfo | None
+ if _dateutil_timezone_name_without_hooks(tzinfo) is not None:
+ # A recognized dateutil tzfile must use its own finite transition
+ # table. Falling through to ZoneInfo would silently reinterpret a
+ # source-selected gap/fold or post-table wall time.
+ canonical_tz = _dateutil_named_offset_without_hooks(value, tzinfo)
+ else:
+ canonical_tz = (
+ _pytz_named_offset_without_hooks(tzinfo)
+ or _dateutil_local_offset_without_hooks(value, tzinfo)
+ or _canonical_timezone(tzinfo)
+ )
+ if canonical_tz is None:
+ return None, "contains a datetime with an unsupported timezone"
+ canonical_value = datetime(
+ value.year,
+ value.month,
+ value.day,
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ tzinfo=canonical_tz,
+ fold=value.fold,
+ )
+ try:
+ # Exercise builtin validation without dispatching through a source
+ # timezone after the reconstruction above.
+ datetime.isoformat(canonical_value)
+ except (OverflowError, TypeError, ValueError):
+ return None, "contains an invalid datetime"
+ return canonical_value, None
+
+
+def _trusted_datetime_text(value: datetime) -> tuple[str | None, str | None]:
+ """Serialize an exact Python datetime through only trusted timezone
types."""
+ canonical_value, reason = _trusted_datetime_value(value)
+ if reason is not None or canonical_value is None:
+ return None, reason or "contains an invalid datetime"
+ return datetime.isoformat(canonical_value), None
+
+
+def _trusted_time_text(value: time) -> tuple[str | None, str | None]:
+ """Serialize an exact Python time through only trusted timezone types."""
+ tzinfo = value.tzinfo
+ canonical_value = value
+ if tzinfo is not None and not any(
+ type(tzinfo) is trusted for trusted in _TRUSTED_TZINFO_TYPES
+ ):
+ canonical_tz = _canonical_timezone(tzinfo)
+ if canonical_tz is None and type(tzinfo) is _DATEUTIL_TZLOCAL_TYPE:
+ namespace = _exact_object_namespace(tzinfo)
+ if namespace is None or type(dict.get(namespace, "_hasdst")) is
not bool:
+ return None, "contains a time with an unsupported timezone"
+ if dict.get(namespace, "_hasdst"):
+ canonical_tz = None
+ else:
+ standard_offset = dict.get(namespace, "_std_offset")
+ if type(standard_offset) is not timedelta:
+ return None, "contains a time with an unsupported timezone"
+ try:
+ canonical_tz = timezone(standard_offset)
+ except ValueError:
+ return None, "contains a time with an unsupported timezone"
+ elif canonical_tz is None:
+ return None, "contains a time with an unsupported timezone"
+ canonical_value = time(
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ tzinfo=canonical_tz,
+ fold=value.fold,
+ )
+ try:
+ return time.isoformat(canonical_value), None
+ except (OverflowError, TypeError, ValueError):
+ return None, "contains an invalid time"
+
+
+def _trusted_timestamp_value(
+ value: pd.Timestamp,
+) -> tuple[pd.Timestamp | None, str | None]:
+ """Return a timestamp rebuilt with only trusted timezone
implementations."""
+ tzinfo = value.tzinfo
+ try:
+ if tzinfo is not None and not any(
+ type(tzinfo) is trusted for trusted in _TRUSTED_TZINFO_TYPES
+ ):
+ if (
+ _canonical_timezone(tzinfo) is None
+ and type(tzinfo) is not _DATEUTIL_TZLOCAL_TYPE
+ ):
+ return (
+ None,
+ "contains a pandas timestamp with an unsupported timezone",
+ )
+ if (canonical_tz := _timestamp_offset_without_hooks(value)) is
None:
+ return None, "contains an invalid pandas timestamp"
+ # Rebuild from the stored instant and resolution. No method on the
+ # original pytz/dateutil object is called, and the recovered fixed
+ # offset preserves the timestamp's selected fold.
+ raw_value = value.asm8.view("i8")
+ value = pd.Timestamp(raw_value, unit=value.unit,
tz="UTC").tz_convert(
+ canonical_tz
+ )
+ # Validate the retained resolution and selected UTC offset.
+ pd.Timestamp.isoformat(value)
+ except (KeyError, OverflowError, TypeError, ValueError):
+ return None, "contains an invalid pandas timestamp"
+ return value, None
+
+
+def _trusted_timestamp_text(value: pd.Timestamp) -> tuple[str | None, str |
None]:
+ """Convert an exact pandas timestamp to its canonical JSON
representation."""
+ canonical_value, reason = _trusted_timestamp_value(value)
+ if reason is not None or canonical_value is None:
+ return None, reason or "contains an invalid pandas timestamp"
+ # ISO output preserves nanoseconds and the UTC offset selected by fold.
+ text = pd.Timestamp.isoformat(canonical_value)
+ if _bounded_utf8_length(text, MAX_QUERY_RESULT_STRING_BYTES) is None:
+ return None, "contains an oversized pandas timestamp"
+ return text, None
+
+
+def _normalize_trusted_scalar( # noqa: C901
+ value: Any, *, max_string_bytes: int = MAX_QUERY_RESULT_STRING_BYTES
+) -> tuple[Any, str | None]:
+ """Normalize one exact trusted pandas/NumPy scalar or validate a builtin.
+
+ Type identity is checked before every conversion. This deliberately does
not
+ accept subclasses or generic ``np.generic``/pandas extension objects, whose
+ conversion hooks are outside the trusted ChartData materialization
contract.
+ """
+ value_type = type(value)
+ enum_seen: set[int] = set()
+ while _mro_contains(_type_mro(value_type), (Enum,)):
+ identity = id(value)
+ if identity in enum_seen or len(enum_seen) >= _MAX_ROW_CONTAINER_DEPTH:
+ return None, "contains a recursive enum"
+ enum_seen.add(identity)
+ try:
+ value = object.__getattribute__(value, "_value_")
+ except Exception:
+ return None, "contains an unsupported enum"
+ value_type = type(value)
+
+ if value is None or value_type is bool:
+ return value, None
+ if value_type is str:
+ size = _bounded_utf8_length(value, max_string_bytes)
+ return (
+ (value, None)
+ if size is not None
+ else (
+ None,
+ "contains an invalid or oversized string",
+ )
+ )
+ if value_type is int:
+ return value, _integer_failure(value)
+ if value_type is float:
+ if math.isnan(value):
+ return None, None
+ if math.isinf(value):
+ return None, "contains a non-finite number"
+ return value, None
+ if value_type is Decimal:
+ return value, _decimal_failure(value)
+
+ if value_type is datetime:
+ return _trusted_datetime_text(value)
+ if value_type is time:
+ return _trusted_time_text(value)
+ if value_type is date:
+ return date.isoformat(value), None
+ if value_type is timedelta:
+ return _trusted_timedelta_text(value), None
+ if value_type is UUID:
+ return UUID.__str__(value), None
+
+ if value_type is _PANDAS_NAT_TYPE or value_type is _PANDAS_NA_TYPE:
+ return None, None
+ if value_type is pd.Timestamp:
+ return _trusted_timestamp_text(value)
+ if value_type is pd.Timedelta:
+ if pd.isna(value):
+ return None, None
+ text = pd.Timedelta.isoformat(value)
+ if _bounded_utf8_length(text, MAX_QUERY_RESULT_STRING_BYTES) is None:
+ return None, "contains an oversized pandas timedelta"
+ return text, None
+ if value_type is _PANDAS_PERIOD_TYPE or value_type is
_PANDAS_INTERVAL_TYPE:
+ # The concrete extension scalar implementations are trusted, unlike an
+ # arbitrary subclass's ``__str__`` implementation.
+ text = str(value)
+ if _bounded_utf8_length(text, MAX_QUERY_RESULT_STRING_BYTES) is None:
+ return None, "contains an oversized pandas scalar"
+ return text, None
+
+ if any(value_type is type_ for type_ in _NUMPY_INTEGER_TYPES):
+ normalized_integer = int(value)
+ return normalized_integer, _integer_failure(normalized_integer)
+ if any(value_type is type_ for type_ in _NUMPY_EXTENDED_FLOAT_TYPES):
+ if np.isnan(value):
+ return None, None
+ if not np.isfinite(value):
+ return None, "contains a non-finite NumPy number"
+ # JSON has no extended floating-point type. Preserve the trusted scalar
+ # as a round-trippable decimal string instead of narrowing to binary64.
+ return np.format_float_scientific(value, unique=True, trim="-"), None
+ if any(value_type is type_ for type_ in _NUMPY_FLOAT_TYPES):
+ normalized_float = float(value)
+ if math.isnan(normalized_float):
+ return None, None
+ if math.isinf(normalized_float):
+ return None, "contains a non-finite NumPy number"
+ return normalized_float, None
+ if value_type is np.bool_:
+ return bool(value), None
+ if value_type is np.str_:
+ text = str(value)
+ if _bounded_utf8_length(text, MAX_QUERY_RESULT_STRING_BYTES) is None:
+ return None, "contains an invalid or oversized NumPy string"
+ return text, None
+ if value_type is np.datetime64:
+ if np.isnat(value):
+ return None, None
+ try:
+ timestamp = pd.Timestamp(value)
+ except (OverflowError, TypeError, ValueError):
+ return None, "contains an invalid NumPy datetime"
+ return _trusted_timestamp_text(timestamp)
+ if value_type is np.timedelta64:
+ if np.isnat(value):
+ return None, None
+ try:
+ delta = pd.Timedelta(value)
+ text = pd.Timedelta.isoformat(delta)
+ except (OverflowError, TypeError, ValueError):
+ return None, "contains an invalid NumPy timedelta"
+ if _bounded_utf8_length(text, MAX_QUERY_RESULT_STRING_BYTES) is None:
+ return None, "contains an oversized NumPy timedelta"
+ return text, None
+
+ if value_type is bytes or value_type is bytearray or value_type is
memoryview:
+ # Exact binary column values follow the documented serialization
+ # contract: UTF-8 text, otherwise a ``base64:``-prefixed string.
+ try:
+ raw_size = value.nbytes if value_type is memoryview else len(value)
+ except (TypeError, ValueError):
+ return None, "contains an unreadable binary value"
+ if raw_size > max_string_bytes:
+ return None, "contains an oversized binary value"
+ try:
+ text = decode_binary(value)
+ except (TypeError, ValueError):
+ return None, "contains an unreadable binary value"
+ if _bounded_utf8_length(text, max_string_bytes) is None:
+ return None, "contains an oversized binary value"
+ return text, None
+
+ return None, "contains an unsupported or subclassed value"
+
+
+def _is_chart_data_temporal_scalar(value: Any) -> bool:
+ """Return whether an exact scalar has Chart Data temporal wire
semantics."""
+ return type(value) in {
+ date,
+ datetime,
+ pd.Timestamp,
+ np.datetime64,
+ _PANDAS_NAT_TYPE,
+ }
+
+
+def _is_chart_data_duration_scalar(value: Any) -> bool:
+ """Return whether an exact scalar has Chart Data duration semantics."""
+ return type(value) in {timedelta, pd.Timedelta, np.timedelta64}
+
+
+def _chart_data_duration_text(value: Any) -> tuple[str | None, str | None]:
+ """Project an exact duration through Chart Data's public JSON spelling.
+
+ Builtin and pandas durations are ``timedelta`` instances consumed by
+ ``format_timedelta``. Exact NumPy durations model the real DataFrame
+ producer boundary: unambiguous units are promoted to a pandas Timedelta,
+ NaT becomes JSON null, and ambiguous or overflowing units fail closed.
+ """
+ value_type = type(value)
+ if value_type is timedelta:
+ text = _chart_data_builtin_timedelta_text(value)
+ elif value_type is pd.Timedelta:
+ text = _chart_data_pandas_timedelta_text(value)
+ elif value_type is np.timedelta64:
+ if np.isnat(value):
+ return None, None
+ try:
+ pandas_value = pd.Timedelta(value)
+ except (OverflowError, TypeError, ValueError):
+ return None, "contains an invalid NumPy duration"
+ text = _chart_data_pandas_timedelta_text(pandas_value)
+ else:
+ return None, "contains an unsupported duration value"
+ if _bounded_utf8_length(text, MAX_QUERY_RESULT_STRING_BYTES) is None:
+ return None, "contains an oversized duration"
+ return text, None
+
+
+def _chart_data_temporal_number( # noqa: C901
+ value: Any,
+) -> tuple[float | None, str | None]:
+ """Project an exact date/datetime through Chart Data's epoch-ms wire form.
+
+ The public Chart Data API uses ``json_int_dttm_ser`` before the browser
+ parses the payload. Trusted canonicalizers first validate or replace
+ timezone implementations without arbitrary hooks, then the production
+ ``datetime_to_epoch`` helper preserves its exact float behavior.
+ """
+ value_type = type(value)
+ if value_type is _PANDAS_NAT_TYPE:
+ # json_int_dttm_ser produces NaN and the Chart Data response's
+ # ``ignore_nan=True`` projects it to JSON null.
+ return None, None
+ if value_type is np.datetime64:
+ if np.isnat(value):
+ return None, None
+ try:
+ value = pd.Timestamp(value)
+ except (OverflowError, TypeError, ValueError):
+ return None, "contains an invalid NumPy datetime"
+ value_type = type(value)
+ if value_type is pd.Timestamp:
+ canonical_timestamp, reason = _trusted_timestamp_value(value)
+ if reason is not None or canonical_timestamp is None:
+ return None, reason or "contains an invalid pandas timestamp"
+ try:
+ return datetime_to_epoch(canonical_timestamp), None
+ except (OverflowError, TypeError, ValueError):
+ return None, "contains an invalid pandas timestamp"
+ if value_type is datetime:
+ canonical_datetime, reason = _trusted_datetime_value(value)
+ if reason is not None or canonical_datetime is None:
+ return None, reason or "contains an invalid datetime"
+ try:
+ return datetime_to_epoch(canonical_datetime), None
+ except (OverflowError, TypeError, ValueError):
+ return None, "contains an invalid datetime"
+ if value_type is date:
+ return (value - EPOCH.date()).total_seconds() * 1000, None
+ return None, "contains an unsupported temporal value"
+
+
+def _add_result_value_to_budget(value: Any, budget: _ResultBudget) -> str |
None:
+ """Account for one normalized row value and its JSON scalar size."""
+ budget.values += 1
+ if budget.values > MAX_QUERY_RESULT_VALUES:
+ return "contains too many total values"
+ if budget.values + budget.metadata_items > MAX_QUERY_RESULT_WORK:
+ return "exceeds the total work limit"
+ if type(value) is not list and type(value) is not dict:
+ if reason := _charge_json_bytes(budget,
_normalized_scalar_json_size(value)):
+ return reason
+ return None
+
+
+def _metadata_failure_for_value( # noqa: C901
+ value: Any, budget: _ResultBudget, *, row_shaped: bool = False
+) -> str | None:
+ """Bound metadata, charging row-shaped indexes to the row-data work
budget."""
+ stack: list[tuple[Any, int, bool]] = [(value, 0, False)]
+ active_containers: set[int] = set()
+ while stack:
+ item, depth, leaving = stack.pop()
+ if leaving:
+ active_containers.remove(id(item))
+ continue
+ if row_shaped:
+ budget.values += 1
+ if budget.values > MAX_QUERY_RESULT_VALUES:
+ return "metadata contains too many total values"
+ else:
+ budget.metadata_items += 1
+ if budget.metadata_items > MAX_QUERY_RESULT_METADATA_ITEMS:
+ return "metadata exceeds the item limit"
+ if budget.values + budget.metadata_items > MAX_QUERY_RESULT_WORK:
+ return "metadata exceeds the total work limit"
+ if depth > _MAX_ROW_CONTAINER_DEPTH:
+ return "metadata exceeds the nesting depth limit"
+
+ if type(item) is dict:
+ identity = id(item)
+ if identity in active_containers:
+ return "metadata contains cyclic containers"
+ active_containers.add(identity)
+ stack.append((item, depth, True))
+ item_count = dict.__len__(item)
+ if item_count > _MAX_ROW_CONTAINER_ITEMS:
+ return "metadata contains an oversized object"
+ if reason := _charge_json_bytes(
+ budget,
+ _container_json_syntax_size(item_count, mapping=True),
+ metadata=True,
+ ):
+ return reason
+ for key, child in dict.items(item):
+ if type(key) is not str:
+ return "metadata contains a non-string object key"
+ key_size = _json_string_size(key, MAX_QUERY_RESULT_KEY_BYTES)
+ if key_size is None:
+ return "metadata contains an invalid or oversized object
key"
+ if reason := _charge_json_bytes(budget, key_size,
metadata=True):
+ return reason
+ stack.append((child, depth + 1, False))
+ continue
+
+ if type(item) is list or type(item) is tuple:
+ identity = id(item)
+ if identity in active_containers:
+ return "metadata contains cyclic containers"
+ active_containers.add(identity)
+ stack.append((item, depth, True))
+ width = len(item)
+ # Chart Data emits one indexname per row. Only that outer array
+ # receives the row limit; containers within an index stay bounded.
+ max_items = (
+ MAX_QUERY_RESULT_ROWS
+ if row_shaped and depth == 0
+ else _MAX_ROW_CONTAINER_ITEMS
+ )
+ if width > max_items:
+ return "metadata contains an oversized array"
+ if reason := _charge_json_bytes(
+ budget,
+ _container_json_syntax_size(width, mapping=False),
+ metadata=True,
+ ):
+ return reason
+ stack.extend((item[index], depth + 1, False) for index in
range(width))
+ continue
+
+ normalized, reason = _normalize_trusted_scalar(
+ item, max_string_bytes=MAX_QUERY_RESULT_METADATA_BYTES
+ )
+ if reason is not None:
+ return f"metadata {reason}"
+ if reason := _charge_json_bytes(
+ budget,
+ _normalized_scalar_json_size(
+ normalized, max_string_bytes=MAX_QUERY_RESULT_METADATA_BYTES
+ ),
+ metadata=True,
+ ):
+ return reason
+ return None
+
+
+def _metadata_failure( # noqa: C901
+ payload: dict[str, Any], label: str
+) -> ChartError | None:
+ """Validate bounded cache and row-count metadata before any consumer."""
+ for key in ("rowcount", "total_rows"):
+ if key in payload and dict.__getitem__(payload, key) is not None:
+ try:
+ bounded_result_row_count(dict.__getitem__(payload, key))
+ except ValueError as ex:
+ return _malformed_result(f"{label} {key} {ex}")
+
+ if "is_cached" in payload:
+ is_cached = dict.__getitem__(payload, "is_cached")
+ if is_cached is not None and type(is_cached) is not bool:
+ return _malformed_result(f"{label} is_cached must be an exact
boolean")
+
+ if "cache_key" in payload:
+ cache_key = dict.__getitem__(payload, "cache_key")
+ if cache_key is not None and (
+ type(cache_key) is not str
+ or _bounded_utf8_length(cache_key, _MAX_CACHE_STRING_BYTES) is None
+ ):
+ return _malformed_result(
+ f"{label} cache_key must be a bounded exact string"
+ )
+
+ # ChartData's production schema emits ``cached_dttm``. ``cache_dttm`` was
+ # used by earlier MCP payloads and remains a bounded compatibility alias.
+ # Validate both before cache utilities parse or compare either value.
+ for key in ("cached_dttm", "cache_dttm"):
+ if key not in payload:
+ continue
+ cache_dttm = dict.__getitem__(payload, key)
+ if cache_dttm is None:
+ continue
+ if type(cache_dttm) is str:
+ if _bounded_utf8_length(cache_dttm, _MAX_CACHE_STRING_BYTES) is
not None:
+ continue
+ return _malformed_result(f"{label} {key} must be a bounded exact
string")
+ if type(cache_dttm) is datetime:
+ tzinfo = cache_dttm.tzinfo
+ if tzinfo is None or any(
+ type(tzinfo) is trusted for trusted in _TRUSTED_TZINFO_TYPES
+ ):
+ continue
+ return _malformed_result(f"{label} {key} has an unsupported
timezone")
+ return _malformed_result(
+ f"{label} {key} must be a bounded exact string or datetime"
+ )
+ return None
+
+
+def _excel_temporal_cell(value: Any) -> Any | None:
+ """Return a naive builtin temporal value an XLSX writer stores natively.
+
+ Called only after ``_normalize_trusted_scalar`` accepted ``value``. XLSX
has
+ no timezone-aware date cells, so aware values keep their ISO text, as do
+ NumPy durations whose unit may be ambiguous. Values outside the 1900 date
+ system's range, including negative durations, also keep their ISO text.
+ """
+ value_type = type(value)
+ if value_type is date:
+ return value if value >= _EXCEL_MIN_DATE else None
+ if value_type is timedelta:
+ return value if timedelta(0) <= value <= _EXCEL_MAX_DURATION else None
+ if value_type is datetime:
+ return (
+ value
+ if value.tzinfo is None
+ and value.date() >= _EXCEL_MIN_DATE
+ and value <= _EXCEL_MAX_DATETIME
+ else None
+ )
+ if value_type is time:
+ return value if value.tzinfo is None and value <= _EXCEL_MAX_TIME else
None
+ if value_type is np.datetime64:
+ value = pd.Timestamp(value)
+ value_type = pd.Timestamp
+ if value_type is pd.Timestamp:
+ if value.tzinfo is not None or not 1900 <= value.year <= 9999:
+ return None
+ # Excel stores millisecond precision; drop nanoseconds explicitly
+ # instead of through pandas' lossy-conversion warning.
+ return _excel_temporal_cell(
+ datetime(
+ value.year,
+ value.month,
+ value.day,
+ value.hour,
+ value.minute,
+ value.second,
+ value.microsecond,
+ )
+ )
+ if value_type is pd.Timedelta:
+ if not 0 <= value.days <= _EXCEL_MAX_DURATION.days:
+ return None
+ return _excel_temporal_cell(
+ timedelta(
+ days=value.days, seconds=value.seconds,
microseconds=value.microseconds
+ )
+ )
+ return None
+
+
+def _normalize_row_value( # noqa: C901
+ value: Any,
+ budget: _ResultBudget,
+ *,
+ temporal_json_numbers: bool = False,
+ preserve_nonfinite_floats: bool = False,
+ preserve_excel_temporals: bool = False,
+) -> str | None:
+ """Normalize trusted scalars and return a bounded serialization failure.
+
+ ChartDataCommand materializes a real DataFrame with ``to_dict(records)``,
so
+ exact pandas and NumPy scalars can remain in an otherwise valid result.
They
+ are converted in place to their canonical JSON-facing builtin values here.
+ Exact containers are inspected through builtin operations only; arbitrary
+ subclasses and scalar hooks remain rejected. ``preserve_excel_temporals``
+ keeps naive row-level date, time, datetime, and duration cells as builtin
+ temporal objects for XLSX exports; size budgets still use their JSON text.
+ """
+ stack: list[
+ tuple[
+ Any,
+ list[Any] | dict[str, Any] | None,
+ int | str | None,
+ int,
+ bool,
+ ]
+ ] = [(value, None, None, 0, False)]
+ active_containers: set[int] = set()
+
+ while stack:
+ item, parent, slot, depth, leaving = stack.pop()
+ if leaving:
+ active_containers.remove(id(item))
+ continue
+ if depth > _MAX_ROW_CONTAINER_DEPTH:
+ return "exceeds the nesting depth limit"
+
+ if type(item) is np.ndarray:
+ # Arrow list cells are exact arrays. Expand one bounded dimension
at
+ # a time, never calling tolist() before shape/work checks or
trusting
+ # ndarray subclasses. Object-array children use the same scalar and
+ # cycle guards as builtin containers.
+ identity = id(item)
+ if identity in active_containers:
+ return "contains cyclic containers"
+ if depth + item.ndim > _MAX_ROW_CONTAINER_DEPTH:
+ return "exceeds the nesting depth limit"
+ if any(width > _MAX_ROW_CONTAINER_ITEMS for width in item.shape):
+ return "contains an oversized array"
+ if item.size > MAX_QUERY_RESULT_VALUES - budget.values:
+ return "exceeds the total value limit"
+ active_containers.add(identity)
+ stack.append((item, None, None, depth, True))
+ scalar_array = item.ndim == 0
+ item = (
+ np.ndarray.__getitem__(item, ())
+ if scalar_array
+ else [
+ np.ndarray.__getitem__(item, index)
+ for index in range(item.shape[0])
+ ]
+ )
+ if type(parent) is list:
+ assert type(slot) is int
+ list.__setitem__(parent, slot, item)
+ elif type(parent) is dict:
+ assert type(slot) is str
+ dict.__setitem__(parent, slot, item)
+ if scalar_array:
+ stack.append((item, parent, slot, depth + 1, False))
+ continue
+
+ if type(item) is list or type(item) is tuple:
+ identity = id(item)
+ if identity in active_containers:
+ return "contains cyclic containers"
+ width = len(item)
+ if width > _MAX_ROW_CONTAINER_ITEMS:
+ return "contains an oversized array"
+ active_containers.add(identity)
+ stack.append((item, None, None, depth, True))
+ if type(item) is tuple:
+ item = list(item)
+ if type(parent) is list:
+ assert type(slot) is int
+ list.__setitem__(parent, slot, item)
+ elif type(parent) is dict:
+ assert type(slot) is str
+ dict.__setitem__(parent, slot, item)
+ if reason := _add_result_value_to_budget(item, budget):
+ return reason
+ if reason := _charge_json_bytes(
+ budget, _container_json_syntax_size(width, mapping=False)
+ ):
+ return reason
+ stack.extend(
+ (list.__getitem__(item, index), item, index, depth + 1, False)
+ for index in range(width)
+ )
+ continue
+
+ if type(item) is dict:
+ if reason := _add_result_value_to_budget(item, budget):
+ return reason
+ identity = id(item)
+ if identity in active_containers:
+ return "contains cyclic containers"
+ active_containers.add(identity)
+ stack.append((item, None, None, depth, True))
+ item_count = dict.__len__(item)
+ if item_count > _MAX_ROW_CONTAINER_ITEMS:
+ return "contains an oversized object"
+ if reason := _charge_json_bytes(
+ budget, _container_json_syntax_size(item_count, mapping=True)
+ ):
+ return reason
+ for key, child in dict.items(item):
+ if type(key) is not str:
+ return "contains a non-string object key"
+ key_size = _json_string_size(key, MAX_QUERY_RESULT_KEY_BYTES)
+ if key_size is None:
+ return "contains an invalid or oversized object key"
+ if reason := _charge_json_bytes(budget, key_size):
+ return reason
+ stack.append((child, item, key, depth + 1, False))
+ continue
+
+ normalized: Any
+ generic_scalar = False
+ if (
+ preserve_nonfinite_floats
+ and type(item) is float
+ and not math.isfinite(item)
+ ):
+ normalized, reason = item, None
+ elif temporal_json_numbers and _is_chart_data_temporal_scalar(item):
+ normalized, reason = _chart_data_temporal_number(item)
+ elif temporal_json_numbers and _is_chart_data_duration_scalar(item):
+ normalized, reason = _chart_data_duration_text(item)
+ else:
+ normalized, reason = _normalize_trusted_scalar(item)
+ generic_scalar = True
+ if reason is not None:
+ return reason
+ if reason := _add_result_value_to_budget(normalized, budget):
+ return reason
+ replacement = normalized
+ # Plugins with Chart Data temporal wire semantics (numbers/duration
+ # text) keep that projection in every export format.
+ if (
+ preserve_excel_temporals
Review Comment:
The Excel-only preservation here means shared normalization still rewrites
temporal cells for every other format, so CSV exports of existing non-Bullet
charts change from `2026-10-05 12:00:00` to `2026-10-05T12:00:00` and a
duration from `1:30:00` to `PT1H30M`. A downstream job that parses the old CSV
text with `strptime("%Y-%m-%d %H:%M:%S")` would start failing after upgrade
with no chart change, and `UPDATING.md` doesn't mention it. Should CSV keep the
established formatting, or is this change intended and worth documenting?
##########
docs/admin_docs/configuration/mcp-server.mdx:
##########
@@ -1349,6 +1359,111 @@ Disabling a plugin only stops new charts of that type
from being created. Existi
- **[Security](/developer-docs/extensions/security)** -- Security best
practices for extensions
- **[Deployment](/developer-docs/extensions/deployment)** -- Package and
deploy Superset extensions
+## Bullet chart compatibility
+
+The MCP Bullet plugin uses `chart_type: "bullet"` and the native ECharts
+`viz_type: "bullet"`. Its optional `dimensions` hierarchy maps to `groupby`.
+Omit `dimensions` (or use `null`) to create a single-metric Bullet without a
+breakdown. On updates, omission or `null` preserves the saved hierarchy; use
+`dimensions: []` to clear it explicitly. An `order_by` update can reference the
+saved dimensions without resending them; unknown targets are rejected after
+resolving the saved hierarchy. When replacing the dataset, saved-chart and
+cached-preview updates retain an omitted hierarchy only if its columns resolve
+in the replacement dataset; incompatible inherited roles and temporal-filter
+provenance are discarded.
+
+Inherited native SQL dimensions remain query expressions when updating metrics
+or `order_by`; reference their output labels to sort by them. Caller-supplied
+`dimensions` must still be physical columns, not SQL expressions.
+
+Dimension and metric output names are case-sensitive: quoted physical columns
+such as `Region` and `region` remain distinct. Reference lookup prefers exact
+names and uses case-insensitive matching only when there is a single candidate;
+ambiguous references require the exact spelling. Bullet metric strings use
+JavaScript numeric spellings: underscore separators and non-ASCII digits are
+rejected rather than interpreted as numbers. Bounded array-valued dimensions
+retain their raw values in data reads and exports; preview category labels use
+JavaScript string conversion (for example, `[1, 2]` displays as `1,2`). Native
+SQL metrics without a label use their SQL expression as the output label.
+
+Presentation-only updates preserve saved predicates, including legacy top-level
+`where`, `having`, and `filters`, and normalize the native `order_by_cols`
alias.
+When both `orderby` and `order_by_cols` are saved, their sort entries are
+concatenated in form-data key order, including when `orderby` is empty.
+Explicit `filters: []` and `order_by: []` clear those inherited controls.
+
+Range, marker, and marker-line label lists may be shorter than their value
lists.
+Labels that become empty after sanitization retain their slots so later labels
+remain aligned with their corresponding values.
+As in Explore, missing or empty range labels are not displayed, and missing or
+empty marker labels use the formatted numeric value. Extra labels have no value
+to annotate and are ignored. These rules also apply to saved-chart previews and
+updates. Omitted label controls preserve the saved state only when their
+corresponding value controls (`ranges`, `markers`, or `marker_lines`) are also
+omitted. Replacing a value control without its labels clears the saved labels;
+resend the labels to retain annotations with the replacement values.
+
+Saved native `ranges`, `markers`, and `marker_lines` controls ignore empty,
+non-numeric, and NaN tokens, matching Explore. If no numeric range remains, the
+preview uses the default band up to 110% of the largest measure. Unrelated
+updates preserve these saved controls. Newly authored typed lists must contain
+finite numbers; infinite native values and oversized controls remain errors.
+
+On same-dataset chart updates, `filters: []` clears both user filters and the
+generated dashboard-time binding. `temporal_column: null` clears only that
+generated binding, preserving user filters. Omitting those controls preserves
+the saved binding. Bullet Vega previews keep separate indexed rows even when
+dimension display labels are identical, and support both `SMART_NUMBER` and
+`SMART_NUMBER_SIGNED` number formats. Bullet numeric format precision is
limited
+to 20 digits before formatting, including on data reads.
+
+Native Bullet temporal filters retain the active filter's subject and range as
a
+pair; `No filter` placeholders do not supply the subject of another active
range.
+Typed configs support one such pair. Multiple active native temporal filters,
or
+conflicts with explicit `temporal_column`/`time_range`, are rejected rather
than
+silently dropping or moving a predicate.
+
+### Query result limits
+
+MCP chart tools accept up to 50,000 rows per query. The row-shaped `indexnames`
+array emitted by Chart Data uses the same 50,000-entry limit, rather than the
+4,096-entry limit for other metadata arrays. Index entries count toward the
+shared row-data work budget and retain the 1 MiB aggregate metadata byte limit.
+Multi-dimension pivot index tuples are serialized as JSON arrays. Nested
+containers within an index entry retain the standard container limits.
+
+The shared query-result validator also enforces **non-configurable hard
limits**:
+
+- **64 KiB (65,536 UTF-8 bytes) per text cell/string value**. A single
Review Comment:
These limits read broader than the code enforces. The 64 KiB cap applies to
row cells, while metadata strings fall under the separate metadata budget
(`test_query_result_accepts_full_sql_above_source_cell_string_limit` accepts a
70 KiB SQL string), and the 20-digit format-precision check at the top of this
section only runs when preview validation is on, so a saved Bullet with
`y_axis_format: ".21e"` still returns raw data. An operator could rewrite a
working saved chart or query unnecessarily; could the wording say these apply
to row cells and to previews respectively (same phrasing in `UPDATING.md`)?
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]