mirror of
https://github.com/apache/superset.git
synced 2026-08-04 04:52:32 +00:00
order_by_cols is a raw-mode-only control (resetOnHide: false), so an aggregate chart can carry a stale value. The form-data query-context rebuild read it in every mode, which could order an aggregate export by stale columns and return a different top-N than the chart shows. Gate it behind is_raw_query_mode so aggregate mode falls back to the metric-based ordering, mirroring the frontend Table buildQuery. Also add frontend-drift pointers naming the mirrored buildQuery / extractQueryFields / processFilters sources, and a regression test for an aggregate table with a stale order_by_cols. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
291 lines
12 KiB
Python
291 lines
12 KiB
Python
# Licensed to the Apache Software Foundation (ASF) under one
|
|
# or more contributor license agreements. See the NOTICE file
|
|
# distributed with this work for additional information
|
|
# regarding copyright ownership. The ASF licenses this file
|
|
# to you under the Apache License, Version 2.0 (the
|
|
# "License"); you may not use this file except in compliance
|
|
# with the License. You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing,
|
|
# software distributed under the License is distributed on an
|
|
# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
|
# KIND, either express or implied. See the License for the
|
|
# specific language governing permissions and limitations
|
|
# under the License.
|
|
"""
|
|
Synthesize a query context from a chart's saved form data (``params``).
|
|
|
|
A chart's ``query_context`` is normally generated client-side by each viz
|
|
plugin's ``buildQuery`` and only persisted when the chart is (re-)saved in
|
|
Explore. Charts that predate that behavior keep their ``params`` (form data) but
|
|
carry no ``query_context``, so server-side consumers that need to run the query
|
|
(e.g. the dashboard Excel export) have nothing to execute.
|
|
|
|
This module rebuilds a best-effort query context from the form data — columns,
|
|
metrics, filters (including free-form SQL and the time range), ordering and time
|
|
grain — mirroring the shared parts of the viz plugins' ``buildQuery``. It does
|
|
**not** reproduce plugin post-processing (pivot, contribution/percent
|
|
transforms, rolling/forecast) or multi-query fan-out, so callers must restrict it
|
|
to viz types whose data maps faithfully to a single plain query.
|
|
|
|
The mirrored logic lives on the frontend in
|
|
``superset-frontend/plugins/plugin-chart-table/src/buildQuery.ts`` (query mode,
|
|
ordering), ``superset-frontend/packages/superset-ui-core/src/query/`` (field
|
|
extraction, ``processFilters``). There is no automated tripwire tying the two
|
|
across the language boundary; the per-helper pointers below must be kept in sync
|
|
when that frontend logic changes.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
|
|
from superset.utils import json
|
|
|
|
|
|
def adhoc_filters_to_query_filters(
|
|
adhoc_filters: list[dict[str, Any]],
|
|
where_only: bool = False,
|
|
) -> list[dict[str, Any]]:
|
|
"""
|
|
Convert ``SIMPLE`` adhoc filters into QueryObject filter clauses.
|
|
|
|
Adhoc filters use ``{subject, operator, comparator}`` while a query object
|
|
expects ``{col, op, val}``; free-form ``SQL`` filters have no ``{col, op,
|
|
val}`` equivalent and are handled separately (see
|
|
:func:`freeform_where_having`).
|
|
|
|
By default all ``SIMPLE`` filters are converted (the behavior the MCP
|
|
compile/preview path relies on). Pass ``where_only=True`` to convert only
|
|
``WHERE``-clause filters, matching the frontend's ``processFilters``
|
|
(``superset-ui-core/src/query/processFilters.ts``) — the dashboard export uses
|
|
this so it applies the same rows the chart shows and does not additionally
|
|
filter on ``SIMPLE`` ``HAVING`` clauses.
|
|
"""
|
|
result: list[dict[str, Any]] = []
|
|
for flt in adhoc_filters or []:
|
|
if flt.get("expressionType") != "SIMPLE":
|
|
continue
|
|
if where_only and (flt.get("clause") or "WHERE").upper() != "WHERE":
|
|
continue
|
|
result.append(
|
|
{
|
|
"col": flt.get("subject"),
|
|
"op": flt.get("operator"),
|
|
"val": flt.get("comparator"),
|
|
}
|
|
)
|
|
return result
|
|
|
|
|
|
def freeform_where_having(form_data: dict[str, Any]) -> dict[str, str]:
|
|
"""
|
|
Collect free-form SQL predicates into a query ``extras`` mapping.
|
|
|
|
Mirrors ``processFilters`` on the frontend
|
|
(``superset-ui-core/src/query/processFilters.ts``): ``SQL`` adhoc filters (and
|
|
a legacy top-level ``where``) join into ``extras.where`` / ``extras.having`` by
|
|
clause, so a chart restricted by a custom SQL predicate exports the same rows
|
|
it displays instead of the full, unrestricted result.
|
|
"""
|
|
where: list[str] = []
|
|
having: list[str] = []
|
|
if form_data.get("where"):
|
|
where.append(form_data["where"])
|
|
for flt in form_data.get("adhoc_filters") or []:
|
|
if flt.get("expressionType") == "SQL" and flt.get("sqlExpression"):
|
|
clause = (flt.get("clause") or "WHERE").upper()
|
|
(having if clause == "HAVING" else where).append(flt["sqlExpression"])
|
|
|
|
extras: dict[str, str] = {}
|
|
if where:
|
|
extras["where"] = " AND ".join(f"({clause})" for clause in where)
|
|
if having:
|
|
extras["having"] = " AND ".join(f"({clause})" for clause in having)
|
|
return extras
|
|
|
|
|
|
def columns_from_form_data(form_data: dict[str, Any]) -> list[Any]:
|
|
"""
|
|
Derive the query's grouping/raw columns from form data.
|
|
|
|
Handles raw-mode tables (``all_columns``/``columns``), an ``x_axis`` (string
|
|
or adhoc column), and ``groupby`` dimensions, de-duplicating while preserving
|
|
order.
|
|
"""
|
|
if form_data.get("query_mode") == "raw" and (
|
|
form_data.get("all_columns") or form_data.get("columns")
|
|
):
|
|
return list(form_data.get("all_columns") or form_data.get("columns") or [])
|
|
|
|
groupby_columns: list[Any] = form_data.get("groupby") or []
|
|
raw_columns: list[Any] = form_data.get("columns") or []
|
|
# Prefer explicit raw columns only when they are actually present; a stale
|
|
# empty ``columns: []`` key must not shadow the group-by dimensions (which
|
|
# would silently drop the grouping and change the aggregation).
|
|
columns = raw_columns.copy() if raw_columns else groupby_columns.copy()
|
|
|
|
x_axis = form_data.get("x_axis")
|
|
if isinstance(x_axis, str) and x_axis and x_axis not in columns:
|
|
columns.insert(0, x_axis)
|
|
elif isinstance(x_axis, dict):
|
|
col_name = x_axis.get("column_name")
|
|
if col_name and col_name not in columns:
|
|
columns.insert(0, col_name)
|
|
return columns
|
|
|
|
|
|
def is_raw_query_mode(form_data: dict[str, Any]) -> bool:
|
|
"""
|
|
Whether the chart runs in raw (non-aggregated) mode, mirroring the frontend's
|
|
``getQueryMode`` (``plugin-chart-table/src/buildQuery.ts``): an explicit
|
|
``query_mode`` wins, otherwise the presence of ``all_columns`` implies raw mode.
|
|
"""
|
|
if mode := form_data.get("query_mode"):
|
|
return mode == "raw"
|
|
return bool(form_data.get("all_columns"))
|
|
|
|
|
|
def orderby_from_form_data(
|
|
form_data: dict[str, Any], metrics: list[Any], viz_type: str | None = None
|
|
) -> list[list[Any]]:
|
|
"""
|
|
Derive ordering so a ``row_limit`` returns the chart's top-N, not an
|
|
arbitrary N.
|
|
|
|
Raw-mode tables order by ``order_by_cols`` (stored as JSON ``[col, asc]``
|
|
pairs). Aggregate charts order by the configured sort metric
|
|
(``timeseries_limit_metric``, or the first metric when ``sort_by_metric`` is
|
|
set), otherwise fall back to the first metric descending — matching the
|
|
table/pie ``buildQuery`` defaults.
|
|
|
|
``order_by_cols`` is a raw-mode-only control (``resetOnHide: false`` in the
|
|
plugin control panels), so an aggregate chart can carry a stale value from a
|
|
previous raw-mode configuration. Aggregate mode must ignore it, mirroring the
|
|
frontend, where ``plugin-chart-table/src/buildQuery.ts:136-145`` overrides
|
|
``orderby`` with the sort metric (``order_by_cols`` reaches ``orderby`` only
|
|
via the alias in ``extractQueryFields.ts``, then gets overwritten in aggregate
|
|
mode).
|
|
"""
|
|
if is_raw_query_mode(form_data):
|
|
parsed: list[list[Any]] = []
|
|
for col in form_data.get("order_by_cols") or []:
|
|
if isinstance(col, str):
|
|
try:
|
|
col = json.loads(col)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
parsed.append(col)
|
|
return parsed
|
|
|
|
if not metrics:
|
|
return []
|
|
|
|
sort_metric = form_data.get("timeseries_limit_metric") or (
|
|
metrics[0] if form_data.get("sort_by_metric") else None
|
|
)
|
|
if sort_metric is not None:
|
|
# The Table plugin defaults ``order_desc`` to False (ascending); Pie and
|
|
# others sort by metric descending. Match that so a row limit keeps the
|
|
# chart's top/bottom-N rather than flipping it.
|
|
default_desc = viz_type != "table"
|
|
order_desc = form_data.get("order_desc", default_desc)
|
|
return [[sort_metric, not order_desc]]
|
|
# No explicit sort metric: default to the first metric, descending.
|
|
return [[metrics[0], False]]
|
|
|
|
|
|
def _columns_and_metrics(
|
|
form_data: dict[str, Any], viz_type: str | None
|
|
) -> tuple[list[Any], list[Any], bool]:
|
|
"""
|
|
Resolve the query's ``(columns, metrics, promoted_time_column)`` from form
|
|
data, honoring raw vs. aggregate mode and the Big Number trendline promotion.
|
|
"""
|
|
if is_raw_query_mode(form_data):
|
|
# Raw mode returns individual rows: use only the selected columns and
|
|
# ignore ``metrics``/``groupby``, which stay in form data as stale values
|
|
# (the controls aren't reset when hidden) but are ignored by the chart.
|
|
columns = list(form_data.get("all_columns") or form_data.get("columns") or [])
|
|
return columns, [], False
|
|
|
|
metrics = list(form_data.get("metrics") or [])
|
|
# Single-metric charts (e.g. Big Number) store ``metric`` rather than
|
|
# ``metrics``.
|
|
if not metrics and form_data.get("metric"):
|
|
metrics = [form_data["metric"]]
|
|
columns = columns_from_form_data(form_data)
|
|
# Only a Big Number *with a trendline* (viz_type ``big_number``) groups by its
|
|
# time column; ``big_number_total`` is a single aggregate and must not be
|
|
# grouped, or it would return one row per timestamp instead of a total.
|
|
if not columns and viz_type == "big_number" and form_data.get("granularity_sqla"):
|
|
return [form_data["granularity_sqla"]], metrics, True
|
|
return columns, metrics, False
|
|
|
|
|
|
def build_query_context_from_form_data(
|
|
form_data: dict[str, Any],
|
|
datasource: dict[str, Any],
|
|
viz_type: str | None = None,
|
|
) -> dict[str, Any]:
|
|
"""
|
|
Build a query-context payload (the JSON shape ``ChartDataQueryContextSchema``
|
|
loads) from a chart's form data and datasource reference.
|
|
|
|
:param form_data: The chart's saved ``params`` parsed to a dict.
|
|
:param datasource: ``{"id": <int>, "type": "table"}`` datasource reference.
|
|
:param viz_type: The chart's viz type, used for viz-specific handling.
|
|
:returns: A single-query query-context dict.
|
|
"""
|
|
columns, metrics, promoted_time_column = _columns_and_metrics(form_data, viz_type)
|
|
|
|
# SIMPLE adhoc filters (+ legacy top-level ``filters``) become query filters;
|
|
# free-form SQL predicates go into ``extras``. Only ``WHERE``-clause SIMPLE
|
|
# filters are applied (matching the chart), so the export never filters on a
|
|
# ``HAVING`` clause the chart itself ignores.
|
|
filters = adhoc_filters_to_query_filters(
|
|
form_data.get("adhoc_filters", []), where_only=True
|
|
)
|
|
for flt in form_data.get("filters") or []:
|
|
if isinstance(flt, dict) and flt.get("col") is not None:
|
|
filters.append(flt)
|
|
|
|
extras = freeform_where_having(form_data)
|
|
if form_data.get("time_grain_sqla"):
|
|
extras["time_grain_sqla"] = form_data["time_grain_sqla"]
|
|
|
|
# Prefer the modern ``time_range``; fall back to the legacy ``since``/``until``
|
|
# pair (older charts store the range that way) before defaulting to no filter.
|
|
time_range = form_data.get("time_range")
|
|
if not time_range and (form_data.get("since") or form_data.get("until")):
|
|
time_range = f"{form_data.get('since') or ''} : {form_data.get('until') or ''}"
|
|
time_range = time_range or "No filter"
|
|
query: dict[str, Any] = {
|
|
"columns": columns,
|
|
"metrics": metrics,
|
|
"orderby": orderby_from_form_data(form_data, metrics, viz_type),
|
|
"filters": filters,
|
|
"time_range": time_range,
|
|
}
|
|
if extras:
|
|
query["extras"] = extras
|
|
# ``granularity`` names the temporal column that applies the time range and
|
|
# that ``time_grain_sqla`` buckets. Set it when there is a real range to apply,
|
|
# or when we've promoted a Big Number trendline's time column (so its time
|
|
# grain takes effect even with no active range). Otherwise leave it off — a
|
|
# non-temporal ``granularity_sqla`` with no range would be forced through date
|
|
# bucketing and fail.
|
|
granularity = form_data.get("granularity") or form_data.get("granularity_sqla")
|
|
if granularity and (time_range != "No filter" or promoted_time_column):
|
|
query["granularity"] = granularity
|
|
if form_data.get("row_limit"):
|
|
query["row_limit"] = form_data["row_limit"]
|
|
|
|
return {
|
|
"datasource": datasource,
|
|
"queries": [query],
|
|
"form_data": form_data,
|
|
}
|