aminghadersohi commented on code in PR #43770:
URL: https://github.com/apache/superset/pull/43770#discussion_r4224607635
##########
superset/mcp_service/chart/chart_utils.py:
##########
@@ -1715,6 +1758,847 @@ def map_histogram_config(config:
"HistogramChartConfig") -> Dict[str, Any]:
return form_data
+def _bullet_token_list(values: Sequence[str | int | float]) -> str:
+ """Serialize typed Bullet controls to the frontend's comma-separated
form."""
+ tokens: list[str] = []
+ for value in values:
+ if isinstance(value, float):
+ token = repr(value)
+ # ``100`` parses back to the same binary float as ``100.0`` and
+ # preserves the frontend's established compact integer spelling.
+ if token.endswith(".0") and not (
+ value == 0.0 and math.copysign(1.0, value) < 0
+ ):
+ token = token[:-2]
+ tokens.append(token)
+ else:
+ tokens.append(str(value))
+ return ",".join(tokens)
+
+
+def map_bullet_config(config: BulletChartConfig) -> Dict[str, Any]: # noqa:
C901
+ """Map typed Bullet config to ``Bullet/buildQuery`` and transformProps.
+
+ The frontend buildQuery replaces the generic query fields with exactly one
+ metric and the groupby hierarchy. Presentation controls stay in native
+ snake_case form_data; the chart plugin camelizes them for transformProps.
+ """
+ if (
+ config.dimensions is None
+ and config._inherited_groupby is None
+ and config.order_by
+ ):
+ # An update resolves its saved hierarchy before mapping. Without one,
+ # creation must validate sort targets against an empty hierarchy.
+ BulletChartConfig.model_validate(
+ {**config.model_dump(exclude_unset=True), "dimensions": []}
+ )
+ metric = create_metric_object(config.metric)
+ form_data: Dict[str, Any] = {
+ "viz_type": "bullet",
+ "metric": metric,
+ }
+
+ # Optional semantic/query fields are emitted only when explicitly supplied.
+ # This lets update_chart and update_chart_preview preserve native saved
state,
+ # while an explicit empty value still clears it through the generic merge
path.
+ if "dimensions" in config.model_fields_set:
+ form_data["groupby"] = [dimension.name for dimension in
config.dimensions or []]
+ if "row_limit" in config.model_fields_set:
+ form_data["row_limit"] = config.row_limit
+ if "time_range" in config.model_fields_set:
+ form_data["time_range"] = config.time_range
+
+ if config.order_by:
+ dimensions = config.order_dimensions
+ orderby: list[list[Any]] = []
+ for order in config.order_by:
+ role, index = resolve_bullet_order_target(
+ order.column, dimensions, config.metric
+ )
+ if role == "metric":
+ order_target: Any = metric
+ else:
+ if index is None: # Defensive: resolver pairs dimensions with
indexes.
+ raise ValueError("Bullet dimension order target has no
index")
+ dimension = dimensions[index]
+ order_target = (
+ dimension.name if isinstance(dimension, ColumnRef) else
dimension
+ )
+ orderby.append([order_target, order.ascending])
+ form_data["orderby"] = orderby
+ elif "order_by" in config.model_fields_set:
+ form_data["orderby"] = []
+
+ presentation_fields: dict[str, tuple[str, Any]] = {
+ "ranges": ("ranges", _bullet_token_list(config.ranges)),
+ "range_labels": (
+ "range_labels",
+ _bullet_token_list(config.range_labels),
+ ),
+ "markers": ("markers", _bullet_token_list(config.markers)),
+ "marker_labels": (
+ "marker_labels",
+ _bullet_token_list(config.marker_labels),
+ ),
+ "marker_lines": (
+ "marker_lines",
+ _bullet_token_list(config.marker_lines),
+ ),
+ "marker_line_labels": (
+ "marker_line_labels",
+ _bullet_token_list(config.marker_line_labels),
+ ),
+ "y_axis_format": ("y_axis_format", config.y_axis_format),
+ "show_labels": ("show_labels", config.show_labels),
+ "show_legend": ("show_legend", config.show_legend),
+ }
+ for field_name, (form_key, value) in presentation_fields.items():
+ if field_name in config.model_fields_set:
+ form_data[form_key] = value
+
+ _add_adhoc_filters(form_data, config.filters)
+ if config.filters == [] and "filters" in config.model_fields_set:
+ form_data["adhoc_filters"] = []
+ if config.time_range and config.temporal_column:
+ _ensure_temporal_adhoc_filter(form_data, config.temporal_column)
+ for filter_ in form_data.get("adhoc_filters", []):
+ if (
+ isinstance(filter_, dict)
+ and filter_.get("operator") ==
FilterOperator.TEMPORAL_RANGE.value
+ and filter_.get("subject") == config.temporal_column
+ and filter_.get("comparator") == NO_TIME_RANGE
+ ):
+ filter_["comparator"] = config.time_range
+ return form_data
+
+
+def _normalize_native_filter_aliases(form_data: Mapping[str, Any]) ->
Dict[str, Any]:
+ """Fold legacy WHERE, HAVING, and filter predicates into adhoc controls."""
+ from superset.utils.core import form_data_to_adhoc, simple_filter_to_adhoc
+
+ normalized = dict(form_data)
+ legacy_filters = [
+ form_data_to_adhoc(normalized, clause)
+ for clause in ("having", "where")
+ if normalized.get(clause)
+ ]
+ legacy_filters.extend(
+ simple_filter_to_adhoc(filter_, "where")
+ for filter_ in normalized.get("filters") or []
+ if filter_ is not None
+ )
+ if legacy_filters:
+ normalized["adhoc_filters"] = [
+ *legacy_filters,
+ *(normalized.get("adhoc_filters") or []),
+ ]
+ for key in ("where", "having", "filters"):
+ normalized.pop(key, None)
+ return normalized
+
+
+def _normalize_bullet_query_aliases(form_data: Mapping[str, Any]) -> Dict[str,
Any]:
+ """Fold inherited native predicates and ordering into canonical
controls."""
+ from superset.mcp_service.chart.chart_helpers import _parse_orderby
+
+ normalized = _normalize_native_filter_aliases(form_data)
+ if "groupby" in normalized and not isinstance(normalized["groupby"], list):
+ # Bullet/buildQuery and transformProps read ensureIsArray(groupby).
+ normalized["groupby"] = bullet_groupby_list(normalized["groupby"])
+ if "order_by_cols" in normalized:
+ # Native extractQueryFields concatenates both aliases in key order.
+ ordering: list[Any] = []
+ for key, value in normalized.items():
+ if key == "order_by_cols":
+ ordering.extend(_parse_orderby(value))
+ elif key == "orderby":
+ ordering.extend(value or [])
+ normalized["orderby"] = ordering
+ normalized.pop("order_by_cols", None)
+ return normalized
+
+
+def merge_bullet_form_data(
+ existing_form_data: Mapping[str, Any], new_form_data: Dict[str, Any]
+) -> None:
+ """Preserve omitted native Bullet controls across update tool paths.
+
+ Query roles and every UI control have an explicit typed representation.
+ Mappers emit optional fields only when the caller supplied them, so copying
+ the bounded native keys below preserves omitted state while explicit empty,
+ false, null, and zero-like values remain authoritative.
+ """
+ if (
+ existing_form_data.get("viz_type") != "bullet"
+ or new_form_data.get("viz_type") != "bullet"
+ ):
+ return
+ existing_form_data = _normalize_bullet_query_aliases(existing_form_data)
+ preserved_keys = {
+ "groupby",
+ "adhoc_filters",
+ "time_range",
+ "row_limit",
+ "orderby",
+ "ranges",
+ "range_labels",
+ "markers",
+ "marker_labels",
+ "marker_lines",
+ "marker_line_labels",
+ "y_axis_format",
+ "show_labels",
+ "show_legend",
+ "url_params",
+ # Native query context (dashboard/native filter predicates and time
+ # overrides) that buildQueryObject applies on top of the controls.
+ "extra_form_data",
+ "extra_filters",
+ MCP_DASHBOARD_TIME_FILTER_SUBJECT,
+ }
+
+ # Threshold and label arrays are one frontend control pair. If callers
+ # replace the values without replacing their labels, clear the stale labels
+ # instead of accidentally reassigning them by position.
+ dependent_controls = {
+ "ranges": "range_labels",
+ "markers": "marker_labels",
+ "marker_lines": "marker_line_labels",
+ }
+ for values_key, labels_key in dependent_controls.items():
+ if values_key in new_form_data and labels_key not in new_form_data:
+ new_form_data[labels_key] = ""
+
+ preserve_orderby = (
+ "orderby" not in new_form_data and "orderby" in existing_form_data
+ )
+ for key in preserved_keys:
+ if (
+ key == MCP_DASHBOARD_TIME_FILTER_SUBJECT
+ and "adhoc_filters" in new_form_data
+ ):
+ # The marker describes a mapper-generated temporal filter. Do not
+ # retain stale provenance when an explicit filter update removed
it.
+ continue
+ if key in existing_form_data and key not in new_form_data:
+ new_form_data[key] = existing_form_data[key]
+ if preserve_orderby:
+ new_form_data["orderby"] = _orderby_for_final_output_roles(
+ existing_form_data, new_form_data
+ )
+
+
+def _bullet_output_labels(
+ form_data: Mapping[str, Any],
+) -> tuple[set[str], dict[str, Any]]:
+ """Return a Bullet state's dimension output labels and metric outputs."""
+ from superset.mcp_service.chart.chart_helpers import _column_label,
_metric_label
+
+ dimensions = {
+ label
+ for column in bullet_groupby_list(form_data.get("groupby"))
+ if (label := _column_label(column)) is not None
+ }
+ metrics = form_data.get("metrics") or []
+ if not isinstance(metrics, (list, tuple)):
+ metrics = [metrics]
+ metric_outputs = {
+ label: metric
+ for metric in [form_data.get("metric"), *metrics]
+ if (label := _metric_label(metric)) is not None
+ }
+ return dimensions, metric_outputs
+
+
+def bullet_groupby_list(groupby: Any) -> list[Any]:
+ """Normalize a saved Bullet hierarchy like the frontend
``ensureIsArray``."""
+ if groupby is None:
+ return []
+ return list(groupby) if isinstance(groupby, (list, tuple)) else [groupby]
+
+
+def _orderby_for_final_output_roles(
+ existing_form_data: Mapping[str, Any], new_form_data: Mapping[str, Any]
+) -> Any:
+ """Drop sorts on removed output roles and rebind inherited metric
expressions.
+
+ Native ordering may also rank by a saved metric or column that is not a
+ displayed output (``get_sqla_query`` resolves it independently). Those
+ sorters never named a Bullet role, so a role change does not remove them.
+ """
+ from superset.mcp_service.chart.chart_helpers import _column_label,
_metric_label
+
+ saved = existing_form_data.get("orderby")
+ if not isinstance(saved, list):
+ return saved
+ outputs, metric_outputs = _bullet_output_labels(new_form_data)
+ outputs.update(metric_outputs)
+ previous_dimensions, previous_metrics =
_bullet_output_labels(existing_form_data)
+ previous_outputs = previous_dimensions | set(previous_metrics)
+ retained = []
+ for entry in saved:
+ if isinstance(entry, (list, tuple)) and entry:
+ target = entry[0]
+ label = (
+ _metric_label(target)
+ or _column_label(target)
+ or target.get("metric_name")
+ if isinstance(target, Mapping)
+ else target
+ )
+ if (
+ isinstance(label, str)
+ and label not in outputs
+ and label in previous_outputs
+ ):
+ continue
+ if (
+ isinstance(target, Mapping)
+ and isinstance(label, str)
+ and label in metric_outputs
+ ):
+ # Label equality identifies an output role, not expression
+ # equality: execute the final metric, never the saved
expression.
+ entry = [metric_outputs[label], *entry[1:]]
+ retained.append(entry)
+ return retained
+
+
+def _filter_identity(filter_: Any) -> tuple[Any, ...] | None:
+ """Return the native identity used when one filter replaces another."""
+ if not isinstance(filter_, Mapping):
+ return None
+ return (
+ filter_.get("clause"),
+ filter_.get("expressionType"),
+ filter_.get("subject"),
+ filter_.get("operator"),
+ )
+
+
+def _temporal_binding_filter(filters: list[Any], subject: Any) -> dict[str,
Any] | None:
+ """Find the unique filter owned by a recorded MCP temporal marker."""
+ if subject is None:
+ return None
+ if not isinstance(subject, str) or not subject:
+ raise ValueError(
+ "MCP temporal binding provenance subject must be a non-empty
string"
+ )
+ matches = [
+ filter_
+ for filter_ in filters
+ if isinstance(filter_, dict)
+ and filter_.get("subject") == subject
+ and filter_.get("operator") == FilterOperator.TEMPORAL_RANGE.value
+ ]
+ if len(matches) != 1:
+ raise ValueError(
+ "MCP temporal binding provenance must match exactly one "
+ f"TEMPORAL_RANGE filter for subject {subject!r}; found
{len(matches)}"
+ )
+ return matches[0]
+
+
+def _append_or_replace_filter(filters: list[Any], filter_: Any) -> None:
+ """Append a filter, replacing the same native role when identifiable."""
+ identity = _filter_identity(filter_)
+ if identity is None:
+ if filter_ not in filters:
+ filters.append(filter_)
+ return
+ filters[:] = [item for item in filters if _filter_identity(item) !=
identity]
+ filters.append(filter_)
+
+
+_NATIVE_TEMPORAL_ROLE_FIELDS: dict[str, frozenset[str]] = {
+ # Typed ``x`` is persisted as native x_axis/granularity_sqla for XY and
+ # Mixed Timeseries. Waterfall exposes the typed field as ``x_axis``.
+ "x_axis": frozenset({"x", "x_axis"}),
+ "granularity_sqla": frozenset({"x", "x_axis", "temporal_column"}),
+ # Chart plugins may designate a chart-specific query role as the implicit
+ # dashboard-time subject.
+ "start_time": frozenset({"start_time"}),
+}
+
+
+def _native_temporal_subject_changed(
+ existing_form_data: Mapping[str, Any],
+ new_form_data: Mapping[str, Any],
+ explicit_fields: set[str],
+) -> bool:
+ """Return whether an authoritative native temporal role was replaced.
+
+ Mapping a partial update can propose a dataset fallback binding even when
+ the caller only changed filters. That proposal is not authoritative. A
+ changed x/granularity/chart-specific role is authoritative only when its
+ corresponding typed field was actually supplied.
+ """
+ for native_key, typed_fields in _NATIVE_TEMPORAL_ROLE_FIELDS.items():
+ if explicit_fields.isdisjoint(typed_fields):
+ continue
+ existing_value = existing_form_data.get(native_key)
+ incoming_value = new_form_data.get(native_key)
+ if existing_value != incoming_value:
+ return True
+ return False
+
+
+def _native_temporal_binding(
+ form_data: Mapping[str, Any], filters: list[Any]
+) -> tuple[str | None, dict[str, Any] | None]:
+ """Resolve one binding for a trusted native temporal role, if present."""
+ for native_key in _NATIVE_TEMPORAL_ROLE_FIELDS:
+ subject = form_data.get(native_key)
+ if not isinstance(subject, str) or not subject:
+ continue
+ matches = [
+ filter_
+ for filter_ in filters
+ if isinstance(filter_, dict)
+ and filter_.get("subject") == subject
+ and filter_.get("operator") == FilterOperator.TEMPORAL_RANGE.value
+ ]
+ if len(matches) > 1:
+ raise ValueError(
+ "An authoritative native temporal subject must match at most
one "
+ f"TEMPORAL_RANGE filter for subject {subject!r}; found "
+ f"{len(matches)}"
+ )
+ if matches:
+ return subject, matches[0]
+ return None, None
+
+
+def merge_update_form_data( # noqa: C901
+ existing_form_data: Mapping[str, Any],
+ new_form_data: Dict[str, Any],
+ config: ChartConfig,
+) -> None:
+ """Apply the shared omission/provenance contract for chart updates.
+
+ Mapper-generated neutral temporal bindings are infrastructure, not evidence
+ that the caller supplied ``filters`` or changed a saved time-range binding.
+ This helper is used by immediate saves, preview-first saved updates, and
+ cached-preview updates so omission, clear, replacement, and temporal
+ overrides have identical behavior.
+
+ State never crosses a visualization boundary: a viz-type change starts from
+ the mapper's output, so the previous chart's predicates are not restored.
+ """
+ existing_viz_type = existing_form_data.get("viz_type")
+ if isinstance(existing_viz_type, str) and existing_viz_type !=
new_form_data.get(
+ "viz_type"
+ ):
+ return
+ existing_form_data = _normalize_native_filter_aliases(existing_form_data)
+ # The initial overlay may carry legacy keys from saved form data. Filter
+ # omission/replacement below owns the complete canonical predicate
sequence.
+ for key in ("where", "having", "filters"):
+ new_form_data.pop(key, None)
+ existing_filters = list(existing_form_data.get("adhoc_filters") or [])
+ incoming_filters = list(new_form_data.get("adhoc_filters") or [])
+ existing_subject =
existing_form_data.get(MCP_DASHBOARD_TIME_FILTER_SUBJECT)
+ incoming_subject = new_form_data.get(MCP_DASHBOARD_TIME_FILTER_SUBJECT)
+ existing_binding = _temporal_binding_filter(existing_filters,
existing_subject)
+ incoming_binding = _temporal_binding_filter(incoming_filters,
incoming_subject)
+
+ explicit_fields = set(getattr(config, "model_fields_set", set()))
+ filters_explicit = "filters" in explicit_fields
+ range_explicit = "time_range" in explicit_fields
+ subject_explicit = "temporal_column" in explicit_fields
+ native_subject_changed = _native_temporal_subject_changed(
+ existing_form_data, new_form_data, explicit_fields
+ )
+ subject_authoritative = subject_explicit or native_subject_changed
+ if incoming_binding is None:
+ native_subject, native_binding = _native_temporal_binding(
+ new_form_data, incoming_filters
+ )
+ if native_binding is not None:
+ incoming_subject = native_subject
+ incoming_binding = native_binding
+ incoming_user_filters = [
+ filter_ for filter_ in incoming_filters if filter_ is not
incoming_binding
+ ]
+ temporal_explicit = range_explicit or subject_authoritative
+ if (
+ existing_binding is None
+ and incoming_binding is not None
+ and isinstance(incoming_subject, str)
+ and "filters" not in explicit_fields
+ and temporal_explicit
+ ):
+ # A saved temporal filter that Explore wrote has no MCP provenance
+ # marker. When it is the only native filter for the incoming subject,
+ # the update replaces it in place instead of appending a duplicate.
+ native_matches = [
+ filter_
+ for filter_ in existing_filters
+ if isinstance(filter_, dict)
+ and filter_.get("subject") == incoming_subject
+ and filter_.get("operator") == FilterOperator.TEMPORAL_RANGE.value
+ ]
+ if len(native_matches) == 1:
+ existing_binding = native_matches[0]
+ existing_subject = incoming_subject
+
+ chosen_binding: dict[str, Any] | None = None
+ chosen_subject: Any = None
+ if not filters_explicit:
+ # Omission is byte-faithful: keep the native sequence in its exact
order,
+ # including SQL/HAVING objects and a provenance-owned binding at any
index.
+ merged_filters = list(existing_filters)
+ chosen_binding = existing_binding
+ chosen_subject = existing_subject
+ if temporal_explicit:
+ if subject_authoritative:
+ chosen_binding = incoming_binding
+ chosen_subject = incoming_subject
+ elif existing_binding is not None:
+ # A range-only update belongs to the saved subject, even when
+ # mapping the partial config proposed the dataset main_dttm.
+ chosen_binding = dict(existing_binding)
+ chosen_subject = existing_subject
+ else:
+ chosen_binding = incoming_binding
+ chosen_subject = incoming_subject
+
+ if chosen_binding is not None:
+ chosen_binding = dict(chosen_binding)
+ if range_explicit:
+ chosen_binding["comparator"] = (
+ getattr(config, "time_range", None) or NO_TIME_RANGE
+ )
+ elif existing_binding is not None:
+ # Subject-only replacement preserves the saved active or
+ # neutral range instead of resetting it to No filter.
+ chosen_binding["comparator"] = existing_binding.get(
+ "comparator", NO_TIME_RANGE
+ )
+ if existing_binding is not None:
+ binding_index = next(
+ index
+ for index, filter_ in enumerate(merged_filters)
+ if filter_ is existing_binding
+ )
+ if chosen_binding is None:
+ merged_filters.pop(binding_index)
+ else:
+ # A temporal override changes infrastructure in place
instead
+ # of moving it past surrounding native filters.
+ merged_filters[binding_index] = chosen_binding
+ elif chosen_binding is not None:
+ merged_filters.append(chosen_binding)
+ else:
+ # An explicit filter array replaces the saved native sequence. The
mapper
+ # deliberately emits [] for an explicit clear; otherwise retain its
+ # generated temporal binding after the replacement filters.
+ merged_filters = list(incoming_user_filters)
+ if incoming_user_filters or temporal_explicit:
+ if subject_authoritative:
+ chosen_binding = incoming_binding
+ chosen_subject = incoming_subject
+ elif range_explicit and existing_binding is not None:
+ chosen_binding = dict(existing_binding)
+ chosen_subject = existing_subject
+ else:
+ # A filter-only replacement keeps the saved provenance binding.
+ # The mapper's incoming binding may merely be a dataset
fallback
+ # and must not reset the saved subject or active range.
+ chosen_binding = existing_binding
+ chosen_subject = existing_subject
+ if chosen_binding is not None:
+ chosen_binding = dict(chosen_binding)
+ if range_explicit:
+ chosen_binding["comparator"] = (
+ getattr(config, "time_range", None) or NO_TIME_RANGE
+ )
+ elif subject_authoritative and existing_binding is not None:
+ chosen_binding["comparator"] = existing_binding.get(
+ "comparator", NO_TIME_RANGE
+ )
+ if chosen_binding is not None:
+ _append_or_replace_filter(merged_filters, chosen_binding)
+
+ # Materialize exactly when saved state had the key or the caller made the
+ # controls authoritative. An omitted update must not turn a missing native
+ # filter key into [] merely because its mapper proposed a neutral binding.
+ if filters_explicit or "adhoc_filters" in existing_form_data or
temporal_explicit:
+ new_form_data["adhoc_filters"] = merged_filters
+ else:
+ new_form_data.pop("adhoc_filters", None)
+ if chosen_binding is not None and isinstance(chosen_subject, str):
+ new_form_data[MCP_DASHBOARD_TIME_FILTER_SUBJECT] = chosen_subject
+ else:
+ new_form_data.pop(MCP_DASHBOARD_TIME_FILTER_SUBJECT, None)
+
+
+def _currency_form_value(value: CurrencyFormat | None) -> dict[str, str] |
None:
+ """Return the native value for an explicitly supplied currency control."""
+ return value.to_form_data() if value is not None else None
+
+
+def _column_names(value: Sequence[ColumnRef] | None) -> list[str | None] |
None:
+ """Return a native column-name list while retaining an explicit null."""
+ return [column.name for column in value] if value is not None else None
+
+
+def _table_sort_value(value: Sequence[str | SortByConfig] | None) -> list[str]
| None:
+ """Return the native Table sort control for an explicit typed value."""
+ if value is None:
+ return None
+ return [
+ json.dumps(
+ [entry.column, entry.ascending]
+ if isinstance(entry, SortByConfig)
+ else [entry, False]
+ )
+ for entry in value
+ ]
+
+
+def _table_column_config_value(value: Any) -> dict[str, Any] | None:
+ """Return Table column config without losing an explicit null or empty
map."""
+ if value is None:
+ return None
+ return {
+ label: column.model_dump(by_alias=True, exclude_unset=True)
+ for label, column in value.items()
+ }
+
+
+# Mappers intentionally omit optional controls so fresh charts use the frontend
+# defaults. During a same-viz update, however, an explicitly supplied false,
+# null, or empty value must block preservation of the saved native key. Keep
the
+# typed-to-native relationship declarative so every update path shares it.
+_FormValueConverter = Callable[[Any], Any]
+_FormControlMap = dict[str, tuple[str, _FormValueConverter]]
+
+_COMMON_EXPLICIT_FORM_CONTROLS: _FormControlMap = {
+ "color_scheme": ("color_scheme", lambda value: value),
+ "currency_format": ("currency_format", _currency_form_value),
+ "show_value": ("show_value", lambda value: value),
+}
+
+_CHART_EXPLICIT_FORM_CONTROLS: dict[str, _FormControlMap] = {
+ "table": {
+ "sort_by": ("order_by_cols", _table_sort_value),
+ "column_config": ("column_config", _table_column_config_value),
+ },
+ "xy": {
+ "group_by": ("groupby", _column_names),
+ "series_limit": ("series_limit", lambda value: value),
+ "stacked": ("stack", lambda value: "Stack" if value else None),
+ "orientation": ("orientation", lambda value: value),
+ "legend_orientation": ("legendOrientation", lambda value: value),
+ "x_axis_time_format": ("x_axis_time_format", lambda value: value),
+ "time_grain": ("time_grain_sqla", lambda value: value),
+ },
+ "mixed_timeseries": {
+ "group_by": ("groupby", _column_names),
+ "group_by_secondary": ("groupby_b", _column_names),
+ "currency_format_secondary": (
+ "currency_format_secondary",
+ _currency_form_value,
+ ),
+ "time_grain": ("time_grain_sqla", lambda value: value),
+ },
+ "waterfall": {
+ "time_grain": ("time_grain_sqla", lambda value: value),
+ },
+ "big_number": {
+ "subheader": ("subheader", lambda value: value),
+ "y_axis_format": ("y_axis_format", lambda value: value),
+ "time_grain": ("time_grain_sqla", lambda value: value),
+ "compare_lag": ("compare_lag", lambda value: value),
+ "time_format": ("time_format", lambda value: value),
+ "aggregation": ("aggregation", lambda value: value),
+ },
+ "handlebars": {
+ "style_template": ("styleTemplate", lambda value: value),
+ "columns": ("all_columns", _column_names),
+ "groupby": ("groupby", _column_names),
+ "metrics": ("metrics", _column_names),
+ },
+ "pivot_table": {
+ "date_format": ("date_format", lambda value: value),
+ },
+ "interactive_pivot": {
+ "time_grain": ("time_grain_sqla", lambda value: value),
+ "series_limit": ("series_limit", lambda value: value),
+ "date_format": ("date_format", lambda value: value),
+ "column_sort": ("colOrder", lambda value: value),
+ },
+}
+
+
+def _apply_explicit_form_controls( # noqa: C901
+ existing_form_data: Mapping[str, Any],
+ new_form_data: Dict[str, Any],
+ config: ChartConfig,
+) -> None:
+ """Apply typed controls whose mapper omission represents a native clear."""
+ explicit_fields = set(getattr(config, "model_fields_set", set()))
+ controls = {
+ **_COMMON_EXPLICIT_FORM_CONTROLS,
+ **_CHART_EXPLICIT_FORM_CONTROLS.get(config.chart_type, {}),
+ }
+ for field_name, (native_key, convert) in controls.items():
+ if field_name in explicit_fields:
+ converted = convert(getattr(config, field_name))
+ is_clear = (
+ converted is None
+ or converted is False
+ or converted == ""
+ or converted in ([], {})
+ )
+ if is_clear:
+ new_form_data[native_key] = converted
+
+ axis_controls = {
+ "xy": (
+ ("x_axis", "x_axis_title", "x_axis_format", None),
+ ("y_axis", "y_axis_title", "y_axis_format", "logAxis"),
+ ),
+ "mixed_timeseries": (
+ ("x_axis", "xAxisTitle", "x_axis_time_format", None),
+ ("y_axis", "yAxisTitle", "y_axis_format", "logAxis"),
+ (
+ "y_axis_secondary",
+ "yAxisTitleSecondary",
+ "y_axis_format_secondary",
+ "logAxisSecondary",
+ ),
+ ),
+ }
+ if config.chart_type in axis_controls:
+ for field_name, title_key, format_key, scale_key in axis_controls[
+ config.chart_type
+ ]:
+ if field_name not in explicit_fields:
+ continue
+ axis = getattr(config, field_name)
+ if axis is None:
+ new_form_data[title_key] = None
+ new_form_data[format_key] = None
+ if scale_key:
+ new_form_data[scale_key] = None
+ continue
+ axis_fields = set(axis.model_fields_set)
+ if "title" in axis_fields:
+ new_form_data[title_key] = axis.title
+ if "format" in axis_fields:
+ new_form_data[format_key] = axis.format
+ if scale_key and "scale" in axis_fields:
+ new_form_data[scale_key] = (
+ None if axis.scale is None else axis.scale == "log"
+ )
+
+ if config.chart_type == "xy" and "legend" in explicit_fields:
+ legend = config.legend
+ if legend is None:
+ new_form_data["show_legend"] = None
+ new_form_data["legendOrientation"] = None
+ else:
+ legend_fields = set(legend.model_fields_set)
+ if "show" in legend_fields:
+ new_form_data["show_legend"] = legend.show
+ if "position" in legend_fields:
+ new_form_data["legendOrientation"] = legend.position
+
+ if config.chart_type == "interactive_pivot":
+ if "temporal_column" in explicit_fields and config.temporal_column is
None:
+ new_form_data["granularity_sqla"] = None
+ new_form_data["temporal_columns_lookup"] = None
+ if (
+ "series_limit_metric" in explicit_fields
+ and config.series_limit_metric is None
+ ):
+ new_form_data["series_limit_metric"] = None
+ if "comparison_period" in explicit_fields and config.comparison_period
is None:
+ new_form_data["time_compare"] = None
+ if "comparison_type" in explicit_fields and config.comparison_type is
None:
+ new_form_data["comparison_type"] = None
+
+ # A Waterfall axis replacement cannot inherit a bucket belonging to the old
+ # temporal subject. Grain omission preserves only while the axis is stable;
+ # explicit null is already handled by the declarative control map above.
+ if (
+ config.chart_type == "waterfall"
+ and existing_form_data.get("x_axis") != new_form_data.get("x_axis")
+ and "time_grain" not in explicit_fields
+ ):
+ new_form_data["time_grain_sqla"] = None
+
+
+def merge_same_viz_form_data(
+ existing_form_data: Mapping[str, Any],
+ new_form_data: Dict[str, Any],
+ config: ChartConfig,
+) -> None:
+ """Preserve saved controls that the typed mapper does not represent.
+
+ The typed MCP surface deliberately models a bounded subset of every Explore
+ control panel. For a replacement within the exact same native ``viz_type``,
+ keys absent from the mapper therefore represent omitted controls and retain
+ their saved values. Mapper output and the chart-specific merge helpers run
+ first and remain authoritative, including explicit empty, false, null, and
+ nested values.
+
+ No generic state crosses a visualization boundary. This prevents query-role
+ keys from the previous plugin (for example ``metric`` or ``groupby``) from
+ leaking into a different plugin whose role contract is unrelated.
+ """
+ existing_viz_type = existing_form_data.get("viz_type")
+ if not isinstance(existing_viz_type, str) or existing_viz_type !=
new_form_data.get(
+ "viz_type"
+ ):
+ return
+
+ _apply_explicit_form_controls(existing_form_data, new_form_data, config)
+
+ for key, value in existing_form_data.items():
+ if key in {"where", "having", "filters"}:
+ # merge_update_form_data already resolved these legacy predicates
+ # into adhoc_filters; restoring them would undo replacement/clear.
+ continue
+ if key == MCP_DASHBOARD_TIME_FILTER_SUBJECT:
+ # merge_update_form_data owns this provenance marker. Its absence
+ # may be an intentional subject clear and must not be undone by the
+ # generic preservation layer.
+ continue
+ if key not in new_form_data:
+ new_form_data[key] = value
+
+
+def validate_merged_bullet_form_data(
+ form_data: Mapping[str, Any],
+ update_config: ChartConfig | None = None,
+) -> BulletChartConfig | None:
+ """Validate final Bullet controls without reinterpreting inherited query
roles.
+
+ Saved Explore state may contain SQL dimensions and SQL WHERE/SIMPLE HAVING
+ filters beyond the typed authoring surface. Omitted roles are validated by
+ the native query contract and compilation, not as newly authored physical
+ columns or SIMPLE WHERE filters. Only this validation copy excludes them
+ and native query metadata; compiled and persisted form data stays intact.
+ Explicit replacements, including ``[]``, retain strict typed validation.
+ """
+ if form_data.get("viz_type") != "bullet":
+ return None
+ validation_data = dict(form_data)
+ for native_query_key in ("url_params", "extra_form_data", "extra_filters"):
+ validation_data.pop(native_query_key, None)
+ if update_config is None or isinstance(update_config, BulletChartConfig):
+ if update_config is None or update_config.dimensions is None:
+ validation_data.pop("groupby", None)
+ if update_config is None or "filters" not in
update_config.model_fields_set:
+ validation_data.pop("adhoc_filters", None)
+ validation_data.pop(MCP_DASHBOARD_TIME_FILTER_SUBJECT, None)
+ return BulletChartConfig.model_validate(validation_data)
Review Comment:
Confirmed: restating `dimensions` kept the inherited `orderby` in the strict
validation copy, so a saved independent ranking sort failed as if newly
authored. Fixed in e3dc3844f2e386b0f34b9a3596431c47c7a680ce:
`validate_merged_bullet_form_data` also drops inherited
`orderby`/`order_by_cols` unless the update supplies `order_by`, matching
`groupby` and `adhoc_filters`; the native query contract still validates the
inherited sort at compile. New
`test_saved_bullet_restated_dimensions_keep_independent_sort` (saved `orderby:
[["SavedRevenue", false]]`, update restates `dimensions: [{"name": "Region"}]`)
fails before the fix with "Merged Bullet Chart configuration is invalid." and
passes after.
##########
superset/mcp_service/chart/chart_utils.py:
##########
@@ -1715,6 +1758,847 @@ def map_histogram_config(config:
"HistogramChartConfig") -> Dict[str, Any]:
return form_data
+def _bullet_token_list(values: Sequence[str | int | float]) -> str:
+ """Serialize typed Bullet controls to the frontend's comma-separated
form."""
+ tokens: list[str] = []
+ for value in values:
+ if isinstance(value, float):
+ token = repr(value)
+ # ``100`` parses back to the same binary float as ``100.0`` and
+ # preserves the frontend's established compact integer spelling.
+ if token.endswith(".0") and not (
+ value == 0.0 and math.copysign(1.0, value) < 0
+ ):
+ token = token[:-2]
+ tokens.append(token)
+ else:
+ tokens.append(str(value))
+ return ",".join(tokens)
+
+
+def map_bullet_config(config: BulletChartConfig) -> Dict[str, Any]: # noqa:
C901
+ """Map typed Bullet config to ``Bullet/buildQuery`` and transformProps.
+
+ The frontend buildQuery replaces the generic query fields with exactly one
+ metric and the groupby hierarchy. Presentation controls stay in native
+ snake_case form_data; the chart plugin camelizes them for transformProps.
+ """
+ if (
+ config.dimensions is None
+ and config._inherited_groupby is None
+ and config.order_by
+ ):
+ # An update resolves its saved hierarchy before mapping. Without one,
+ # creation must validate sort targets against an empty hierarchy.
+ BulletChartConfig.model_validate(
+ {**config.model_dump(exclude_unset=True), "dimensions": []}
+ )
+ metric = create_metric_object(config.metric)
+ form_data: Dict[str, Any] = {
+ "viz_type": "bullet",
+ "metric": metric,
+ }
+
+ # Optional semantic/query fields are emitted only when explicitly supplied.
+ # This lets update_chart and update_chart_preview preserve native saved
state,
+ # while an explicit empty value still clears it through the generic merge
path.
+ if "dimensions" in config.model_fields_set:
+ form_data["groupby"] = [dimension.name for dimension in
config.dimensions or []]
+ if "row_limit" in config.model_fields_set:
Review Comment:
Closed by 1a7455b21d4d6339beb650b6f093c4f7f55c5084 (same issue as the later
thread on this line): `map_bullet_config` always writes `row_limit` (10000 when
omitted), and the Bullet update merge restores the saved value when the update
omits `row_limit`. Covered by
`test_bullet_mapper_preserves_omission_and_honors_explicit_values` and
`test_bullet_update_merge_row_limit_omission_and_explicit_value`.
##########
superset/mcp_service/chart/preview_utils.py:
##########
@@ -1398,6 +2412,101 @@ def fallback_vega_lite_preview(
return None
+def generate_xy_pivot_vega_lite_preview(
+ data: list[dict[str, Any]], form_data: dict[str, Any], *, mark: str
+) -> VegaLitePreview | None:
+ """Render flattened timeseries pivot columns without dropping grouped
series.
+
+ Folding escaped field paths resolves literal output keys without splitting
+ category values that contain escaped commas. The legend retains each
+ complete metric/category label.
+ Long-form results continue through the generic renderer.
+ """
+ from superset.mcp_service.chart.chart_helpers import _as_list
+ from superset.utils.pandas_postprocessing.utils import (
+ escape_separator,
+ FLAT_COLUMN_SEPARATOR,
+ )
+
+ if not data:
+ return None
+ dimensions = [
+ label
+ for column in _as_list(form_data.get("groupby"))
+ if (label := _form_column_label(column))
+ ]
+ if not dimensions or any(label in data[0] for label in dimensions):
+ return None
+ x_axis = _form_column_label(form_data.get("x_axis")) or "__timestamp"
+ if x_axis not in data[0]:
+ return None
+ metric_labels = [
+ escape_separator(label)
Review Comment:
Confirmed: `query_context_processor` unescapes flattened column names, so
the result key is `Revenue, net, EU` while the prefix was built from `Revenue\,
net`. Fixed in 4211e3aad46e2fe8d8e528b348ed4eb875982ae2: field discovery
matches the unescaped label (and still the escaped spelling from raw
post-processing output), and the y-axis title uses the plain label; escaping
stays only in the Vega fold field paths. New
`test_xy_preview_renders_series_for_metric_label_with_separator` unescapes the
real post-processed columns as the processor does and asserts both `Revenue,
net, EU`/`Revenue, net, US` series fold; it fails without the fix.
##########
superset/mcp_service/chart/chart_helpers.py:
##########
@@ -760,7 +1639,56 @@ def build_single_query_dict(
order_desc if order_desc is not None else
form_data.get("order_desc", True)
)
qd["orderby"] = [(sort_metric, not descending)]
+ if orderby:
+ qd["orderby"] = orderby
+ for key in (
+ "annotation_layers",
+ "row_offset",
+ "series_columns",
+ "group_others_when_limit_reached",
+ "is_timeseries",
+ "time_offsets",
+ "time_compare_full_range",
+ ):
+ if key in form_data and form_data[key] is not None:
+ qd[key] = form_data[key]
+
+ # ``buildQueryObject`` keeps the modern series-limit controls, falls back
+ # to their legacy Timeseries names, and defaults the limit to zero. A
+ # malformed modern metric does not mask a valid legacy metric.
+ series_limit = form_data.get("series_limit")
+ if series_limit is None:
+ series_limit = form_data.get("limit")
+ if series_limit is not None:
+ qd["series_limit"] = series_limit
+ series_limit_metric = form_data.get("series_limit_metric")
+ if not _is_query_form_metric(series_limit_metric):
+ series_limit_metric = form_data.get("timeseries_limit_metric")
+ if series_limit_metric is not None:
+ qd["series_limit_metric"] = series_limit_metric
+ if apply_chart_fields is not None:
+ apply_chart_fields(qd, effective_row_limit)
apply_form_data_filters_to_query(qd, form_data)
+ # Mirror the common ``buildQueryObject``/``extractExtras`` translation used
+ # by native frontend plugins. ``granularity_sqla`` is a form-data control,
+ # while QueryObject calls the field ``granularity``; the SQL time grain is
+ # carried inside ``extras`` rather than as a top-level query field.
+ if include_common_temporal:
+ granularity = form_data.get("granularity") or
form_data.get("granularity_sqla")
+ if granularity:
+ qd["granularity"] = granularity
+ if time_grain := form_data.get("time_grain_sqla"):
+ qd["extras"] = {
+ **(qd.get("extras") or {}),
+ "time_grain_sqla": time_grain,
Review Comment:
Confirmed: the MCP Table builder copied a cached `time_grain_sqla` into
extras with no temporal column, and the semantic layer rejects it
(`superset/semantic_layers/mapper.py`, "A time column must be specified when a
time grain is provided."). Fixed in 951a56289db79cd1dbfbb6e788d1987a869e7eaf
(plus the mypy narrowing in f92dbdfe0d6b169b9af9e69c6809515836ffb1f5):
`build_table_query_dicts` ports `omitDormantGrain` for semantic-view aggregate
mode, applied to every final query including totals/rowcount, with the same
duration-grain set and the same
`granularity`/`is_timeseries`/all-columns-`false`-in-`temporal_columns_lookup`
conditions. New `test_semantic_table_aggregate_omits_dormant_time_grain`
asserts the grain is dropped for `semantic_view` and kept for `table` and for a
temporal groupby; it fails without the fix.
##########
docs/admin_docs/configuration/mcp-server.mdx:
##########
@@ -1408,6 +1421,129 @@ Disabling a plugin only stops new charts of that type
from being created. Existi
- **[Security](/developer-docs/extensions/security)** -- Security best
practices for extensions
- **[Deployment](/developer-docs/extensions/deployment)** -- Package and
deploy Superset extensions
+## Bullet chart compatibility
+
+The MCP Bullet plugin uses `chart_type: "bullet"` and the native ECharts
+`viz_type: "bullet"`. Its optional `dimensions` hierarchy maps to `groupby`.
+Omit `dimensions` (or use `null`) to create a single-metric Bullet without a
+breakdown. On updates, omission or `null` preserves the saved hierarchy; use
+`dimensions: []` to clear it explicitly. An `order_by` update can reference the
+saved dimensions without resending them; unknown targets are rejected after
+resolving the saved hierarchy. When replacing the dataset, saved-chart and
+cached-preview updates retain an omitted hierarchy only if its columns resolve
+in the replacement dataset; incompatible inherited roles and temporal-filter
+provenance are discarded.
+
+Inherited native SQL dimensions remain query expressions when updating metrics
+or `order_by`; reference their output labels to sort by them. Caller-supplied
+`dimensions` must still be physical columns, not SQL expressions.
+
+Dimension and metric output names are case-sensitive: quoted physical columns
+such as `Region` and `region` remain distinct. Reference lookup prefers exact
+names and uses case-insensitive matching only when there is a single candidate;
+ambiguous references require the exact spelling. Bullet metric strings use
+JavaScript numeric spellings: underscore separators and non-ASCII digits are
+rejected rather than interpreted as numbers. Bounded array-valued dimensions
+retain their raw values in data reads and exports; preview category labels use
+JavaScript string conversion (for example, `[1, 2]` displays as `1,2`). Native
+SQL metrics without a label use their SQL expression as the output label.
+
+Presentation-only updates preserve saved predicates, including legacy top-level
+`where`, `having`, and `filters`, and normalize the native `order_by_cols`
alias.
+When both `orderby` and `order_by_cols` are saved, their sort entries are
+concatenated in form-data key order, including when `orderby` is empty.
+Explicit `filters: []` and `order_by: []` clear those inherited controls.
+
+Range, marker, and marker-line label lists may be shorter than their value
lists.
+Labels that become empty after sanitization retain their slots so later labels
+remain aligned with their corresponding values.
+As in Explore, missing or empty range labels are not displayed, and missing or
+empty marker labels use the formatted numeric value. Extra labels have no value
+to annotate and are ignored. These rules also apply to saved-chart previews and
+updates. Omitted label controls preserve the saved state only when their
+corresponding value controls (`ranges`, `markers`, or `marker_lines`) are also
+omitted. Replacing a value control without its labels clears the saved labels;
+resend the labels to retain annotations with the replacement values.
+
+Saved native `ranges`, `markers`, and `marker_lines` controls ignore empty,
+non-numeric, and NaN tokens, matching Explore. If no numeric range remains, the
+preview uses the default band up to 110% of the largest measure. Unrelated
+updates preserve these saved controls. Newly authored typed lists must contain
+finite numbers; infinite native values and oversized controls remain errors.
+
+On same-dataset chart updates, `filters: []` clears both user filters and the
+generated dashboard-time binding. `temporal_column: null` clears only that
+generated binding, preserving user filters. Omitting those controls preserves
+the saved binding. Bullet Vega previews keep separate indexed rows even when
+dimension display labels are identical, and support both `SMART_NUMBER` and
+`SMART_NUMBER_SIGNED` number formats. Bullet preview numeric format
+precision is limited to 20 digits before formatting; raw data reads do not
+validate presentation formats.
+
+Native Bullet temporal filters retain the active filter's subject and range as
a
+pair; `No filter` placeholders do not supply the subject of another active
range.
+Typed configs support one such pair. Multiple active native temporal filters,
or
+conflicts with explicit `temporal_column`/`time_range`, are rejected rather
than
+silently dropping or moving a predicate.
+
+### Query result limits
+
+MCP chart tools accept up to **50,000 rows per query** and **100,000 total
rows**
+across all queries in one result. The aggregate counts each returned data row,
+including repeated rows in different queries; it does not use `rowcount` or
+`total_rows` metadata. The row-shaped `indexnames`
+array emitted by Chart Data uses the same 50,000-entry limit, rather than the
+4,096-entry limit for other metadata arrays. Index entries count toward the
+shared row-data work budget and retain the 1 MiB aggregate metadata byte limit.
+Multi-dimension pivot index tuples are serialized as JSON arrays. Nested
+containers within an index entry retain the standard container limits.
+
+The shared query-result validator also enforces **non-configurable hard
limits**:
+
+- **2,500,000 values/containers across all queries**. Each row object, cell
+ scalar (including null), and nested list, tuple, array, or object contributes
+ to this shared work budget; nested elements count individually and repeated
+ occurrences count again. Object keys count toward byte limits, not this value
+ count. Row-shaped `indexnames` and their entries also consume this budget.
+ For example, 50,000 rows with 50 scalar columns require 2,550,000 values
+ (50,000 row objects plus 2,500,000 cells) and exceed the budget even if their
+ encoded size is below 16 MiB.
+- **64 KiB (65,536 UTF-8 bytes) per row-cell string value**. A single
+ 65,537-byte cell fails even in a one-row result. Binary cells must also fit
+ the cell limit before and after conversion to UTF-8 or a `base64:` string.
+- **16 MiB (16,777,216 bytes) of aggregate JSON-encoded result data and
+ metadata** across all queries, including escaped strings, keys, and container
+ syntax. Metadata strings (including SQL text) are not subject to the row-cell
+ string cap; all metadata shares a separate 1 MiB aggregate allowance.
+
+These limits apply before rendering or exporting in `get_chart_data`,
+`get_chart_preview`, chart generation/update compile checks, `query_dataset`,
+and semantic-layer `get_table`. They include inline/table responses and MCP
+CSV, Excel, and Parquet export paths; exports do not bypass source validation.
Review Comment:
Agreed: `GetChartDataRequest.format` is `Literal["json", "csv", "excel"]`
and no MCP tool exports Parquet. Fixed in
49dee0e733ca17d06cdfd18fa614489266d67514: both
`docs/admin_docs/configuration/mcp-server.mdx` and `UPDATING.md` now say the
`get_chart_data` CSV and Excel export formats.
##########
superset/mcp_service/chart/chart_helpers.py:
##########
@@ -809,10 +1738,564 @@ def build_mixed_timeseries_secondary(
return qd
-# Deck.gl viz types that conditionally set is_timeseries from time_grain_sqla
-_DECK_TIMESERIES_VIZ_TYPES: frozenset[str] = frozenset(
- {"deck_arc", "deck_path", "deck_polygon", "deck_scatter",
"deck_screengrid"}
-)
+def build_histogram_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Histogram buildQuery, including its histogram post-processing."""
+ column = form_data.get("column")
+ histogram_groupby = _as_list(form_data.get("groupby"))
+ query = build_single_query_dict(
+ form_data,
+ [*histogram_groupby, column] if column is not None else
histogram_groupby,
+ [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ having_filter = bool(form_data.get("having")) or any(
+ isinstance(filter_, dict) and filter_.get("clause") == "HAVING"
+ for filter_ in form_data.get("adhoc_filters") or []
+ )
+ if having_filter:
+ query["metrics"] = [
+ {
+ "expressionType": "SQL",
+ "sqlExpression": "COUNT(*)",
+ "label": "COUNT(*)",
+ }
+ ]
+ bins = form_data.get("bins", 5)
+ try:
+ parsed_bins = float(bins)
+ parsed_bins = int(parsed_bins) if parsed_bins.is_integer() else
parsed_bins
+ except (TypeError, ValueError):
+ parsed_bins = 5
+ query["post_processing"] = [
+ {
+ "operation": "histogram",
+ "options": {
+ "column": _column_label(column),
+ "groupby": [
+ label
+ for item in histogram_groupby
+ if (label := _column_label(item))
+ ],
+ "bins": parsed_bins,
+ "cumulative": bool(form_data.get("cumulative")),
+ "normalize": bool(form_data.get("normalize")),
+ },
+ }
+ ]
+ return [query]
+
+
+def build_box_plot_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Box Plot buildQuery, including its boxplot post-processing."""
+ distribute = _as_list(form_data.get("columns"))
+ if not distribute and form_data.get("granularity_sqla"):
+ distribute = [form_data["granularity_sqla"]]
+ box_groupby = _as_list(form_data.get("groupby"))
+ query = build_single_query_dict(
+ form_data,
+ [
+ *(_temporal_column(column, form_data) for column in distribute),
+ *box_groupby,
+ ],
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ query["series_columns"] = box_groupby
+ if whisker := form_data.get("whiskerOptions"):
+ whisker_type = "tukey"
+ percentiles: list[int] | None = None
+ if whisker == "Min/max (no outliers)":
+ whisker_type = "min/max"
+ elif match := re.fullmatch(r"(\d{1,3})/(\d{1,3}) percentiles",
str(whisker)):
+ whisker_type = "percentile"
+ percentiles = [int(match.group(1)), int(match.group(2))]
+ elif whisker != "Tukey":
+ raise ValueError(f"Unsupported whisker type: {whisker}")
+ query["post_processing"] = [
+ {
+ "operation": "boxplot",
+ "options": {
+ "whisker_type": whisker_type,
+ "percentiles": percentiles,
+ "groupby": [
+ label
+ for column in box_groupby
+ if (label := _column_label(column))
+ ],
+ "metrics": [
+ label
+ for metric in query["metrics"]
+ if (label := _metric_label(metric))
+ ],
+ },
+ }
+ ]
+ return [query]
+
+
+def build_pivot_table_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Pivot Table buildQuery, including subtotal grouping sets."""
+ rows = _as_list(form_data.get("groupbyRows"))
+ pivot_columns = _as_list(form_data.get("groupbyColumns"))
+ if form_data.get("transposePivot"):
+ rows, pivot_columns = pivot_columns, rows
+ columns = _dedupe_query_fields([*rows, *pivot_columns], _column_label)
+ query = build_single_query_dict(
+ form_data,
+ [_temporal_column(column, form_data) for column in columns],
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ )
+ sort_metric = query.get("series_limit_metric")
+ if sort_metric is None and query["metrics"]:
+ sort_metric = query["metrics"][0]
+ if sort_metric is not None:
+ query["orderby"] = [[sort_metric, not query.get("order_desc", True)]]
+ if grouping_sets := _pivot_grouping_sets(form_data, rows, pivot_columns):
+ query["grouping_sets"] = grouping_sets
+ return [query]
+
+
+def build_pie_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ contribution: bool,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Pie/Sunburst buildQuery; Pie adds a contribution operator."""
+ metric = form_data.get("metric")
+ query = build_single_query_dict(
+ form_data,
+ _as_list(form_data.get("groupby")),
+ [metric] if metric is not None else [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ if form_data.get("sort_by_metric") and metric is not None:
+ query["orderby"] = [[metric, False]]
+ if contribution and (label := _metric_label(metric)):
+ query["post_processing"] = [
+ {
+ "operation": "contribution",
+ "options": {
+ "columns": [label],
+ "rename_columns": [f"{label}__contribution"],
+ },
+ }
+ ]
+ return [query]
+
+
+def _positive_int(value: Any) -> int:
+ """Coerce a stored limit (int, numeric string, or empty) to a positive int
or 0."""
+ try:
+ coerced = int(value)
+ except (TypeError, ValueError):
+ return 0
+ return coerced if coerced > 0 else 0
+
+
+def build_table_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Table buildQuery: percent metrics, comparisons, totals,
paging."""
+ raw_mode = form_data.get("query_mode") == "raw" or (
+ form_data.get("query_mode") not in {"raw", "aggregate"}
+ and bool(form_data.get("all_columns"))
+ )
+ # Native extractQueryFields excludes empty-string column references.
+ table_columns = [
+ column
+ for column in _as_list(
+ form_data.get("all_columns") if raw_mode else
form_data.get("groupby")
+ )
+ if column != ""
+ ]
+ table_metrics = [] if raw_mode else _as_list(form_data.get("metrics"))
+ percent_metrics = [] if raw_mode else
_as_list(form_data.get("percent_metrics"))
+ query_metrics = _dedupe_query_fields(
+ [*table_metrics, *percent_metrics], _metric_label
+ )
+ table_orderby = _parse_orderby(form_data.get("order_by_cols"))
+ if not raw_mode:
+ sort_metrics = _as_list(form_data.get("timeseries_limit_metric"))
+ if sort_metrics:
+ table_orderby = [[sort_metrics[0], not form_data.get("order_desc",
False)]]
+ elif table_metrics:
+ table_orderby = [[table_metrics[0], False]]
+ query = build_single_query_dict(
+ form_data,
+ table_columns,
+ query_metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=table_orderby,
+ )
+ if not raw_mode:
+ # Table selects one temporal axis and places it before the other roles.
+ for index, column in enumerate(table_columns):
+ temporal_column = _temporal_column(column, form_data)
+ if temporal_column is not column:
+ query["columns"] = [
+ temporal_column,
+ *table_columns[:index],
+ *table_columns[index + 1 :],
+ ]
+ break
+ # Native comparisons use ordinary metrics, before percentage-only metrics
+ # are added to the selected query and contribution operator.
+ has_time_comparison = _time_comparison(form_data, table_metrics)
+ offsets = _table_time_offsets(form_data, {**query, "metrics":
table_metrics})
+ query["time_offsets"] = offsets
+ post_processing: list[dict[str, Any]] = []
+ contribution: dict[str, Any] | None = None
+ if percent_metrics:
+ labels: list[str] = []
+ for metric in percent_metrics:
+ if label := _metric_label(metric):
+ candidates = [label]
+ if has_time_comparison:
+ candidates.extend(f"{label}__{offset}" for offset in
offsets)
+ for candidate in candidates:
+ if candidate not in labels:
+ labels.append(candidate)
+ contribution = {
+ "operation": "contribution",
+ "options": {
+ "columns": labels,
+ "rename_columns": [f"%{label}" for label in labels],
+ },
+ }
+ post_processing.append(contribution)
+ if has_time_comparison and offsets and form_data.get("comparison_type") !=
"values":
+ source: list[str] = []
+ shifted: list[str] = []
+ for metric in table_metrics:
+ if label := _metric_label(metric):
+ for offset in offsets:
+ source.append(label)
+ shifted.append(f"{label}__{offset}")
+ post_processing.append(
+ {
+ "operation": "compare",
+ "options": {
+ "source_columns": source,
+ "compare_columns": shifted,
+ "compare_type": form_data.get("comparison_type"),
+ "drop_original_columns": True,
+ },
+ }
+ )
+ query["post_processing"] = post_processing
+
+ # ``query["row_limit"]`` is the normalized caller limit (explicit request
+ # limit or the saved row_limit, which may be stored as a string); page
+ # sizing narrows it but never replaces it.
+ configured_limit = _positive_int(query.get("row_limit"))
+ if form_data.get("server_pagination"):
+ if page_size := _positive_int(form_data.get("server_page_length")):
+ query["row_limit"] = (
+ min(page_size, configured_limit) if configured_limit else
page_size
+ )
+ query["row_offset"] = 0
+
+ extra_queries: list[dict[str, Any]] = []
+ if form_data.get("percent_metric_calculation") == "all_records" and
percent_metrics:
+ extra_queries.append(
+ {
+ **query,
+ "columns": [],
+ "metrics": percent_metrics,
+ "post_processing": [],
+ "row_limit": 0,
+ "row_offset": 0,
+ "orderby": [],
+ "is_timeseries": False,
+ }
+ )
+ if query_metrics and form_data.get("show_totals") and not raw_mode:
+ totals = {
+ **query,
+ "columns": [],
+ "metrics": _table_totals_metrics(
+ query_metrics, form_data.get("totals_aggregate")
+ ),
+ "row_limit": 0,
+ "row_offset": 0,
+ "post_processing": [contribution] if contribution else [],
+ }
+ totals.pop("orderby", None)
+ totals.pop("order_desc", None)
+ extra_queries.append(totals)
+ if form_data.get("server_pagination"):
+ rowcount = {
+ **query,
+ "time_offsets": [],
+ "row_limit": configured_limit or 0,
+ "row_offset": 0,
+ "post_processing": [],
+ "is_rowcount": True,
+ }
+ return [query, rowcount, *extra_queries]
+ return [query, *extra_queries]
+
+
+def build_gantt_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Gantt buildQuery with its interval columns and series."""
+ (
+ gantt_columns,
+ gantt_metrics,
+ gantt_orderby,
+ gantt_groupby,
+ ) = resolve_gantt_query_fields(form_data)
+ query = build_single_query_dict(
+ form_data,
+ gantt_columns,
+ gantt_metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=gantt_orderby,
+ )
+ query["series_columns"] = gantt_groupby
+ return [query]
+
+
+def build_interactive_pivot_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Interactive Pivot Table buildQuery."""
+ interactive_columns = [
+ _temporal_column(column, form_data)
+ for column in _as_list(form_data.get("groupby"))
+ ]
+ query = build_single_query_dict(
+ form_data,
+ interactive_columns,
+ list(form_data.get("metrics") or []),
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ _normalize_orderby(query)
+ return [query]
+
+
+def build_big_number_query_dicts( # noqa: C901
+ form_data: dict[str, Any],
+ *,
+ trendline: bool,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Big Number (with or without trendline) buildQuery."""
+ metric = form_data.get("metric")
+ if metric is None:
+ plural_metrics = _as_list(form_data.get("metrics"))
+ metric = plural_metrics[0] if plural_metrics else None
+ columns = _resolve_big_number_query_columns(form_data) if trendline else []
+ query = build_single_query_dict(
+ form_data,
+ columns,
+ [metric] if metric is not None else [],
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=form_data.get("orderby"),
+ )
+ if trendline:
+ # Big Number has no series dimension. Its frontend pivot receives
+ # the common base QueryObject (whose columns are empty), not the
+ # final QueryObject after the explicit x-axis is added. Preserve
+ # that distinction instead of falling back to the final columns.
+ query["series_columns"] = []
+ if not form_data.get("x_axis"):
+ query["is_timeseries"] = True
+ query["post_processing"] = _timeseries_post_processing(form_data,
query)
+ if form_data.get("aggregation") == "raw":
+ return [
+ query,
+ {
+ **query,
+ "columns": [],
+ "is_timeseries": False,
+ "post_processing": [],
+ },
+ ]
+ return [query]
+
+
+def build_waterfall_query_dicts(
+ form_data: dict[str, Any],
+ *,
+ engine: str,
+ row_limit: int | None,
+ order_desc: bool | None,
+) -> list[dict[str, Any]]:
+ """Render Waterfall buildQuery with raw-axis ordering."""
+ metrics, groupby = resolve_metrics_and_groupby(form_data)
+ # normalizeTimeColumn runs after Waterfall's buildQuery callback. It
+ # wraps only the final x-axis column; orderby deliberately retains the
+ # raw control value produced inside the callback.
+ raw_axis = form_data.get("x_axis") or form_data.get("granularity_sqla")
+ query_axis = (
+ _normalized_x_axis_query_field(form_data)
+ if form_data.get("x_axis")
+ else raw_axis
+ )
+ waterfall_columns = ([query_axis] if query_axis else []) + groupby
+ raw_ordering_columns = ([raw_axis] if raw_axis else []) + groupby
+ query = build_single_query_dict(
+ form_data,
+ waterfall_columns,
+ metrics,
+ row_limit=row_limit,
+ order_desc=order_desc,
+ orderby=None,
+ )
+ query["orderby"] = [[column, True] for column in raw_ordering_columns]
+ if form_data.get("x_axis"):
+ query.pop("is_timeseries", None)
+ return [query]
+
+
+def build_mixed_timeseries_query_dicts( # noqa: C901
Review Comment:
Agreed: neither `build_mixed_timeseries_secondary` nor `with_x_axis_column`
has a caller in `superset/` or `tests/`. Removed in
9989731953eebcbb9ea154db115ebcadfa74e22c; their remaining helpers
(`build_single_query_dict`, `extract_x_axis_col`,
`split_adhoc_filters_into_base_filters`) still have other callers, and
ruff/mypy pass.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]