This is an automated email from the ASF dual-hosted git repository.
jli pushed a commit to branch 6.0
in repository https://gitbox.apache.org/repos/asf/superset.git
The following commit(s) were added to refs/heads/6.0 by this push:
new d5a523db25 fix(databricks): string escaper v2 (#34991)
d5a523db25 is described below
commit d5a523db2546cb3e81ad68f2608252ee71c27286
Author: Vitor Avila <[email protected]>
AuthorDate: Tue Sep 2 18:11:45 2025 -0300
fix(databricks): string escaper v2 (#34991)
(cherry picked from commit 0de5b28716fd4aa7195e22a6d737be8d6768ac55)
---
superset/db_engine_specs/databricks.py | 34 ++++++++++++++++++++++++----------
1 file changed, 24 insertions(+), 10 deletions(-)
diff --git a/superset/db_engine_specs/databricks.py
b/superset/db_engine_specs/databricks.py
index 7f84802080..55ec2d1479 100644
--- a/superset/db_engine_specs/databricks.py
+++ b/superset/db_engine_specs/databricks.py
@@ -25,6 +25,7 @@ from flask_babel import gettext as __
from marshmallow import fields, Schema
from marshmallow.validate import Range
from sqlalchemy import types
+from sqlalchemy.engine.default import DefaultDialect
from sqlalchemy.engine.reflection import Inspector
from sqlalchemy.engine.url import URL
@@ -72,19 +73,32 @@ class DatabricksStringType(types.TypeDecorator):
def monkeypatch_dialect() -> None:
"""
- Monkeypatch dialect to correctly escape single quotes.
+ Monkeypatch dialect to correctly escape single quotes for Databricks.
- The Databricks SQLAlchemy dialect we currently use does not escape single
quotes
- correctly -- it doubles the single quotes, instead of adding a backslash.
The fixed
- version requires SQLAlchemy 2.0, which is not yet available in Superset.
+ The Databricks SQLAlchemy dialect (<3.0) incorrectly escapes single quotes
by
+ doubling them ('O''Hara') instead of using backslash escaping ('O\'Hara').
The
+ fixed version requires SQLAlchemy>=2.0, which is not yet compatible with
Superset.
+
+ Since the DatabricksDialect.colspecs points to the base class
(HiveDialect.colspecs)
+ we can't patch it without affecting other Hive-based dialects. The
solution is to
+ introduce a dialect-aware string type so that the change applies only to
Databricks.
"""
try:
- from sqlalchemy_databricks._dialect import DatabricksDialect
+ from pyhive.sqlalchemy_hive import HiveDialect
+
+ class ContextAwareStringType(types.TypeDecorator):
+ impl = types.String
+ cache_ok = True
+
+ def literal_processor(
+ self, dialect: DefaultDialect
+ ) -> Callable[[Any], str]:
+ if dialect.__class__.__name__ == "DatabricksDialect":
+ return DatabricksStringType().literal_processor(dialect)
+ return super().literal_processor(dialect)
+
+ HiveDialect.colspecs[types.String] = ContextAwareStringType
- # copy because the dictionary is shared with other subclasses if it
hasn't been
- # overwritten
- DatabricksDialect.colspecs = DatabricksDialect.colspecs.copy()
- DatabricksDialect.colspecs[types.String] = DatabricksStringType
except ImportError:
pass
@@ -646,5 +660,5 @@ class
DatabricksPythonConnectorEngineSpec(DatabricksDynamicBaseEngineSpec):
return uri, connect_args
-# remove once we've upgraded to SQLAlchemy 2.0 and the 2.x
databricks-sqlalchemy lib
+# TODO: remove once we've upgraded to SQLAlchemy>=2.0 and
databricks-sql-python>=3.x
monkeypatch_dialect()