This is an automated email from the ASF dual-hosted git repository.

jli pushed a commit to branch 6.0
in repository https://gitbox.apache.org/repos/asf/superset.git


The following commit(s) were added to refs/heads/6.0 by this push:
     new d5a523db25 fix(databricks): string escaper v2 (#34991)
d5a523db25 is described below

commit d5a523db2546cb3e81ad68f2608252ee71c27286
Author: Vitor Avila <[email protected]>
AuthorDate: Tue Sep 2 18:11:45 2025 -0300

    fix(databricks): string escaper v2 (#34991)
    
    (cherry picked from commit 0de5b28716fd4aa7195e22a6d737be8d6768ac55)
---
 superset/db_engine_specs/databricks.py | 34 ++++++++++++++++++++++++----------
 1 file changed, 24 insertions(+), 10 deletions(-)

diff --git a/superset/db_engine_specs/databricks.py 
b/superset/db_engine_specs/databricks.py
index 7f84802080..55ec2d1479 100644
--- a/superset/db_engine_specs/databricks.py
+++ b/superset/db_engine_specs/databricks.py
@@ -25,6 +25,7 @@ from flask_babel import gettext as __
 from marshmallow import fields, Schema
 from marshmallow.validate import Range
 from sqlalchemy import types
+from sqlalchemy.engine.default import DefaultDialect
 from sqlalchemy.engine.reflection import Inspector
 from sqlalchemy.engine.url import URL
 
@@ -72,19 +73,32 @@ class DatabricksStringType(types.TypeDecorator):
 
 def monkeypatch_dialect() -> None:
     """
-    Monkeypatch dialect to correctly escape single quotes.
+    Monkeypatch dialect to correctly escape single quotes for Databricks.
 
-    The Databricks SQLAlchemy dialect we currently use does not escape single 
quotes
-    correctly -- it doubles the single quotes, instead of adding a backslash. 
The fixed
-    version requires SQLAlchemy 2.0, which is not yet available in Superset.
+    The Databricks SQLAlchemy dialect (<3.0) incorrectly escapes single quotes 
by
+    doubling them ('O''Hara') instead of using backslash escaping ('O\'Hara'). 
The
+    fixed version requires SQLAlchemy>=2.0, which is not yet compatible with 
Superset.
+
+    Since the DatabricksDialect.colspecs points to the base class 
(HiveDialect.colspecs)
+    we can't patch it without affecting other Hive-based dialects. The 
solution is to
+    introduce a dialect-aware string type so that the change applies only to 
Databricks.
     """
     try:
-        from sqlalchemy_databricks._dialect import DatabricksDialect
+        from pyhive.sqlalchemy_hive import HiveDialect
+
+        class ContextAwareStringType(types.TypeDecorator):
+            impl = types.String
+            cache_ok = True
+
+            def literal_processor(
+                self, dialect: DefaultDialect
+            ) -> Callable[[Any], str]:
+                if dialect.__class__.__name__ == "DatabricksDialect":
+                    return DatabricksStringType().literal_processor(dialect)
+                return super().literal_processor(dialect)
+
+        HiveDialect.colspecs[types.String] = ContextAwareStringType
 
-        # copy because the dictionary is shared with other subclasses if it 
hasn't been
-        # overwritten
-        DatabricksDialect.colspecs = DatabricksDialect.colspecs.copy()
-        DatabricksDialect.colspecs[types.String] = DatabricksStringType
     except ImportError:
         pass
 
@@ -646,5 +660,5 @@ class 
DatabricksPythonConnectorEngineSpec(DatabricksDynamicBaseEngineSpec):
         return uri, connect_args
 
 
-# remove once we've upgraded to SQLAlchemy 2.0 and the 2.x 
databricks-sqlalchemy lib
+# TODO: remove once we've upgraded to SQLAlchemy>=2.0 and 
databricks-sql-python>=3.x
 monkeypatch_dialect()

Reply via email to