bito-code-review[bot] commented on code in PR #38807:
URL: https://github.com/apache/superset/pull/38807#discussion_r3611071251
##########
superset/daos/dataset.py:
##########
@@ -666,6 +668,346 @@ def get_filterable_columns_and_operators(cls) ->
Dict[str, List[str]]:
filterable.update(DATASET_CUSTOM_FIELDS)
return filterable
+ @staticmethod
+ def get_rls_filters_for_datasets(
+ dataset_ids: list[int],
+ ) -> dict[int, list[dict[str, Any]]]:
+ """
+ Return a mapping of dataset_id -> list of RLS filter summaries
+ for the given dataset IDs. Only returns datasets that have at least
+ one RLS filter attached. For virtual datasets, also includes RLS
+ filters from physical tables referenced in the dataset's SQL.
+ """
+
+ if not dataset_ids:
+ return {}
+
+ # Get direct RLS filters for all requested datasets
+ rows = (
+ db.session.query(
+ RLSFilterTables.c.table_id,
+ RowLevelSecurityFilter.id,
+ RowLevelSecurityFilter.name,
+ RowLevelSecurityFilter.filter_type,
+ RowLevelSecurityFilter.group_key,
+ )
+ .join(
+ RowLevelSecurityFilter,
+ RLSFilterTables.c.rls_filter_id == RowLevelSecurityFilter.id,
+ )
+ .filter(RLSFilterTables.c.table_id.in_(dataset_ids))
+ .all()
+ )
+
+ result: dict[int, list[dict[str, Any]]] = {}
+ for table_id, rls_id, name, filter_type, group_key in rows:
+ result.setdefault(table_id, []).append(
+ {
+ "id": rls_id,
+ "name": name,
+ "filter_type": filter_type,
+ "group_key": group_key,
+ }
+ )
+
+ # For virtual datasets, also check underlying physical tables
+ virtual_datasets = (
+ db.session.query(
+ SqlaTable.id, SqlaTable.sql, SqlaTable.schema,
SqlaTable.database_id
+ )
+ .filter(SqlaTable.id.in_(dataset_ids), SqlaTable.sql.isnot(None))
# type: ignore[attr-defined,unused-ignore]
+ .all()
+ )
+
+ if virtual_datasets:
+ inherited = DatasetDAO._get_inherited_rls_for_virtual_datasets(
+ virtual_datasets
+ )
+ for ds_id, filters in inherited.items():
+ existing_ids = {f["id"] for f in result.get(ds_id, [])}
+ for f in filters:
+ if f["id"] not in existing_ids:
+ existing_ids.add(f["id"])
+ result.setdefault(ds_id, []).append(f)
+
+ return result
+
+ @staticmethod
+ def _parse_tables_from_virtual_datasets(
+ virtual_datasets: list[tuple[int, str, str | None, int]],
+ db_engines: dict[int, str] | None = None,
+ ) -> tuple[dict[int, set[Table]], dict[int, int]]:
+ """
+ Parse SQL from virtual datasets and return:
+ - ds_to_tables: mapping of dataset_id -> set of referenced Table
objects
+ (with schema/catalog preserved from the SQL, unqualified refs
resolved
+ to the virtual dataset's own schema)
+ - ds_db_map: mapping of dataset_id -> database_id
+ """
+ if db_engines is None:
+ db_engines = {}
+ ds_to_tables: dict[int, set[Table]] = {}
+ ds_db_map: dict[int, int] = {}
+ for ds_id, sql, default_schema, database_id in virtual_datasets:
+ ds_db_map[ds_id] = database_id
+ engine = db_engines.get(database_id, "")
+ try:
+ parsed = SQLScript(sql, engine=engine)
+ table_refs: set[Table] = set()
+ for statement in parsed.statements:
+ for table_ref in statement.tables:
+ # Qualify unqualified references with the virtual
dataset's
+ # own schema so we match the correct physical dataset.
+
table_refs.add(table_ref.qualify(schema=default_schema))
+ if table_refs:
+ ds_to_tables[ds_id] = table_refs
+ except Exception: # noqa: BLE001
+ logger.warning(
+ "Failed to parse SQL for virtual dataset %d", ds_id,
exc_info=True
+ )
+ return ds_to_tables, ds_db_map
+
+ @staticmethod
+ def _fetch_physical_rls_map(
+ all_tables: set[Table],
+ db_ids: set[int],
+ ) -> tuple[dict[tuple[str, str | None, int], int], dict[int,
list[dict[str, Any]]]]:
+ """
+ Look up physical datasets matching the given Table objects and
database IDs,
+ then fetch their RLS filters.
+
+ Returns:
+ - physical_map: (table_name, schema, database_id) -> physical dataset
id
+ - phys_rls: physical dataset id -> list of RLS filter summaries
+ """
+
+ all_table_names = {t.table for t in all_tables}
+ physical_tables = (
+ db.session.query(
+ SqlaTable.id,
+ SqlaTable.table_name,
+ SqlaTable.schema,
+ SqlaTable.database_id,
+ )
+ .filter(
+ SqlaTable.table_name.in_(all_table_names),
+ SqlaTable.database_id.in_(db_ids),
+ SqlaTable.sql.is_(None),
+ )
+ .all()
+ )
+
+ physical_map: dict[tuple[str, str | None, int], int] = {}
+ physical_ids: set[int] = set()
+ for phys_id, table_name, schema, db_id in physical_tables:
+ physical_map[(table_name, schema, db_id)] = phys_id
+ physical_ids.add(phys_id)
+
+ if not physical_ids:
+ return physical_map, {}
+
+ rls_rows = (
+ db.session.query(
+ RLSFilterTables.c.table_id,
+ RowLevelSecurityFilter.id,
+ RowLevelSecurityFilter.name,
+ RowLevelSecurityFilter.filter_type,
+ RowLevelSecurityFilter.group_key,
+ )
+ .join(
+ RowLevelSecurityFilter,
+ RLSFilterTables.c.rls_filter_id == RowLevelSecurityFilter.id,
+ )
+ .filter(RLSFilterTables.c.table_id.in_(physical_ids))
+ .all()
+ )
+
+ phys_rls: dict[int, list[dict[str, Any]]] = {}
+ for table_id, rls_id, name, filter_type, group_key in rls_rows:
+ phys_rls.setdefault(table_id, []).append(
+ {
+ "id": rls_id,
+ "name": name,
+ "filter_type": filter_type,
+ "group_key": group_key,
+ }
+ )
+ return physical_map, phys_rls
+
+ @staticmethod
+ def _get_inherited_rls_for_virtual_datasets(
+ virtual_datasets: list[tuple[int, str, str | None, int]],
+ ) -> dict[int, list[dict[str, Any]]]:
+ """
+ For virtual datasets, parse their SQL to find referenced physical
+ tables and return any RLS filters attached to those tables.
+
+ Each tuple is (dataset_id, sql, schema, database_id).
+ """
+ # Batch-fetch database engines for accurate SQL parsing
+ unique_db_ids = {row[3] for row in virtual_datasets}
+ db_engines: dict[int, str] = {}
+ if unique_db_ids:
+ db_objs = (
+ db.session.query(Database)
+ .filter(Database.id.in_(unique_db_ids)) # type:
ignore[attr-defined,unused-ignore]
+ .all()
+ )
+ for database_obj in db_objs:
+ try:
+ db_engines[database_obj.id] = database_obj.backend
+ except Exception: # noqa: BLE001
+ db_engines[database_obj.id] = ""
+
+ ds_to_tables, ds_db_map =
DatasetDAO._parse_tables_from_virtual_datasets(
+ virtual_datasets, db_engines=db_engines
+ )
+
+ if not ds_to_tables:
+ return {}
+
+ all_table_refs: set[Table] = set()
+ db_ids: set[int] = set()
+ for ds_id, table_refs in ds_to_tables.items():
+ all_table_refs.update(table_refs)
+ db_ids.add(ds_db_map[ds_id])
+
+ physical_map, phys_rls = DatasetDAO._fetch_physical_rls_map(
+ all_table_refs, db_ids
+ )
+
+ result: dict[int, list[dict[str, Any]]] = {}
+ for ds_id, table_refs in ds_to_tables.items():
+ database_id = ds_db_map[ds_id]
+ for table_ref in table_refs:
+ phys_id = physical_map.get(
+ (table_ref.table, table_ref.schema, database_id)
+ )
+ if phys_id and phys_id in phys_rls:
+ result.setdefault(ds_id, []).extend(phys_rls[phys_id])
Review Comment:
<!-- Bito Reply -->
The reviewer is correct. Using `extend()` directly on the list of filters
can introduce duplicates if the same RLS filter is associated with multiple
physical tables referenced in the virtual dataset's SQL. To ensure each filter
is only added once, you should check for existing IDs before appending, as
suggested.
**superset/daos/dataset.py**
```
if phys_id and phys_id in phys_rls:
existing_ids = {f["id"] for f in result.get(ds_id, [])}
for f in phys_rls[phys_id]:
if f["id"] not in existing_ids:
existing_ids.add(f["id"])
result.setdefault(ds_id, []).append(f)
```
##########
superset-frontend/src/pages/DatasetList/index.tsx:
##########
@@ -57,6 +57,8 @@ import {
DatasetTypeLabel,
Loading,
List,
+ RlsBadge,
+ type RlsFilterSummary,
Review Comment:
<!-- Bito Reply -->
The suggestion to add test coverage for the RLS badge is a standard best
practice for ensuring new UI components function as expected. Since the RLS
badge is a new feature, verifying its behavior with both positive (filters
present) and negative (no filters) test cases is appropriate to prevent
regressions and ensure reliability. You should proceed with adding these test
fixtures.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]