This is an automated email from the ASF dual-hosted git repository.

xuang7 pushed a commit to branch release/v1.2
in repository https://gitbox.apache.org/repos/asf/texera.git


The following commit(s) were added to refs/heads/release/v1.2 by this push:
     new 2b6bcda60a fix(query, v1.2): Remove duplicated rows when datasets 
shared Publicly in the hub page  (#6992)
2b6bcda60a is described below

commit 2b6bcda60a6cbbd2901ae5284fcda18f448b7426
Author: Yicong Huang <[email protected]>
AuthorDate: Wed Jul 29 14:13:04 2026 -0400

    fix(query, v1.2): Remove duplicated rows when datasets shared Publicly in 
the hub page  (#6992)
    
    ### What changes were proposed in this PR?
    
    Backport of #6016 to `release/v1.2`, cherry-picked from
    3da5f49c0a3727b1b57756265e84a5957551c036. The cherry-pick applied
    cleanly.
    
    Follows the Direct Backport Push convention; opened as a PR (rather than
    a direct push) per a backport-coverage audit.
    
    ### Any related issues, documentation, discussions?
    
    Backport of #6016. Originally linked #5957.
    
    ### How was this PR tested?
    
    Release-branch CI runs on this PR. Cherry-pick applied cleanly onto
    `release/v1.2`; no manual conflict resolution was needed.
    
    ### Was this PR authored or co-authored using generative AI tooling?
    
    Yes — backport prepared with Claude Code (mechanical cherry-pick; the
    change itself is #6016 by its original author).
    
    🤖 Generated with [Claude Code](https://claude.com/claude-code)
    
    Co-authored-by: Mrudhulraj <[email protected]>
---
 .../dashboard/DatasetSearchQueryBuilder.scala      |  47 +++--
 .../dashboard/file/DatasetResourceSpec.scala       | 228 +++++++++++++++++++++
 2 files changed, 258 insertions(+), 17 deletions(-)

diff --git 
a/amber/src/main/scala/org/apache/texera/web/resource/dashboard/DatasetSearchQueryBuilder.scala
 
b/amber/src/main/scala/org/apache/texera/web/resource/dashboard/DatasetSearchQueryBuilder.scala
index 64c8c31106..0cda3eecdc 100644
--- 
a/amber/src/main/scala/org/apache/texera/web/resource/dashboard/DatasetSearchQueryBuilder.scala
+++ 
b/amber/src/main/scala/org/apache/texera/web/resource/dashboard/DatasetSearchQueryBuilder.scala
@@ -66,30 +66,38 @@ object DatasetSearchQueryBuilder extends SearchQueryBuilder 
with LazyLogging {
       params: DashboardResource.SearchQueryParams,
       includePublic: Boolean = false
   ): TableLike[_] = {
+    // Case 1: if `uid` is (set) and `includePublic` is false
+    // -> return ONLY datasets that given `uid` has explicit access to.
+    // Case 2: if `uid` is (null) and `includePublic` is true
+    // -> return ONLY datasets that are public
+    // Case 3: if `uid` is (set) and `includePublic` is true
+    // -> Union of datasets that are public and explicitly shared with user is 
returned
+    // Case 4: if `uid` is (null) and `includePublic` is false
+    // -> return public datasets by default as user might not be logged in
     val baseJoin = DATASET
       .leftJoin(DATASET_USER_ACCESS)
       .on(DATASET_USER_ACCESS.DID.eq(DATASET.DID))
+      .and(if (uid == null) DSL.falseCondition() else 
DATASET_USER_ACCESS.UID.eq(uid))
       .leftJoin(USER)
       .on(USER.UID.eq(DATASET.OWNER_UID))
 
-    // Default condition starts as true, ensuring all datasets are selected 
initially.
-    var condition: Condition = DSL.trueCondition()
-
-    if (uid == null) {
-      // If `uid` is null, the user is not logged in or performing a public 
search
-      // We only select datasets marked as public
-      condition = DATASET.IS_PUBLIC.eq(true)
-    } else {
-      // When `uid` is present, we add a condition to only include datasets 
with direct user access.
-      val userAccessCondition = DATASET_USER_ACCESS.UID.eq(uid)
-
-      if (includePublic) {
-        // If `includePublic` is true, we extend visibility to public datasets 
as well.
-        condition = userAccessCondition.or(DATASET.IS_PUBLIC.eq(true))
+    // Set the `condition` where clause here
+    val condition: Condition =
+      if (uid == null) {
+        // Case 2 and 4
+        // Get all the public datasets by default
+        DATASET.IS_PUBLIC.eq(true)
       } else {
-        condition = userAccessCondition
+        if (includePublic) {
+          // Case 3
+          // Get all the datasets that `uid` has access to and the public 
datasets
+          DATASET.IS_PUBLIC.eq(true).or(DATASET_USER_ACCESS.UID.isNotNull)
+        } else {
+          // Case 1
+          // If `includePublic` is false get only user accessible datasets
+          DATASET_USER_ACCESS.UID.isNotNull
+        }
       }
-    }
     baseJoin.where(condition)
   }
 
@@ -140,7 +148,12 @@ object DatasetSearchQueryBuilder extends 
SearchQueryBuilder with LazyLogging {
     val dd = DashboardDataset(
       dataset,
       owner.getEmail,
-      record.get(DATASET_USER_ACCESS.PRIVILEGE, classOf[PrivilegeEnum]),
+      Option(
+        record.get(
+          DATASET_USER_ACCESS.PRIVILEGE,
+          classOf[PrivilegeEnum]
+        )
+      ).getOrElse(PrivilegeEnum.NONE),
       dataset.getOwnerUid == uid,
       size
     )
diff --git 
a/amber/src/test/scala/org/apache/texera/web/resource/dashboard/file/DatasetResourceSpec.scala
 
b/amber/src/test/scala/org/apache/texera/web/resource/dashboard/file/DatasetResourceSpec.scala
new file mode 100644
index 0000000000..1d4b5635e0
--- /dev/null
+++ 
b/amber/src/test/scala/org/apache/texera/web/resource/dashboard/file/DatasetResourceSpec.scala
@@ -0,0 +1,228 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements.  See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership.  The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License.  You may obtain a copy of the License at
+ *
+ *   http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied.  See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.texera.web.resource.dashboard.file
+
+import org.apache.texera.auth.SessionUser
+import org.apache.texera.dao.MockTexeraDB
+import org.apache.texera.dao.jooq.generated.enums.UserRoleEnum
+import org.apache.texera.dao.jooq.generated.tables.pojos.User
+import 
org.apache.texera.web.resource.dashboard.DashboardResource.SearchQueryParams
+import 
org.apache.texera.web.resource.dashboard.user.dataset.DatasetResource.DashboardDataset
+import org.apache.texera.web.resource.dashboard.{FulltextSearchQueryUtils}
+import org.apache.texera.web.resource.dashboard.DatasetSearchQueryBuilder
+import org.scalatest.flatspec.AnyFlatSpec
+import org.apache.texera.dao.jooq.generated.tables.daos.{UserDao, DatasetDao, 
DatasetUserAccessDao}
+import org.apache.texera.dao.jooq.generated.enums.PrivilegeEnum
+import org.apache.texera.dao.jooq.generated.tables.pojos.{Dataset, 
DatasetUserAccess}
+import org.scalatest.{BeforeAndAfterAll, BeforeAndAfterEach}
+import java.time.OffsetDateTime
+import java.util
+import 
org.apache.texera.web.resource.dashboard.SearchQueryBuilder.DATASET_RESOURCE_TYPE
+
+class DatasetResourceSpec
+    extends AnyFlatSpec
+    with BeforeAndAfterAll
+    with BeforeAndAfterEach
+    with MockTexeraDB {
+
+  // An example creation time to test Account Creation Time attribute
+  private val exampleCreationTime: OffsetDateTime =
+    OffsetDateTime.parse("2025-01-01T00:00:00Z")
+
+  private val ownerUser: User = {
+    val user = new User
+    user.setUid(Integer.valueOf(1))
+    user.setName("owner_user")
+    user.setRole(UserRoleEnum.ADMIN)
+    user.setEmail("[email protected]")
+    user.setPassword("123")
+    user.setComment("test_comment")
+    user.setAccountCreationTime(exampleCreationTime)
+    user
+  }
+
+  private val testUser: User = {
+    val user = new User
+    user.setUid(Integer.valueOf(2))
+    user.setName("test_user")
+    user.setEmail("[email protected]")
+    user.setRole(UserRoleEnum.REGULAR)
+    user.setPassword("123")
+    user.setComment("test_comment2")
+    user.setAccountCreationTime(exampleCreationTime)
+    user
+  }
+
+  private val testDatasetRecord: Dataset = {
+    val dataset = new Dataset()
+    dataset.setName("test_dataset1")
+    dataset.setDescription("keyword_in_dataset_description")
+    dataset.setIsPublic(true)
+    dataset.setDid(Integer.valueOf(1))
+    dataset
+  }
+
+  private val sessionUser1: SessionUser = {
+    new SessionUser(ownerUser)
+  }
+
+  private val sessionUser2: SessionUser = {
+    new SessionUser(testUser)
+  }
+
+  // get context lazily
+  private lazy val datasetDao: DatasetDao = {
+    new DatasetDao(getDSLContext.configuration())
+  }
+
+  private lazy val datasetUserAccessDao: DatasetUserAccessDao = {
+    new DatasetUserAccessDao(getDSLContext.configuration())
+  }
+
+  override protected def beforeAll(): Unit = {
+    initializeDBAndReplaceDSLContext()
+    FulltextSearchQueryUtils.usePgroonga = false // disable pgroonga
+    // add test user directly
+    val userDao = new UserDao(getDSLContext.configuration())
+    userDao.insert(ownerUser)
+    userDao.insert(testUser)
+  }
+
+  override protected def beforeEach(): Unit = {
+    // Clean up environment before each test case
+  }
+
+  override protected def afterEach(): Unit = {
+    // 1. Delete access rows before the dataset
+    val datasetUserAccessDao = new 
DatasetUserAccessDao(getDSLContext.configuration())
+    getDSLContext
+      
.deleteFrom(org.apache.texera.dao.jooq.generated.tables.DatasetUserAccess.DATASET_USER_ACCESS)
+      .execute()
+    // 2. Fetch all datasets owned by the owner
+    val datasets = datasetDao.fetchByOwnerUid(ownerUser.getUid())
+    if (!datasets.isEmpty) {
+      datasetDao.delete(datasets)
+    }
+  }
+
+  override protected def afterAll(): Unit = {
+    shutdownDB()
+  }
+
+  private def getKeywordsArray(keywords: String*): util.ArrayList[String] = {
+    val keywordsList = new util.ArrayList[String]()
+    for (keyword <- keywords) {
+      keywordsList.add(keyword)
+    }
+    keywordsList
+  }
+
+  private def assertSameDataset(a: Dataset, b: DashboardDataset): Unit = {
+    assert(a.getName == b.dataset.getName)
+  }
+
+  "User.accountCreationTime" should "be persisted and retrievable via UserDao" 
in {
+    val userDao = new UserDao(getDSLContext.configuration())
+    val u1 = userDao.fetchOneByUid(Integer.valueOf(1))
+    val u2 = userDao.fetchOneByUid(Integer.valueOf(2))
+
+    assert(u1.getAccountCreationTime != null)
+    assert(u2.getAccountCreationTime != null)
+
+    assert(u1.getAccountCreationTime.isEqual(exampleCreationTime))
+    assert(u2.getAccountCreationTime.isEqual(exampleCreationTime))
+  }
+
+  it should "remain unchanged when updating unrelated fields" in {
+    val userDao = new UserDao(getDSLContext.configuration())
+    val u1 = userDao.fetchOneByUid(Integer.valueOf(1))
+    val originalTime = u1.getAccountCreationTime
+
+    u1.setComment("updated_comment")
+    userDao.update(u1)
+
+    val test_u1 = userDao.fetchOneByUid(Integer.valueOf(1))
+    assert(test_u1.getAccountCreationTime.isEqual(originalTime))
+  }
+
+  "DatasetResource /owner view" should "get deduplicated datasets created by 
owner and shared publicly" in {
+    // Only metadatas of dataset and the user is maintained - no dataset is 
actually created in LakeFS
+    // Create dataset
+    val datasetDao = new DatasetDao(getDSLContext.configuration())
+    testDatasetRecord.setOwnerUid(ownerUser.getUid)
+    datasetDao.insert(testDatasetRecord)
+
+    // Give write access to the owner user
+    datasetUserAccessDao.insert(
+      new DatasetUserAccess(
+        testDatasetRecord.getDid,
+        ownerUser.getUid,
+        PrivilegeEnum.WRITE
+      )
+    )
+
+    // Build the query - bypasses DashboarResource
+    val query =
+      DatasetSearchQueryBuilder.constructQuery(
+        ownerUser.getUid,
+        SearchQueryParams(resourceType = DATASET_RESOURCE_TYPE),
+        includePublic = true
+      )
+    // Assert the length of returned dataset
+    val datasetEntryList = getDSLContext.fetch(query)
+    assert(datasetEntryList.size() == 1)
+  }
+
+  "/search API" should "deduplicate datasets shared both publicly and 
explicitly with NON-WRITE permissions" in {
+    // Only metadatas of dataset and the user is maintained - no dataset is 
actually created in LakeFS
+    // Create dataset
+    val datasetDao = new DatasetDao(getDSLContext.configuration())
+    testDatasetRecord.setOwnerUid(ownerUser.getUid)
+    datasetDao.insert(testDatasetRecord)
+
+    // Give write access to the owner user
+    datasetUserAccessDao.insert(
+      new DatasetUserAccess(
+        ownerUser.getUid,
+        testDatasetRecord.getDid,
+        PrivilegeEnum.WRITE
+      )
+    )
+
+    datasetUserAccessDao.insert(
+      new DatasetUserAccess(
+        testDatasetRecord.getDid,
+        testUser.getUid,
+        PrivilegeEnum.READ
+      )
+    )
+
+    // Build the query
+    val query =
+      DatasetSearchQueryBuilder.constructQuery(
+        testUser.getUid,
+        SearchQueryParams(resourceType = DATASET_RESOURCE_TYPE),
+        includePublic = true
+      )
+    // Assert the length of returned dataset
+    val datasetEntryList = getDSLContext.fetch(query)
+    assert(datasetEntryList.size() == 1)
+  }
+}

Reply via email to