This is an automated email from the ASF dual-hosted git repository.

xiangfu0 pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/pinot.git


The following commit(s) were added to refs/heads/master by this push:
     new eb077bc5a39 [UUID 4/8] Server-side predicate evaluation for the 
logical UUID type (#18872)
eb077bc5a39 is described below

commit eb077bc5a396bdcc607d1db255b995b90d58ea9b
Author: Xiang Fu <[email protected]>
AuthorDate: Sat Aug 8 14:23:35 2026 -0700

    [UUID 4/8] Server-side predicate evaluation for the logical UUID type 
(#18872)
---
 .../request/context/predicate/BaseInPredicate.java |  15 ++
 .../predicate/EqualsPredicateEvaluatorFactory.java |   6 +
 .../predicate/InPredicateEvaluatorFactory.java     |  13 ++
 .../NotEqualsPredicateEvaluatorFactory.java        |   6 +
 .../predicate/NotInPredicateEvaluatorFactory.java  |  13 ++
 .../operator/filter/predicate/PredicateUtils.java  |  20 +++
 .../predicate/RangePredicateEvaluatorFactory.java  |   9 ++
 .../query/reduce/filter/PredicateRowMatcher.java   |   4 +
 .../NoDictionaryEqualsPredicateEvaluatorsTest.java |  43 +++++
 .../NoDictionaryInPredicateEvaluatorTest.java      |  73 +++++++++
 .../NoDictionaryRangePredicateEvaluatorTest.java   |  69 ++++++++
 .../UuidDictionaryPredicateEvaluatorTest.java      | 173 +++++++++++++++++++++
 .../core/query/reduce/HavingFilterHandlerTest.java |  21 +++
 13 files changed, 465 insertions(+)

diff --git 
a/pinot-common/src/main/java/org/apache/pinot/common/request/context/predicate/BaseInPredicate.java
 
b/pinot-common/src/main/java/org/apache/pinot/common/request/context/predicate/BaseInPredicate.java
index 0bbe77c14fe..a5350e46e1a 100644
--- 
a/pinot-common/src/main/java/org/apache/pinot/common/request/context/predicate/BaseInPredicate.java
+++ 
b/pinot-common/src/main/java/org/apache/pinot/common/request/context/predicate/BaseInPredicate.java
@@ -25,6 +25,7 @@ import org.apache.pinot.spi.utils.BooleanUtils;
 import org.apache.pinot.spi.utils.ByteArray;
 import org.apache.pinot.spi.utils.BytesUtils;
 import org.apache.pinot.spi.utils.TimestampUtils;
+import org.apache.pinot.spi.utils.UuidUtils;
 
 
 /// Base predicate for `IN` and `NOT_IN`.
@@ -49,6 +50,7 @@ public abstract class BaseInPredicate extends BasePredicate {
   private volatile int[] _booleanValues;
   private volatile long[] _timestampValues;
   private volatile ByteArray[] _bytesValues;
+  private volatile ByteArray[] _uuidValues;
 
   public BaseInPredicate(ExpressionContext lhs, List<String> values) {
     super(lhs);
@@ -162,4 +164,17 @@ public abstract class BaseInPredicate extends 
BasePredicate {
     }
     return bigDecimalValues;
   }
+
+  public ByteArray[] getUuidValues() {
+    ByteArray[] uuidValues = _uuidValues;
+    if (uuidValues == null) {
+      int numValues = _values.size();
+      uuidValues = new ByteArray[numValues];
+      for (int i = 0; i < numValues; i++) {
+        uuidValues[i] = new ByteArray(UuidUtils.toBytes(_values.get(i)));
+      }
+      _uuidValues = uuidValues;
+    }
+    return uuidValues;
+  }
 }
diff --git 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/EqualsPredicateEvaluatorFactory.java
 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/EqualsPredicateEvaluatorFactory.java
index c2fc2d7477f..e8026fbedb2 100644
--- 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/EqualsPredicateEvaluatorFactory.java
+++ 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/EqualsPredicateEvaluatorFactory.java
@@ -31,6 +31,7 @@ import org.apache.pinot.spi.data.FieldSpec.DataType;
 import org.apache.pinot.spi.utils.BooleanUtils;
 import org.apache.pinot.spi.utils.BytesUtils;
 import org.apache.pinot.spi.utils.TimestampUtils;
+import org.apache.pinot.spi.utils.UuidUtils;
 
 
 /// Factory for EQ predicate evaluators.
@@ -76,6 +77,11 @@ public class EqualsPredicateEvaluatorFactory {
         return new StringRawValueBasedEqPredicateEvaluator(eqPredicate, value);
       case BYTES:
         return new BytesRawValueBasedEqPredicateEvaluator(eqPredicate, 
BytesUtils.toBytes(value));
+      // UUID is a logical type stored as 16 raw bytes, so -- like TIMESTAMP 
over LONG above -- convert the literal to
+      // its stored form and reuse the stored-type evaluator. getDataType() 
then correctly reports the type applySV
+      // consumes (BYTES), per the PredicateEvaluator contract.
+      case UUID:
+        return new BytesRawValueBasedEqPredicateEvaluator(eqPredicate, 
UuidUtils.toBytes(value));
       default:
         throw new IllegalStateException("Unsupported data type: " + dataType);
     }
diff --git 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/InPredicateEvaluatorFactory.java
 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/InPredicateEvaluatorFactory.java
index 5b74eae0186..25e316f7400 100644
--- 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/InPredicateEvaluatorFactory.java
+++ 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/InPredicateEvaluatorFactory.java
@@ -144,6 +144,19 @@ public class InPredicateEvaluatorFactory {
         }
         return new BytesRawValueBasedInPredicateEvaluator(inPredicate, 
matchingValues);
       }
+      // UUID is a logical type stored as 16 raw bytes, so -- like TIMESTAMP 
over LONG above -- convert the
+      // literals to their stored form and reuse the stored-type evaluator.
+      case UUID: {
+        ByteArray[] uuidValues = inPredicate.getUuidValues();
+        Set<ByteArray> matchingValues = new 
ObjectOpenHashSet<>(HashUtil.getMinHashSetSize(uuidValues.length));
+        // NOTE: Add value-by-value to avoid overhead
+        //noinspection ManualArrayToCollectionCopy
+        for (ByteArray value : uuidValues) {
+          //noinspection UseBulkOperation
+          matchingValues.add(value);
+        }
+        return new BytesRawValueBasedInPredicateEvaluator(inPredicate, 
matchingValues);
+      }
       default:
         throw new IllegalStateException("Unsupported data type: " + dataType);
     }
diff --git 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/NotEqualsPredicateEvaluatorFactory.java
 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/NotEqualsPredicateEvaluatorFactory.java
index 412e4ec15e5..902031f845c 100644
--- 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/NotEqualsPredicateEvaluatorFactory.java
+++ 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/NotEqualsPredicateEvaluatorFactory.java
@@ -27,6 +27,7 @@ import org.apache.pinot.spi.data.FieldSpec.DataType;
 import org.apache.pinot.spi.utils.BooleanUtils;
 import org.apache.pinot.spi.utils.BytesUtils;
 import org.apache.pinot.spi.utils.TimestampUtils;
+import org.apache.pinot.spi.utils.UuidUtils;
 
 
 /// Factory for NEQ predicate evaluators.
@@ -72,6 +73,11 @@ public class NotEqualsPredicateEvaluatorFactory {
         return new StringRawValueBasedNeqPredicateEvaluator(notEqPredicate, 
value);
       case BYTES:
         return new BytesRawValueBasedNeqPredicateEvaluator(notEqPredicate, 
BytesUtils.toBytes(value));
+      // UUID is a logical type stored as 16 raw bytes, so -- like TIMESTAMP 
over LONG above -- convert the literal to
+      // its stored form and reuse the stored-type evaluator. getDataType() 
then correctly reports the type applySV
+      // consumes (BYTES), per the PredicateEvaluator contract.
+      case UUID:
+        return new BytesRawValueBasedNeqPredicateEvaluator(notEqPredicate, 
UuidUtils.toBytes(value));
       default:
         throw new IllegalStateException("Unsupported data type: " + dataType);
     }
diff --git 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/NotInPredicateEvaluatorFactory.java
 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/NotInPredicateEvaluatorFactory.java
index f744c3eb917..f210599773b 100644
--- 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/NotInPredicateEvaluatorFactory.java
+++ 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/NotInPredicateEvaluatorFactory.java
@@ -144,6 +144,19 @@ public class NotInPredicateEvaluatorFactory {
         }
         return new BytesRawValueBasedNotInPredicateEvaluator(notInPredicate, 
nonMatchingValues);
       }
+      // UUID is a logical type stored as 16 raw bytes, so -- like TIMESTAMP 
over LONG above -- convert the
+      // literals to their stored form and reuse the stored-type evaluator.
+      case UUID: {
+        ByteArray[] uuidValues = notInPredicate.getUuidValues();
+        Set<ByteArray> nonMatchingValues = new 
ObjectOpenHashSet<>(HashUtil.getMinHashSetSize(uuidValues.length));
+        // NOTE: Add value-by-value to avoid overhead
+        //noinspection ManualArrayToCollectionCopy
+        for (ByteArray value : uuidValues) {
+          //noinspection UseBulkOperation
+          nonMatchingValues.add(value);
+        }
+        return new BytesRawValueBasedNotInPredicateEvaluator(notInPredicate, 
nonMatchingValues);
+      }
       default:
         throw new IllegalStateException("Unsupported data type: " + dataType);
     }
diff --git 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/PredicateUtils.java
 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/PredicateUtils.java
index a9c6baaac1b..5b514dc9c42 100644
--- 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/PredicateUtils.java
+++ 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/PredicateUtils.java
@@ -32,8 +32,10 @@ import org.apache.pinot.segment.spi.index.reader.Dictionary;
 import org.apache.pinot.spi.data.FieldSpec.DataType;
 import org.apache.pinot.spi.utils.BooleanUtils;
 import org.apache.pinot.spi.utils.ByteArray;
+import org.apache.pinot.spi.utils.BytesUtils;
 import 
org.apache.pinot.spi.utils.CommonConstants.Broker.Request.QueryOptionKey;
 import org.apache.pinot.spi.utils.TimestampUtils;
+import org.apache.pinot.spi.utils.UuidUtils;
 
 
 public class PredicateUtils {
@@ -51,6 +53,15 @@ public class PredicateUtils {
         return getStoredBooleanValue(value);
       case TIMESTAMP:
         return getStoredTimestampValue(value);
+      case UUID:
+        // The hex here is a transport encoding for the String-typed lookup 
APIs, NOT the storage format -- a UUID
+        // column is stored as its raw 16 bytes, and the bytes round-trip 
unchanged (encode here, decode in the
+        // dictionary). Range bounds reach the dictionary only through 
Dictionary#insertionIndexOf(String) and
+        // #getDictIdsInRange(String, ...), and the canonical "550e8400-..." 
form cannot be passed through as-is
+        // because the dashes are not valid hex. Equality and IN avoid this 
entirely: they resolve UUIDs
+        // byte-natively via Dictionary#indexOf(ByteArray). Adding a matching 
insertionIndexOf(ByteArray) overload
+        // would let range bounds do the same.
+        return BytesUtils.toHexString(UuidUtils.toBytes(value));
       default:
         return value;
     }
@@ -177,6 +188,15 @@ public class PredicateUtils {
           }
         }
         break;
+      case UUID:
+        ByteArray[] uuidValues = inPredicate.getUuidValues();
+        for (ByteArray value : uuidValues) {
+          int dictId = dictionary.indexOf(value);
+          if (dictId >= 0) {
+            dictIdSet.add(dictId);
+          }
+        }
+        break;
       default:
         throw new IllegalStateException("Unsupported data type: " + dataType);
     }
diff --git 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/RangePredicateEvaluatorFactory.java
 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/RangePredicateEvaluatorFactory.java
index 01ca8d9cc8f..74b594bd441 100644
--- 
a/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/RangePredicateEvaluatorFactory.java
+++ 
b/pinot-core/src/main/java/org/apache/pinot/core/operator/filter/predicate/RangePredicateEvaluatorFactory.java
@@ -34,6 +34,7 @@ import org.apache.pinot.spi.utils.BooleanUtils;
 import org.apache.pinot.spi.utils.ByteArray;
 import org.apache.pinot.spi.utils.BytesUtils;
 import org.apache.pinot.spi.utils.TimestampUtils;
+import org.apache.pinot.spi.utils.UuidUtils;
 
 
 /// Factory for RANGE predicate evaluators.
@@ -107,6 +108,14 @@ public class RangePredicateEvaluatorFactory {
         return new BytesRawValueBasedRangePredicateEvaluator(rangePredicate,
             lowerUnbounded ? null : BytesUtils.toBytes(lowerBound),
             upperUnbounded ? null : BytesUtils.toBytes(upperBound), 
lowerInclusive, upperInclusive);
+      // UUID is stored as 16 raw bytes and its unsigned bytewise ordering is 
exactly UUID ordering, so -- like
+      // TIMESTAMP over LONG above -- convert the bounds to the stored form 
and reuse the BYTES evaluator. UUID
+      // dictionaries also report getValueType() == BYTES, so the unsorted 
dictionary-based evaluator dispatches on
+      // BYTES and feeds this the raw 16-byte values, directly comparable to 
the bounds.
+      case UUID:
+        return new BytesRawValueBasedRangePredicateEvaluator(rangePredicate,
+            lowerUnbounded ? null : UuidUtils.toBytes(lowerBound),
+            upperUnbounded ? null : UuidUtils.toBytes(upperBound), 
lowerInclusive, upperInclusive);
       default:
         throw new IllegalStateException("Unsupported data type: " + dataType);
     }
diff --git 
a/pinot-core/src/main/java/org/apache/pinot/core/query/reduce/filter/PredicateRowMatcher.java
 
b/pinot-core/src/main/java/org/apache/pinot/core/query/reduce/filter/PredicateRowMatcher.java
index 65845734a7b..e22e86af6fd 100644
--- 
a/pinot-core/src/main/java/org/apache/pinot/core/query/reduce/filter/PredicateRowMatcher.java
+++ 
b/pinot-core/src/main/java/org/apache/pinot/core/query/reduce/filter/PredicateRowMatcher.java
@@ -20,11 +20,13 @@ package org.apache.pinot.core.query.reduce.filter;
 
 import java.math.BigDecimal;
 import java.sql.Timestamp;
+import java.util.UUID;
 import javax.annotation.Nullable;
 import org.apache.pinot.common.request.context.predicate.Predicate;
 import org.apache.pinot.core.operator.filter.predicate.PredicateEvaluator;
 import 
org.apache.pinot.core.operator.filter.predicate.PredicateEvaluatorProvider;
 import org.apache.pinot.spi.data.FieldSpec.DataType;
+import org.apache.pinot.spi.utils.UuidUtils;
 
 
 /// Predicate matcher.
@@ -78,6 +80,8 @@ public class PredicateRowMatcher implements RowMatcher {
         return _predicateEvaluator.applySV((String) value);
       case BYTES:
         return _predicateEvaluator.applySV((byte[]) value);
+      case UUID:
+        return _predicateEvaluator.applySV(UuidUtils.toBytes((UUID) value));
       default:
         throw new IllegalStateException();
     }
diff --git 
a/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryEqualsPredicateEvaluatorsTest.java
 
b/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryEqualsPredicateEvaluatorsTest.java
index 24b8b305200..3eecc5afc26 100644
--- 
a/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryEqualsPredicateEvaluatorsTest.java
+++ 
b/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryEqualsPredicateEvaluatorsTest.java
@@ -19,7 +19,10 @@
 package org.apache.pinot.core.operator.filter.predicate;
 
 import java.math.BigDecimal;
+import java.util.Arrays;
+import java.util.Locale;
 import java.util.Random;
+import java.util.UUID;
 import org.apache.commons.lang3.ArrayUtils;
 import org.apache.commons.lang3.RandomStringUtils;
 import org.apache.pinot.common.request.context.ExpressionContext;
@@ -27,6 +30,7 @@ import 
org.apache.pinot.common.request.context.predicate.EqPredicate;
 import org.apache.pinot.common.request.context.predicate.NotEqPredicate;
 import org.apache.pinot.spi.data.FieldSpec;
 import org.apache.pinot.spi.utils.BytesUtils;
+import org.apache.pinot.spi.utils.UuidUtils;
 import org.testng.Assert;
 import org.testng.annotations.BeforeClass;
 import org.testng.annotations.Test;
@@ -322,4 +326,43 @@ public class NoDictionaryEqualsPredicateEvaluatorsTest {
           !ArrayUtils.contains(randomBytesArray, stringValue));
     }
   }
+
+  @Test
+  public void testUuidPredicateEvaluators() {
+    UUID uuidValue = UUID.fromString("550e8400-e29b-41d4-a716-446655440000");
+    byte[] uuidBytes = UuidUtils.toBytes(uuidValue);
+    // Predicate literals reach the evaluator as UUID strings. Use an 
upper-cased one to pin down that the hex digits
+    // are matched case-insensitively rather than compared as raw strings.
+    String stringValue = uuidValue.toString().toUpperCase(Locale.ROOT);
+
+    EqPredicate eqPredicate = new EqPredicate(COLUMN_EXPRESSION, stringValue);
+    PredicateEvaluator eqPredicateEvaluator =
+        EqualsPredicateEvaluatorFactory.newRawValueBasedEvaluator(eqPredicate, 
FieldSpec.DataType.UUID);
+
+    NotEqPredicate notEqPredicate = new NotEqPredicate(COLUMN_EXPRESSION, 
stringValue);
+    PredicateEvaluator neqPredicateEvaluator =
+        
NotEqualsPredicateEvaluatorFactory.newRawValueBasedEvaluator(notEqPredicate, 
FieldSpec.DataType.UUID);
+
+    // getDataType() reports the type applySV consumes, not the column's 
logical type -- exactly as a TIMESTAMP
+    // column's evaluator reports LONG. UUID literals are converted to their 
16-byte stored form up front, so the
+    // BYTES raw evaluator is reused as-is and reports BYTES.
+    Assert.assertEquals(eqPredicateEvaluator.getDataType(), 
FieldSpec.DataType.BYTES);
+    Assert.assertEquals(neqPredicateEvaluator.getDataType(), 
FieldSpec.DataType.BYTES);
+
+    Assert.assertTrue(eqPredicateEvaluator.applySV(uuidBytes));
+    Assert.assertFalse(neqPredicateEvaluator.applySV(uuidBytes));
+
+    // A UUID differing only in the last byte must not match, guarding against 
a truncated comparison.
+    byte[] nearMissBytes = Arrays.copyOf(uuidBytes, uuidBytes.length);
+    nearMissBytes[nearMissBytes.length - 1] ^= 0x01;
+    Assert.assertFalse(eqPredicateEvaluator.applySV(nearMissBytes));
+    Assert.assertTrue(neqPredicateEvaluator.applySV(nearMissBytes));
+
+    for (int i = 0; i < 100; i++) {
+      byte[] randomUuidBytes = UuidUtils.toBytes(UUID.randomUUID());
+      boolean matches = Arrays.equals(randomUuidBytes, uuidBytes);
+      Assert.assertEquals(eqPredicateEvaluator.applySV(randomUuidBytes), 
matches);
+      Assert.assertEquals(neqPredicateEvaluator.applySV(randomUuidBytes), 
!matches);
+    }
+  }
 }
diff --git 
a/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryInPredicateEvaluatorTest.java
 
b/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryInPredicateEvaluatorTest.java
index affa76f49e8..feadb532c89 100644
--- 
a/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryInPredicateEvaluatorTest.java
+++ 
b/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryInPredicateEvaluatorTest.java
@@ -39,6 +39,7 @@ import 
org.apache.pinot.common.request.context.predicate.InPredicate;
 import org.apache.pinot.common.request.context.predicate.NotInPredicate;
 import org.apache.pinot.spi.data.FieldSpec;
 import org.apache.pinot.spi.utils.BytesUtils;
+import org.apache.pinot.spi.utils.UuidUtils;
 import org.testng.Assert;
 import org.testng.annotations.BeforeClass;
 import org.testng.annotations.Test;
@@ -363,4 +364,76 @@ public class NoDictionaryInPredicateEvaluatorTest {
     Assert.assertTrue(inPredicateEvaluator.applyMV(multiValues, 
NUM_MULTI_VALUES));
     Assert.assertFalse(notInPredicateEvaluator.applyMV(multiValues, 
NUM_MULTI_VALUES));
   }
+
+  @Test
+  public void testUuidPredicateEvaluators() {
+    List<String> uuidStrings = new ArrayList<>(NUM_PREDICATE_VALUES);
+    Set<String> uuidStringSet = new HashSet<>();
+
+    for (int i = 0; i < NUM_PREDICATE_VALUES; i++) {
+      String uuidString = java.util.UUID.randomUUID().toString();
+      uuidStrings.add(uuidString);
+      uuidStringSet.add(uuidString);
+    }
+
+    InPredicate inPredicate = new InPredicate(COLUMN_EXPRESSION, uuidStrings);
+    PredicateEvaluator inPredicateEvaluator =
+        InPredicateEvaluatorFactory.newRawValueBasedEvaluator(inPredicate, 
FieldSpec.DataType.UUID);
+
+    NotInPredicate notInPredicate = new NotInPredicate(COLUMN_EXPRESSION, 
uuidStrings);
+    PredicateEvaluator notInPredicateEvaluator =
+        
NotInPredicateEvaluatorFactory.newRawValueBasedEvaluator(notInPredicate, 
FieldSpec.DataType.UUID);
+
+    // getDataType() reports the type applySV consumes, not the column's 
logical type -- exactly as a TIMESTAMP
+    // column's evaluator reports LONG. UUID literals are converted to their 
16-byte stored form up front, so the
+    // BYTES raw evaluator is reused as-is and reports BYTES.
+    Assert.assertEquals(inPredicateEvaluator.getDataType(), 
FieldSpec.DataType.BYTES);
+    Assert.assertEquals(notInPredicateEvaluator.getDataType(), 
FieldSpec.DataType.BYTES);
+
+    for (String uuidString : uuidStringSet) {
+      byte[] uuidBytes = UuidUtils.toBytes(uuidString);
+      Assert.assertTrue(inPredicateEvaluator.applySV(uuidBytes));
+      Assert.assertFalse(notInPredicateEvaluator.applySV(uuidBytes));
+    }
+
+    for (int i = 0; i < NUM_PREDICATE_VALUES; i++) {
+      byte[] value = UuidUtils.toBytes(java.util.UUID.randomUUID());
+      boolean expected = uuidStringSet.contains(UuidUtils.toString(value));
+      Assert.assertEquals(inPredicateEvaluator.applySV(value), expected);
+      Assert.assertEquals(notInPredicateEvaluator.applySV(value), !expected);
+    }
+  }
+
+  /// The BYTES/UUID raw evaluators key their matching set on the raw `byte[]` 
so that `applySV` does not
+  /// wrap every scanned value. That only works if the set compares by 
*content*: with identity semantics a
+  /// scanned array would never match a predicate array, and IN would silently 
return nothing while NOT IN returned
+  /// everything. Probe with arrays that are equal but deliberately not the 
same instance.
+  @Test
+  public void testBytesAndUuidPredicatesMatchByValueNotIdentity() {
+    String uuidString = "550e8400-e29b-41d4-a716-446655440000";
+    String bytesHex = "0a1b2c3d";
+
+    for (Object[] testCase : new Object[][]{
+        {FieldSpec.DataType.UUID, uuidString, UuidUtils.toBytes(uuidString)},
+        {FieldSpec.DataType.BYTES, bytesHex, BytesUtils.toBytes(bytesHex)}
+    }) {
+      FieldSpec.DataType dataType = (FieldSpec.DataType) testCase[0];
+      List<String> values = List.of((String) testCase[1]);
+      byte[] probe = ((byte[]) testCase[2]).clone();
+
+      PredicateEvaluator inEvaluator = 
InPredicateEvaluatorFactory.newRawValueBasedEvaluator(
+          new InPredicate(COLUMN_EXPRESSION, values), dataType);
+      PredicateEvaluator notInEvaluator = 
NotInPredicateEvaluatorFactory.newRawValueBasedEvaluator(
+          new NotInPredicate(COLUMN_EXPRESSION, values), dataType);
+
+      Assert.assertTrue(inEvaluator.applySV(probe), dataType + " IN must match 
an equal-but-distinct array");
+      Assert.assertFalse(notInEvaluator.applySV(probe),
+          dataType + " NOT IN must not match an equal-but-distinct array");
+
+      byte[] different = probe.clone();
+      different[0] ^= 0xff;
+      Assert.assertFalse(inEvaluator.applySV(different), dataType + " IN must 
not match a different value");
+      Assert.assertTrue(notInEvaluator.applySV(different), dataType + " NOT IN 
must match a different value");
+    }
+  }
 }
diff --git 
a/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryRangePredicateEvaluatorTest.java
 
b/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryRangePredicateEvaluatorTest.java
index 1804eb76c8c..a88a76d2837 100644
--- 
a/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryRangePredicateEvaluatorTest.java
+++ 
b/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/NoDictionaryRangePredicateEvaluatorTest.java
@@ -19,10 +19,12 @@
 package org.apache.pinot.core.operator.filter.predicate;
 
 import java.math.BigDecimal;
+import java.util.UUID;
 import org.apache.pinot.common.request.context.ExpressionContext;
 import org.apache.pinot.common.request.context.predicate.RangePredicate;
 import org.apache.pinot.spi.data.FieldSpec;
 import org.apache.pinot.spi.utils.ByteArray;
+import org.apache.pinot.spi.utils.UuidUtils;
 import org.testng.Assert;
 import org.testng.annotations.Test;
 
@@ -390,6 +392,73 @@ public class NoDictionaryRangePredicateEvaluatorTest {
     }
   }
 
+  @Test
+  public void testUuidPredicateEvaluator() {
+    // Spread the probes across the full unsigned byte range of the first 
octet. 0x80..0xff are negative as signed
+    // bytes, so a signed comparison would order them below the bounds and 
this test would fail.
+    String[] uuidStrings = new String[]{
+        "00000000-0000-4000-8000-000000000000", 
"20000000-0000-4000-8000-000000000000",
+        "40000000-0000-4000-8000-000000000000", 
"60000000-0000-4000-8000-000000000000",
+        "80000000-0000-4000-8000-000000000000", 
"a0000000-0000-4000-8000-000000000000",
+        "c0000000-0000-4000-8000-000000000000", 
"e0000000-0000-4000-8000-000000000000",
+        "ffffffff-ffff-4fff-bfff-ffffffffffff"
+    };
+    String lower = "40000000-0000-4000-8000-000000000000";
+    String upper = "c0000000-0000-4000-8000-000000000000";
+    byte[] lowerBytes = UuidUtils.toBytes(lower);
+    byte[] upperBytes = UuidUtils.toBytes(upper);
+
+    PredicateEvaluator predicateEvaluator = buildRangePredicate("[" + lower + 
"\000" + upper + "]",
+        FieldSpec.DataType.UUID);
+    for (String uuidString : uuidStrings) {
+      byte[] value = UuidUtils.toBytes(uuidString);
+      Assert.assertEquals(predicateEvaluator.applySV(value),
+          ByteArray.compare(value, lowerBytes) >= 0 && 
ByteArray.compare(value, upperBytes) <= 0, uuidString);
+    }
+
+    predicateEvaluator = buildRangePredicate("(" + lower + "\000" + upper + 
"]", FieldSpec.DataType.UUID);
+    for (String uuidString : uuidStrings) {
+      byte[] value = UuidUtils.toBytes(uuidString);
+      Assert.assertEquals(predicateEvaluator.applySV(value),
+          ByteArray.compare(value, lowerBytes) > 0 && ByteArray.compare(value, 
upperBytes) <= 0, uuidString);
+    }
+
+    predicateEvaluator = buildRangePredicate("(" + lower + "\000" + upper + 
")", FieldSpec.DataType.UUID);
+    for (String uuidString : uuidStrings) {
+      byte[] value = UuidUtils.toBytes(uuidString);
+      Assert.assertEquals(predicateEvaluator.applySV(value),
+          ByteArray.compare(value, lowerBytes) > 0 && ByteArray.compare(value, 
upperBytes) < 0, uuidString);
+    }
+
+    predicateEvaluator = buildRangePredicate("(*\000" + upper + "]", 
FieldSpec.DataType.UUID);
+    for (String uuidString : uuidStrings) {
+      byte[] value = UuidUtils.toBytes(uuidString);
+      Assert.assertEquals(predicateEvaluator.applySV(value), 
ByteArray.compare(value, upperBytes) <= 0, uuidString);
+    }
+
+    predicateEvaluator = buildRangePredicate("[" + lower + "\000*)", 
FieldSpec.DataType.UUID);
+    for (String uuidString : uuidStrings) {
+      byte[] value = UuidUtils.toBytes(uuidString);
+      Assert.assertEquals(predicateEvaluator.applySV(value), 
ByteArray.compare(value, lowerBytes) >= 0, uuidString);
+    }
+
+    predicateEvaluator = buildRangePredicate("(*\000*)", 
FieldSpec.DataType.UUID);
+    for (String uuidString : uuidStrings) {
+      
Assert.assertTrue(predicateEvaluator.applySV(UuidUtils.toBytes(uuidString)), 
uuidString);
+    }
+
+    // Range bounds are canonical UUID strings; the ordering must match 
java.util.UUID's unsigned-word comparison.
+    UUID lowerUuid = UUID.fromString(lower);
+    UUID upperUuid = UUID.fromString(upper);
+    predicateEvaluator = buildRangePredicate("[" + lower + "\000" + upper + 
"]", FieldSpec.DataType.UUID);
+    for (int i = 0; i < 100; i++) {
+      UUID randomUuid = UUID.randomUUID();
+      boolean expected = UuidUtils.compare(UuidUtils.toBytes(randomUuid), 
UuidUtils.toBytes(lowerUuid)) >= 0
+          && UuidUtils.compare(UuidUtils.toBytes(randomUuid), 
UuidUtils.toBytes(upperUuid)) <= 0;
+      
Assert.assertEquals(predicateEvaluator.applySV(UuidUtils.toBytes(randomUuid)), 
expected, randomUuid.toString());
+    }
+  }
+
   private PredicateEvaluator buildRangePredicate(String rangeString, 
FieldSpec.DataType dataType) {
     RangePredicate predicate = new RangePredicate(COLUMN_EXPRESSION, 
rangeString);
     return RangePredicateEvaluatorFactory.newRawValueBasedEvaluator(predicate, 
dataType);
diff --git 
a/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/UuidDictionaryPredicateEvaluatorTest.java
 
b/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/UuidDictionaryPredicateEvaluatorTest.java
new file mode 100644
index 00000000000..62520d07e97
--- /dev/null
+++ 
b/pinot-core/src/test/java/org/apache/pinot/core/operator/filter/predicate/UuidDictionaryPredicateEvaluatorTest.java
@@ -0,0 +1,173 @@
+/**
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements.  See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership.  The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License.  You may obtain a copy of the License at
+ *
+ *   http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied.  See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+package org.apache.pinot.core.operator.filter.predicate;
+
+import it.unimi.dsi.fastutil.ints.IntSet;
+import java.io.IOException;
+import java.util.ArrayList;
+import java.util.Arrays;
+import java.util.List;
+import java.util.Locale;
+import java.util.UUID;
+import org.apache.pinot.common.request.context.ExpressionContext;
+import org.apache.pinot.common.request.context.predicate.EqPredicate;
+import org.apache.pinot.common.request.context.predicate.InPredicate;
+import org.apache.pinot.common.request.context.predicate.NotEqPredicate;
+import org.apache.pinot.common.request.context.predicate.NotInPredicate;
+import org.apache.pinot.segment.local.io.writer.impl.DirectMemoryManager;
+import 
org.apache.pinot.segment.local.realtime.impl.dictionary.BytesOffHeapMutableDictionary;
+import org.apache.pinot.segment.spi.memory.PinotDataBufferMemoryManager;
+import org.apache.pinot.spi.data.FieldSpec.DataType;
+import org.apache.pinot.spi.utils.ByteArray;
+import org.apache.pinot.spi.utils.BytesUtils;
+import org.apache.pinot.spi.utils.UuidUtils;
+import org.testng.annotations.AfterClass;
+import org.testng.annotations.BeforeClass;
+import org.testng.annotations.Test;
+
+import static org.testng.Assert.assertEquals;
+import static org.testng.Assert.assertTrue;
+
+
+/// Unit test for dictionary-based predicate evaluators over a logical `UUID` 
column.
+///
+/// UUID has stored type BYTES, so it is backed by a plain bytes dictionary 
whose `getValueType()` reports BYTES. The
+/// predicate value, however, arrives as a canonical UUID string rather than a 
hex string, so the UUID branches in the
+/// evaluator factories look the value up as raw 16 bytes. These tests run 
against a real
+/// [BytesOffHeapMutableDictionary] rather than a mock so that the lookup 
contract is exercised end to end.
+public class UuidDictionaryPredicateEvaluatorTest {
+  private static final ExpressionContext COLUMN_EXPRESSION = 
ExpressionContext.forIdentifier("column");
+  private static final int NUM_VALUES = 32;
+
+  private PinotDataBufferMemoryManager _memoryManager;
+  private BytesOffHeapMutableDictionary _dictionary;
+  private final List<String> _uuidStrings = new ArrayList<>(NUM_VALUES);
+
+  @BeforeClass
+  public void setUp() {
+    _memoryManager = new 
DirectMemoryManager(UuidDictionaryPredicateEvaluatorTest.class.getSimpleName());
+    _dictionary = new BytesOffHeapMutableDictionary(NUM_VALUES, 0, 
_memoryManager, "uuidDictionary",
+        UuidUtils.UUID_NUM_BYTES);
+    for (int i = 0; i < NUM_VALUES; i++) {
+      // Deterministic UUIDs so a failure is reproducible.
+      UUID uuid = new UUID(0x0123456789abcdefL, i);
+      _uuidStrings.add(uuid.toString());
+      _dictionary.index(UuidUtils.toBytes(uuid));
+    }
+  }
+
+  @AfterClass
+  public void tearDown()
+      throws IOException {
+    _dictionary.close();
+    _memoryManager.close();
+  }
+
+  /// The UUID fast path skips the hex round-trip that 
[PredicateUtils#getStoredValue] performs and looks the raw
+  /// bytes up directly. Both must resolve to the same dictionary id, 
otherwise the fast path would silently change
+  /// which rows match.
+  @Test
+  public void testStoredValueMatchesRawByteLookup() {
+    for (String uuidString : _uuidStrings) {
+      byte[] uuidBytes = UuidUtils.toBytes(uuidString);
+      String storedValue = PredicateUtils.getStoredValue(uuidString, 
DataType.UUID);
+      assertEquals(storedValue, BytesUtils.toHexString(uuidBytes));
+      assertEquals(_dictionary.indexOf(storedValue), _dictionary.indexOf(new 
ByteArray(uuidBytes)));
+    }
+  }
+
+  @Test
+  public void testEqAndNeqEvaluators() {
+    for (int i = 0; i < NUM_VALUES; i++) {
+      String uuidString = _uuidStrings.get(i);
+      int expectedDictId = _dictionary.indexOf(new 
ByteArray(UuidUtils.toBytes(uuidString)));
+      assertTrue(expectedDictId >= 0);
+
+      BaseDictionaryBasedPredicateEvaluator eqEvaluator = 
EqualsPredicateEvaluatorFactory.newDictionaryBasedEvaluator(
+          new EqPredicate(COLUMN_EXPRESSION, uuidString), _dictionary, 
DataType.UUID);
+      assertEquals(eqEvaluator.getMatchingDictIds(), new 
int[]{expectedDictId});
+
+      BaseDictionaryBasedPredicateEvaluator neqEvaluator =
+          NotEqualsPredicateEvaluatorFactory.newDictionaryBasedEvaluator(
+              new NotEqPredicate(COLUMN_EXPRESSION, uuidString), _dictionary, 
DataType.UUID);
+      assertEquals(neqEvaluator.getNonMatchingDictIds(), new 
int[]{expectedDictId});
+    }
+  }
+
+  @Test
+  public void testEqEvaluatorIsCaseInsensitive() {
+    String uuidString = _uuidStrings.get(0);
+    int expectedDictId = _dictionary.indexOf(new 
ByteArray(UuidUtils.toBytes(uuidString)));
+
+    BaseDictionaryBasedPredicateEvaluator eqEvaluator = 
EqualsPredicateEvaluatorFactory.newDictionaryBasedEvaluator(
+        new EqPredicate(COLUMN_EXPRESSION, 
uuidString.toUpperCase(Locale.ROOT)), _dictionary, DataType.UUID);
+    assertEquals(eqEvaluator.getMatchingDictIds(), new int[]{expectedDictId});
+  }
+
+  @Test
+  public void testEqAndNeqEvaluatorsOnAbsentValue() {
+    // Same high bits as the indexed values but a low half outside the indexed 
range.
+    String absentUuid = new UUID(0x0123456789abcdefL, NUM_VALUES).toString();
+    assertTrue(_dictionary.indexOf(new 
ByteArray(UuidUtils.toBytes(absentUuid))) < 0);
+
+    BaseDictionaryBasedPredicateEvaluator eqEvaluator = 
EqualsPredicateEvaluatorFactory.newDictionaryBasedEvaluator(
+        new EqPredicate(COLUMN_EXPRESSION, absentUuid), _dictionary, 
DataType.UUID);
+    assertTrue(eqEvaluator.isAlwaysFalse());
+
+    BaseDictionaryBasedPredicateEvaluator neqEvaluator = 
NotEqualsPredicateEvaluatorFactory.newDictionaryBasedEvaluator(
+        new NotEqPredicate(COLUMN_EXPRESSION, absentUuid), _dictionary, 
DataType.UUID);
+    assertTrue(neqEvaluator.isAlwaysTrue());
+  }
+
+  @Test
+  public void testGetDictIdSet() {
+    List<String> values = List.of(_uuidStrings.get(1), _uuidStrings.get(5), 
_uuidStrings.get(9),
+        // An absent UUID must simply not contribute a dict id rather than 
fail the lookup.
+        new UUID(0x0123456789abcdefL, NUM_VALUES + 1).toString());
+    InPredicate inPredicate = new InPredicate(COLUMN_EXPRESSION, values);
+
+    IntSet dictIdSet = PredicateUtils.getDictIdSet(inPredicate, _dictionary, 
DataType.UUID, null);
+    assertEquals(dictIdSet.size(), 3);
+    for (int i = 0; i < 3; i++) {
+      assertTrue(dictIdSet.contains(_dictionary.indexOf(new 
ByteArray(UuidUtils.toBytes(values.get(i))))));
+    }
+  }
+
+  @Test
+  public void testInAndNotInEvaluators() {
+    List<String> values = List.of(_uuidStrings.get(2), _uuidStrings.get(7));
+    int[] expectedDictIds = new int[values.size()];
+    for (int i = 0; i < values.size(); i++) {
+      expectedDictIds[i] = _dictionary.indexOf(new 
ByteArray(UuidUtils.toBytes(values.get(i))));
+    }
+
+    BaseDictionaryBasedPredicateEvaluator inEvaluator = 
InPredicateEvaluatorFactory.newDictionaryBasedEvaluator(
+        new InPredicate(COLUMN_EXPRESSION, values), _dictionary, 
DataType.UUID, null);
+    int[] matchingDictIds = inEvaluator.getMatchingDictIds();
+    Arrays.sort(matchingDictIds);
+    Arrays.sort(expectedDictIds);
+    assertEquals(matchingDictIds, expectedDictIds);
+
+    BaseDictionaryBasedPredicateEvaluator notInEvaluator = 
NotInPredicateEvaluatorFactory.newDictionaryBasedEvaluator(
+        new NotInPredicate(COLUMN_EXPRESSION, values), _dictionary, 
DataType.UUID, null);
+    int[] nonMatchingDictIds = notInEvaluator.getNonMatchingDictIds();
+    Arrays.sort(nonMatchingDictIds);
+    assertEquals(nonMatchingDictIds, expectedDictIds);
+  }
+}
diff --git 
a/pinot-core/src/test/java/org/apache/pinot/core/query/reduce/HavingFilterHandlerTest.java
 
b/pinot-core/src/test/java/org/apache/pinot/core/query/reduce/HavingFilterHandlerTest.java
index 174b54072ee..f109d457198 100644
--- 
a/pinot-core/src/test/java/org/apache/pinot/core/query/reduce/HavingFilterHandlerTest.java
+++ 
b/pinot-core/src/test/java/org/apache/pinot/core/query/reduce/HavingFilterHandlerTest.java
@@ -18,6 +18,7 @@
  */
 package org.apache.pinot.core.query.reduce;
 
+import java.util.UUID;
 import org.apache.pinot.common.utils.DataSchema;
 import org.apache.pinot.common.utils.DataSchema.ColumnDataType;
 import org.apache.pinot.core.query.request.context.QueryContext;
@@ -98,6 +99,26 @@ public class HavingFilterHandlerTest {
     }
   }
 
+  /// A UUID column reaches [PredicateRowMatcher] as a `java.util.UUID`, not 
as the internal `ByteArray`:
+  /// `GroupByDataTableReducer` runs every column through 
`ColumnDataType#convert` immediately before calling
+  /// `isMatch`, and that returns `UuidUtils.toUUID(...)` for UUID. This pins 
that contract, since the matcher
+  /// casts directly rather than accepting several input forms.
+  @Test
+  public void testHavingFilterOnUuidColumn() {
+    QueryContext queryContext = QueryContextConverterUtils.getQueryContext(
+        "SELECT COUNT(*) FROM testTable GROUP BY d1 HAVING d1 = 
'550e8400-e29b-41d4-a716-446655440000'");
+    DataSchema dataSchema = new DataSchema(new String[]{"d1", "count(*)"},
+        new ColumnDataType[]{ColumnDataType.UUID, ColumnDataType.LONG});
+    PostAggregationHandler postAggregationHandler = new 
PostAggregationHandler(queryContext, dataSchema);
+    HavingFilterHandler havingFilterHandler =
+        new HavingFilterHandler(queryContext.getHavingFilter(), 
postAggregationHandler, false);
+
+    UUID matching = UUID.fromString("550e8400-e29b-41d4-a716-446655440000");
+    UUID other = UUID.fromString("550e8400-e29b-41d4-a716-446655440001");
+    assertTrue(havingFilterHandler.isMatch(new Object[]{matching, 5L}));
+    assertFalse(havingFilterHandler.isMatch(new Object[]{other, 5L}));
+  }
+
   @Test
   public void testIsNullWhenNullHandlingEnabled() {
     QueryContext queryContext = QueryContextConverterUtils.getQueryContext(


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to