Manya0407 commented on code in PR #6746:
URL: https://github.com/apache/hive/pull/6746#discussion_r4122161185
##########
ql/src/java/org/apache/hadoop/hive/ql/optimizer/calcite/stats/FilterSelectivityEstimator.java:
##########
@@ -484,19 +484,164 @@ private Double
computeRangePredicateSelectivity(Supplier<Double> defaultSelectiv
}
final List<ColStatistics> colStats =
scan.getColStat(Collections.singletonList(inputRefIndex));
- if (colStats.isEmpty() || !isHistogramAvailable(colStats.get(0))) {
+ if (colStats.isEmpty()) {
return defaultSelectivity.get();
}
- final KllFloatsSketch kll =
KllFloatsSketch.heapify(Memory.wrap(colStats.get(0).getHistogram()));
- double rawSelectivity = rangedSelectivity(kll, boundaries);
+ final ColStatistics cs = colStats.get(0);
+ if (isHistogramAvailable(cs)) {
+ final KllFloatsSketch kll =
KllFloatsSketch.heapify(Memory.wrap(cs.getHistogram()));
+ double rawSelectivity = rangedSelectivity(kll, boundaries);
+ if (inverseBool) {
+ // when inverseBool == true, this is a NOT_BETWEEN and selectivity
must be inverted
+ // if there's a cast, the inversion is with respect to its codomain
(range of the values of the cast)
+ double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
+ rawSelectivity = typeRangeSelectivity - rawSelectivity;
+ }
+ return scaleSelectivityToNullableValues(kll, rawSelectivity, scan);
+ }
+
+ if (isUniformWithinRangeEnabled() && hasUsableMinMax(cs)) {
+ RelDataType columnType =
scan.getRowType().getFieldList().get(inputRefIndex).getType();
+ Double uniformSelectivity = computeUniformRangeSelectivity(cs,
boundaries, scan, inverseBool, typeRange,
+ columnType);
+ if (uniformSelectivity != null) {
+ return uniformSelectivity;
+ }
+ }
+
+ return defaultSelectivity.get();
+ }
+
+ private boolean isUniformWithinRangeEnabled() {
+ HiveConfPlannerContext ctx =
+
childRel.getCluster().getPlanner().getContext().unwrap(HiveConfPlannerContext.class);
+ return ctx == null || ctx.isUniformWithinRange();
+ }
+
+ private static boolean hasUsableMinMax(ColStatistics cs) {
+ ColStatistics.Range range = cs.getRange();
+ return range != null && range.minValue != null && range.maxValue != null;
+ }
+
+ /**
+ * Converts column MIN/MAX statistics into the same numeric space used by
{@link #extractLiteral}.
+ * DATE column stats from HMS are stored as days since epoch; literals use
epoch seconds.
+ */
+ private static Optional<Range<Float>>
convertColRangeToFloatRange(ColStatistics cs, RelDataType columnType) {
+ ColStatistics.Range range = cs.getRange();
+ if (range == null || range.minValue == null || range.maxValue == null) {
+ return Optional.empty();
+ }
+ final Number minValue = range.minValue;
+ final Number maxValue = range.maxValue;
+ switch (columnType.getSqlTypeName()) {
+ case DATE:
+ return Optional.of(Range.closed((float) (minValue.longValue() * 86400L),
+ (float) (maxValue.longValue() * 86400L)));
+ case TINYINT:
+ case SMALLINT:
+ case INTEGER:
+ case BIGINT:
+ case FLOAT:
+ case DOUBLE:
+ case DECIMAL:
+ case TIMESTAMP:
+ return Optional.of(Range.closed(minValue.floatValue(),
maxValue.floatValue()));
Review Comment:
Acknowledged — CBO still maps min/max through floatValue() for overlap;
annotate can differ slightly. Added min/max uniform tests for f_bigint,
f_double, and f_decimal10s3 at mid-range (~0.5) so we don’t get wildly wrong
selectivity; full parity with StatsRulesProcFactory can be a follow-up if we
want.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]