thomasrebele commented on code in PR #6746:
URL: https://github.com/apache/hive/pull/6746#discussion_r3903234102


##########
ql/src/test/org/apache/hadoop/hive/ql/optimizer/calcite/stats/TestFilterSelectivityEstimator.java:
##########
@@ -1188,6 +1189,13 @@ private static RexLiteral literalTimestamp(String 
timestamp) {
         REX_BUILDER.getTypeFactory().createSqlType(SqlTypeName.TIMESTAMP));
   }
 
+  private static RexLiteral literalDate(String date) {
+    GregorianCalendar calendar =
+        
GregorianCalendar.from(LocalDate.parse(date).atStartOfDay(ZoneOffset.UTC));
+    return (RexLiteral) REX_BUILDER.makeLiteral(calendar,
+        REX_BUILDER.getTypeFactory().createSqlType(SqlTypeName.DATE), true);
+  }
+

Review Comment:
   I think we could use org.apache.calcite.rex.RexBuilder#makeDateLiteral here.



##########
ql/src/java/org/apache/hadoop/hive/ql/optimizer/calcite/stats/FilterSelectivityEstimator.java:
##########
@@ -484,19 +484,261 @@ private Double 
computeRangePredicateSelectivity(Supplier<Double> defaultSelectiv
     }
 
     final List<ColStatistics> colStats = 
scan.getColStat(Collections.singletonList(inputRefIndex));
-    if (colStats.isEmpty() || !isHistogramAvailable(colStats.get(0))) {
+    if (colStats.isEmpty()) {
       return defaultSelectivity.get();
     }
 
-    final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(colStats.get(0).getHistogram()));
-    double rawSelectivity = rangedSelectivity(kll, boundaries);
+    final ColStatistics cs = colStats.get(0);
+    if (isHistogramAvailable(cs)) {
+      final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(cs.getHistogram()));
+      double rawSelectivity = rangedSelectivity(kll, boundaries);
+      if (inverseBool) {
+        // when inverseBool == true, this is a NOT_BETWEEN and selectivity 
must be inverted
+        // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
+        double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
+        rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      }
+      return scaleSelectivityToNullableValues(kll, rawSelectivity, scan);
+    }
+
+    if (isUniformWithinRangeEnabled() && hasUsableMinMax(cs)) {
+      RelDataType columnType = 
scan.getRowType().getFieldList().get(inputRefIndex).getType();
+      Double uniformSelectivity = computeUniformRangeSelectivity(cs, 
boundaries, scan, inverseBool, typeRange,
+          columnType);
+      if (uniformSelectivity != null) {
+        return uniformSelectivity;
+      }
+    }
+
+    return defaultSelectivity.get();
+  }
+
+  private boolean isUniformWithinRangeEnabled() {
+    HiveConfPlannerContext ctx =
+        
childRel.getCluster().getPlanner().getContext().unwrap(HiveConfPlannerContext.class);
+    return ctx == null || ctx.isUniformWithinRange();
+  }
+
+  private static boolean hasUsableMinMax(ColStatistics cs) {
+    ColStatistics.Range range = cs.getRange();
+    return range != null && range.minValue != null && range.maxValue != null;
+  }
+
+  /**
+   * Converts column MIN/MAX statistics into the same numeric space used by 
{@link #extractLiteral}.
+   * DATE column stats from HMS are stored as days since epoch; literals use 
epoch seconds.
+   */
+  private static Optional<float[]> convertColRangeToFloatBounds(ColStatistics 
cs, RelDataType columnType) {
+    ColStatistics.Range range = cs.getRange();
+    if (range == null || range.minValue == null || range.maxValue == null) {
+      return Optional.empty();
+    }
+    final float min;
+    final float max;
+    switch (columnType.getSqlTypeName()) {
+    case DATE:
+      min = range.minValue.longValue() * 86400L;
+      max = range.maxValue.longValue() * 86400L;
+      break;
+    case TIMESTAMP:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case TINYINT:
+      min = range.minValue.byteValue();
+      max = range.maxValue.byteValue();
+      break;
+    case SMALLINT:
+      min = range.minValue.shortValue();
+      max = range.maxValue.shortValue();
+      break;
+    case INTEGER:
+      min = range.minValue.intValue();
+      max = range.maxValue.intValue();
+      break;
+    case BIGINT:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case FLOAT:
+      min = range.minValue.floatValue();
+      max = range.maxValue.floatValue();
+      break;
+    case DOUBLE:
+      min = (float) range.minValue.doubleValue();
+      max = (float) range.maxValue.doubleValue();
+      break;

Review Comment:
   Couldn't we combine these cases by using java.lang.Number#floatValue?
   
   We could even change it to java.lang.Number#doubleValue. The method 
FilterSelectivityEstimator#extractLiteral(org.apache.calcite.rex.RexNode) 
returns float because the histogram stores float values. The range and boundary 
type could be changed to Double as well. Maybe this is out-of-scope for 
HIVE-29652, as it would require a bit of refactoring to not lose information.



##########
ql/src/java/org/apache/hadoop/hive/ql/optimizer/calcite/stats/FilterSelectivityEstimator.java:
##########
@@ -484,19 +484,261 @@ private Double 
computeRangePredicateSelectivity(Supplier<Double> defaultSelectiv
     }
 
     final List<ColStatistics> colStats = 
scan.getColStat(Collections.singletonList(inputRefIndex));
-    if (colStats.isEmpty() || !isHistogramAvailable(colStats.get(0))) {
+    if (colStats.isEmpty()) {
       return defaultSelectivity.get();
     }
 
-    final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(colStats.get(0).getHistogram()));
-    double rawSelectivity = rangedSelectivity(kll, boundaries);
+    final ColStatistics cs = colStats.get(0);
+    if (isHistogramAvailable(cs)) {
+      final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(cs.getHistogram()));
+      double rawSelectivity = rangedSelectivity(kll, boundaries);
+      if (inverseBool) {
+        // when inverseBool == true, this is a NOT_BETWEEN and selectivity 
must be inverted
+        // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
+        double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
+        rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      }
+      return scaleSelectivityToNullableValues(kll, rawSelectivity, scan);
+    }
+
+    if (isUniformWithinRangeEnabled() && hasUsableMinMax(cs)) {
+      RelDataType columnType = 
scan.getRowType().getFieldList().get(inputRefIndex).getType();
+      Double uniformSelectivity = computeUniformRangeSelectivity(cs, 
boundaries, scan, inverseBool, typeRange,
+          columnType);
+      if (uniformSelectivity != null) {
+        return uniformSelectivity;
+      }
+    }
+
+    return defaultSelectivity.get();
+  }
+
+  private boolean isUniformWithinRangeEnabled() {
+    HiveConfPlannerContext ctx =
+        
childRel.getCluster().getPlanner().getContext().unwrap(HiveConfPlannerContext.class);
+    return ctx == null || ctx.isUniformWithinRange();
+  }
+
+  private static boolean hasUsableMinMax(ColStatistics cs) {
+    ColStatistics.Range range = cs.getRange();
+    return range != null && range.minValue != null && range.maxValue != null;
+  }
+
+  /**
+   * Converts column MIN/MAX statistics into the same numeric space used by 
{@link #extractLiteral}.
+   * DATE column stats from HMS are stored as days since epoch; literals use 
epoch seconds.
+   */
+  private static Optional<float[]> convertColRangeToFloatBounds(ColStatistics 
cs, RelDataType columnType) {
+    ColStatistics.Range range = cs.getRange();
+    if (range == null || range.minValue == null || range.maxValue == null) {
+      return Optional.empty();
+    }
+    final float min;
+    final float max;
+    switch (columnType.getSqlTypeName()) {
+    case DATE:
+      min = range.minValue.longValue() * 86400L;
+      max = range.maxValue.longValue() * 86400L;
+      break;
+    case TIMESTAMP:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case TINYINT:
+      min = range.minValue.byteValue();
+      max = range.maxValue.byteValue();
+      break;
+    case SMALLINT:
+      min = range.minValue.shortValue();
+      max = range.maxValue.shortValue();
+      break;
+    case INTEGER:
+      min = range.minValue.intValue();
+      max = range.maxValue.intValue();
+      break;
+    case BIGINT:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case FLOAT:
+      min = range.minValue.floatValue();
+      max = range.maxValue.floatValue();
+      break;
+    case DOUBLE:
+      min = (float) range.minValue.doubleValue();
+      max = (float) range.maxValue.doubleValue();
+      break;
+    case DECIMAL:
+      min = new BigDecimal(range.minValue.toString()).floatValue();
+      max = new BigDecimal(range.maxValue.toString()).floatValue();
+      break;
+    default:
+      return Optional.empty();
+    }
+    return Optional.of(new float[] { min, max });
+  }
+
+  private Double computeUniformRangeSelectivity(ColStatistics cs, Range<Float> 
boundaries, HiveTableScan scan,
+      boolean inverseBool, Range<Float> typeRange, RelDataType columnType) {

Review Comment:
   I have the feeling that this method could be simplified by using the 
approach of computeTwoSidedUniformSelectivity 
(`intersect(intersect(minMaxRange, typeRange), boundaries)`) also for one-sided 
predicates. Could you try that, please?



##########
ql/src/java/org/apache/hadoop/hive/ql/optimizer/calcite/stats/FilterSelectivityEstimator.java:
##########
@@ -484,19 +484,261 @@ private Double 
computeRangePredicateSelectivity(Supplier<Double> defaultSelectiv
     }
 
     final List<ColStatistics> colStats = 
scan.getColStat(Collections.singletonList(inputRefIndex));
-    if (colStats.isEmpty() || !isHistogramAvailable(colStats.get(0))) {
+    if (colStats.isEmpty()) {
       return defaultSelectivity.get();
     }
 
-    final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(colStats.get(0).getHistogram()));
-    double rawSelectivity = rangedSelectivity(kll, boundaries);
+    final ColStatistics cs = colStats.get(0);
+    if (isHistogramAvailable(cs)) {
+      final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(cs.getHistogram()));
+      double rawSelectivity = rangedSelectivity(kll, boundaries);
+      if (inverseBool) {
+        // when inverseBool == true, this is a NOT_BETWEEN and selectivity 
must be inverted
+        // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
+        double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
+        rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      }
+      return scaleSelectivityToNullableValues(kll, rawSelectivity, scan);
+    }
+
+    if (isUniformWithinRangeEnabled() && hasUsableMinMax(cs)) {
+      RelDataType columnType = 
scan.getRowType().getFieldList().get(inputRefIndex).getType();
+      Double uniformSelectivity = computeUniformRangeSelectivity(cs, 
boundaries, scan, inverseBool, typeRange,
+          columnType);
+      if (uniformSelectivity != null) {
+        return uniformSelectivity;
+      }
+    }
+
+    return defaultSelectivity.get();
+  }
+
+  private boolean isUniformWithinRangeEnabled() {
+    HiveConfPlannerContext ctx =
+        
childRel.getCluster().getPlanner().getContext().unwrap(HiveConfPlannerContext.class);
+    return ctx == null || ctx.isUniformWithinRange();
+  }
+
+  private static boolean hasUsableMinMax(ColStatistics cs) {
+    ColStatistics.Range range = cs.getRange();
+    return range != null && range.minValue != null && range.maxValue != null;
+  }
+
+  /**
+   * Converts column MIN/MAX statistics into the same numeric space used by 
{@link #extractLiteral}.
+   * DATE column stats from HMS are stored as days since epoch; literals use 
epoch seconds.
+   */
+  private static Optional<float[]> convertColRangeToFloatBounds(ColStatistics 
cs, RelDataType columnType) {
+    ColStatistics.Range range = cs.getRange();
+    if (range == null || range.minValue == null || range.maxValue == null) {
+      return Optional.empty();
+    }
+    final float min;
+    final float max;
+    switch (columnType.getSqlTypeName()) {
+    case DATE:
+      min = range.minValue.longValue() * 86400L;
+      max = range.maxValue.longValue() * 86400L;
+      break;
+    case TIMESTAMP:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case TINYINT:
+      min = range.minValue.byteValue();
+      max = range.maxValue.byteValue();
+      break;
+    case SMALLINT:
+      min = range.minValue.shortValue();
+      max = range.maxValue.shortValue();
+      break;
+    case INTEGER:
+      min = range.minValue.intValue();
+      max = range.maxValue.intValue();
+      break;
+    case BIGINT:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case FLOAT:
+      min = range.minValue.floatValue();
+      max = range.maxValue.floatValue();
+      break;
+    case DOUBLE:
+      min = (float) range.minValue.doubleValue();
+      max = (float) range.maxValue.doubleValue();
+      break;
+    case DECIMAL:
+      min = new BigDecimal(range.minValue.toString()).floatValue();
+      max = new BigDecimal(range.maxValue.toString()).floatValue();
+      break;
+    default:
+      return Optional.empty();
+    }
+    return Optional.of(new float[] { min, max });
+  }
+
+  private Double computeUniformRangeSelectivity(ColStatistics cs, Range<Float> 
boundaries, HiveTableScan scan,
+      boolean inverseBool, Range<Float> typeRange, RelDataType columnType) {
+    Optional<float[]> minMax = convertColRangeToFloatBounds(cs, columnType);
+    if (minMax.isEmpty()) {
+      return null;
+    }
+    float min = minMax.get()[0];
+    float max = minMax.get()[1];
+
+    float lowerInfinite = Float.NEGATIVE_INFINITY;
+    float upperInfinite = Float.POSITIVE_INFINITY;
+    boolean isOneSidedUpper = Float.compare(boundaries.lowerEndpoint(), 
lowerInfinite) == 0
+        && Float.compare(boundaries.upperEndpoint(), upperInfinite) != 0;
+    boolean isOneSidedLower = Float.compare(boundaries.upperEndpoint(), 
upperInfinite) == 0
+        && Float.compare(boundaries.lowerEndpoint(), lowerInfinite) != 0;
+
+    double rawSelectivity;
+    if (isOneSidedUpper) {
+      rawSelectivity = computeOneSidedUniformSelectivity(min, max, 
boundaries.upperEndpoint(), true,
+          BoundType.CLOSED.equals(boundaries.upperBoundType()));
+    } else if (isOneSidedLower) {
+      rawSelectivity = computeOneSidedUniformSelectivity(min, max, 
boundaries.lowerEndpoint(), false,
+          BoundType.CLOSED.equals(boundaries.lowerBoundType()));
+    } else {
+      rawSelectivity = computeTwoSidedUniformSelectivity(min, max, boundaries, 
inverseBool, typeRange);
+    }
+
+    if (rawSelectivity < 0 || Double.isNaN(rawSelectivity) || 
Double.isInfinite(rawSelectivity)) {
+      return null;
+    }
+    return scaleSelectivityForNulls(cs, Math.min(1.0, Math.max(0.0, 
rawSelectivity)), scan);
+  }
+
+  /**
+   * Mirrors {@code StatsRulesProcFactory.EvaluateComparatorWithRange} 
semantics for one-sided predicates.
+   */
+  private static double computeOneSidedUniformSelectivity(float min, float 
max, float value, boolean upperBound,
+      boolean closedBound) {
+    Optional<Double> earlyReturn = applyOneSidedEarlyReturn(min, max, value, 
upperBound, closedBound);
+    if (earlyReturn.isPresent()) {
+      return earlyReturn.get();
+    }
+    float domainWidth = max - min;
+    if (domainWidth <= 0) {
+      return 0;
+    }
+    if (upperBound) {
+      return (value - min) / domainWidth;
+    }
+    return (max - value) / domainWidth;
+  }
+
+  private static Optional<Double> applyOneSidedEarlyReturn(float min, float 
max, float value, boolean upperBound,
+      boolean closedBound) {
+    if (upperBound) {
+      if (max < value || (Float.compare(max, value) == 0 && closedBound)) {
+        return Optional.of(1.0);
+      }
+      if (min > value || (Float.compare(min, value) == 0 && !closedBound)) {
+        return Optional.of(0.0);
+      }
+    } else {
+      if (min > value || (Float.compare(min, value) == 0 && closedBound)) {
+        return Optional.of(1.0);
+      }
+      if (max < value || (Float.compare(max, value) == 0 && !closedBound)) {
+        return Optional.of(0.0);
+      }
+    }
+    return Optional.empty();
+  }
+
+  private static double computeTwoSidedUniformSelectivity(float min, float 
max, Range<Float> boundaries,
+      boolean inverseBool, Range<Float> typeRange) {
+    if (Float.compare(min, max) == 0) {
+      double betweenSelectivity = isPointInClosedRange(boundaries, min) ? 1.0 
: 0.0;
+      return inverseBool ? 1.0 - betweenSelectivity : betweenSelectivity;
+    }
+
+    Range<Float> domain = Range.closedOpen(min, Math.nextUp(max));
+    Range<Float> predicateRange = convertRangeToClosedOpen(boundaries);
+
     if (inverseBool) {
-      // when inverseBool == true, this is a NOT_BETWEEN and selectivity must 
be inverted
-      // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
-      double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
-      rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      Range<Float> universe = domain;
+      if (typeRange != null) {
+        Range<Float> typeRangeClosedOpen = convertRangeToClosedOpen(typeRange);
+        universe = intersectClosedOpenRanges(domain, typeRangeClosedOpen);
+        if (universe == null) {
+          return 0;
+        }
+      }
+      float universeWidth = rangeWidth(universe);
+      if (universeWidth <= 0) {
+        return 0;
+      }
+      Range<Float> betweenIntersect = intersectClosedOpenRanges(universe, 
predicateRange);
+      float betweenWidth = betweenIntersect == null ? 0 : 
rangeWidth(betweenIntersect);
+      return 1.0 - betweenWidth / universeWidth;
+    }
+
+    Range<Float> intersect = intersectClosedOpenRanges(domain, predicateRange);
+    float overlapWidth = intersect == null ? 0 : rangeWidth(intersect);
+    float domainWidth = max - min;
+    if (domainWidth <= 0) {
+      return 0;
+    }
+    return overlapWidth / domainWidth;
+  }
+
+  private static boolean isPointInClosedRange(Range<Float> boundaries, float 
point) {
+    if (boundaries.isEmpty()) {
+      return false;
+    }
+    float lower = boundaries.lowerEndpoint();
+    float upper = boundaries.upperEndpoint();
+    boolean lowerOk = BoundType.CLOSED.equals(boundaries.lowerBoundType())
+        ? Float.compare(point, lower) >= 0
+        : Float.compare(point, lower) > 0;
+    boolean upperOk = BoundType.CLOSED.equals(boundaries.upperBoundType())
+        ? Float.compare(point, upper) <= 0
+        : Float.compare(point, upper) < 0;
+    return lowerOk && upperOk;
+  }
+
+  private static Range<Float> intersectClosedOpenRanges(Range<Float> left, 
Range<Float> right) {
+    if (!left.isConnected(right)) {
+      return null;
+    }
+    Range<Float> intersection = left.intersection(right);
+    if (intersection.isEmpty()) {
+      return null;
+    }

Review Comment:
   Please return a valid range. If the Range is empty, then return 
`Range.closedOpen(0f, 0f)`. The callers that check for `null` can then be 
simplified.



##########
ql/src/test/org/apache/hadoop/hive/ql/optimizer/calcite/stats/TestFilterSelectivityEstimator.java:
##########
@@ -1202,4 +1210,184 @@ private static long timestampMillis(String timestamp) {
   private static long timestamp(String timestamp) {
     return timestampMillis(timestamp) / 1000;
   }
+
+  private static final int INTEGER_FIELD_INDEX = 6; // f_integer
+  private static final int DATE_FIELD_INDEX = 9; // f_date
+
+  private void setupMinMaxNoHistogram(float min, float max) {
+    setupMinMaxNoHistogram(min, max, 0);
+  }
+
+  private void setupMinMaxNoHistogram(float min, float max, long numNulls) {
+    stats = new ColStatistics();
+    stats.setHistogram(null);
+    stats.setRange(min, max);
+    stats.setNumNulls(numNulls);
+    currentInputRef = REX_BUILDER.makeInputRef(scan, INTEGER_FIELD_INDEX);
+    doReturn(Collections.singletonList(stats)).when(tableMock)
+        .getColStat(Collections.singletonList(INTEGER_FIELD_INDEX));
+  }
+
+  private RelNode createScanWithPlanner(HiveConf conf) {
+    RelOptPlanner planner = CalcitePlanner.createPlanner(conf);
+    RelOptCluster cluster = RelOptCluster.create(planner, REX_BUILDER);
+    RelBuilder relBuilder = HiveRelFactories.HIVE_BUILDER.create(cluster, 
schemaMock);
+    HiveTableScan tableScan =
+        new HiveTableScan(cluster, cluster.traitSetOf(HiveRelNode.CONVENTION), 
tableMock, "table", null, false, false);
+    return relBuilder.push(tableScan).build();
+  }
+
+  @Test
+  public void testComparisonMinMaxNoHistogram() {
+    setupMinMaxNoHistogram(0, 100);
+    RexNode int50 = REX_BUILDER.makeLiteral(50, 
TYPE_FACTORY.createSqlType(INTEGER), true);
+    RexNode filter = 
REX_BUILDER.makeCall(SqlStdOperatorTable.LESS_THAN_OR_EQUAL, currentInputRef, 
int50);
+    FilterSelectivityEstimator estimator = new 
FilterSelectivityEstimator(scan, mq);
+    Assert.assertEquals(0.5, estimator.estimateSelectivity(filter), DELTA);
+  }
+
+  @Test
+  public void testComparisonMinMaxNoHistogramNoRange() {
+    stats = new ColStatistics();
+    stats.setHistogram(null);
+    currentInputRef = REX_BUILDER.makeInputRef(scan, INTEGER_FIELD_INDEX);
+    doReturn(Collections.singletonList(stats)).when(tableMock)
+        .getColStat(Collections.singletonList(INTEGER_FIELD_INDEX));
+    RexNode filter = REX_BUILDER.makeCall(SqlStdOperatorTable.LESS_THAN, 
currentInputRef, int3);
+    FilterSelectivityEstimator estimator = new 
FilterSelectivityEstimator(scan, mq);
+    Assert.assertEquals(0.3333333333333333, 
estimator.estimateSelectivity(filter), DELTA);
+  }
+
+  @Test
+  public void testComparisonDateMinMaxNoHistogram() {
+    long minDays = LocalDate.parse("2020-11-01").toEpochDay();
+    long maxDays = LocalDate.parse("2020-11-07").toEpochDay();
+    stats = new ColStatistics();
+    stats.setHistogram(null);
+    stats.setRange(minDays, maxDays);
+    currentInputRef = REX_BUILDER.makeInputRef(scan, DATE_FIELD_INDEX);
+    doReturn(Collections.singletonList(stats)).when(tableMock)
+        .getColStat(Collections.singletonList(DATE_FIELD_INDEX));
+    RexNode filter = 
REX_BUILDER.makeCall(SqlStdOperatorTable.LESS_THAN_OR_EQUAL, currentInputRef,
+        literalDate("2020-11-04"));
+    FilterSelectivityEstimator estimator = new 
FilterSelectivityEstimator(scan, mq);
+    Assert.assertEquals(0.5, estimator.estimateSelectivity(filter), DELTA);

Review Comment:
   Similar to https://github.com/apache/hive/pull/6746/changes#r3903821835, I 
would have expected a selectivity of 4/7 or 0.5714286.



##########
ql/src/java/org/apache/hadoop/hive/ql/optimizer/calcite/stats/FilterSelectivityEstimator.java:
##########
@@ -484,19 +484,261 @@ private Double 
computeRangePredicateSelectivity(Supplier<Double> defaultSelectiv
     }
 
     final List<ColStatistics> colStats = 
scan.getColStat(Collections.singletonList(inputRefIndex));
-    if (colStats.isEmpty() || !isHistogramAvailable(colStats.get(0))) {
+    if (colStats.isEmpty()) {
       return defaultSelectivity.get();
     }
 
-    final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(colStats.get(0).getHistogram()));
-    double rawSelectivity = rangedSelectivity(kll, boundaries);
+    final ColStatistics cs = colStats.get(0);
+    if (isHistogramAvailable(cs)) {
+      final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(cs.getHistogram()));
+      double rawSelectivity = rangedSelectivity(kll, boundaries);
+      if (inverseBool) {
+        // when inverseBool == true, this is a NOT_BETWEEN and selectivity 
must be inverted
+        // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
+        double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
+        rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      }
+      return scaleSelectivityToNullableValues(kll, rawSelectivity, scan);
+    }
+
+    if (isUniformWithinRangeEnabled() && hasUsableMinMax(cs)) {
+      RelDataType columnType = 
scan.getRowType().getFieldList().get(inputRefIndex).getType();
+      Double uniformSelectivity = computeUniformRangeSelectivity(cs, 
boundaries, scan, inverseBool, typeRange,
+          columnType);
+      if (uniformSelectivity != null) {
+        return uniformSelectivity;
+      }
+    }
+
+    return defaultSelectivity.get();
+  }
+
+  private boolean isUniformWithinRangeEnabled() {
+    HiveConfPlannerContext ctx =
+        
childRel.getCluster().getPlanner().getContext().unwrap(HiveConfPlannerContext.class);
+    return ctx == null || ctx.isUniformWithinRange();
+  }
+
+  private static boolean hasUsableMinMax(ColStatistics cs) {
+    ColStatistics.Range range = cs.getRange();
+    return range != null && range.minValue != null && range.maxValue != null;
+  }
+
+  /**
+   * Converts column MIN/MAX statistics into the same numeric space used by 
{@link #extractLiteral}.
+   * DATE column stats from HMS are stored as days since epoch; literals use 
epoch seconds.
+   */
+  private static Optional<float[]> convertColRangeToFloatBounds(ColStatistics 
cs, RelDataType columnType) {
+    ColStatistics.Range range = cs.getRange();
+    if (range == null || range.minValue == null || range.maxValue == null) {
+      return Optional.empty();
+    }
+    final float min;
+    final float max;
+    switch (columnType.getSqlTypeName()) {
+    case DATE:
+      min = range.minValue.longValue() * 86400L;
+      max = range.maxValue.longValue() * 86400L;
+      break;
+    case TIMESTAMP:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case TINYINT:
+      min = range.minValue.byteValue();
+      max = range.maxValue.byteValue();
+      break;
+    case SMALLINT:
+      min = range.minValue.shortValue();
+      max = range.maxValue.shortValue();
+      break;
+    case INTEGER:
+      min = range.minValue.intValue();
+      max = range.maxValue.intValue();
+      break;
+    case BIGINT:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case FLOAT:
+      min = range.minValue.floatValue();
+      max = range.maxValue.floatValue();
+      break;
+    case DOUBLE:
+      min = (float) range.minValue.doubleValue();
+      max = (float) range.maxValue.doubleValue();
+      break;
+    case DECIMAL:
+      min = new BigDecimal(range.minValue.toString()).floatValue();
+      max = new BigDecimal(range.maxValue.toString()).floatValue();
+      break;
+    default:
+      return Optional.empty();
+    }
+    return Optional.of(new float[] { min, max });

Review Comment:
   Please return a Optional<Range<...>>.



##########
ql/src/java/org/apache/hadoop/hive/ql/optimizer/calcite/stats/FilterSelectivityEstimator.java:
##########
@@ -484,19 +484,261 @@ private Double 
computeRangePredicateSelectivity(Supplier<Double> defaultSelectiv
     }
 
     final List<ColStatistics> colStats = 
scan.getColStat(Collections.singletonList(inputRefIndex));
-    if (colStats.isEmpty() || !isHistogramAvailable(colStats.get(0))) {
+    if (colStats.isEmpty()) {
       return defaultSelectivity.get();
     }
 
-    final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(colStats.get(0).getHistogram()));
-    double rawSelectivity = rangedSelectivity(kll, boundaries);
+    final ColStatistics cs = colStats.get(0);
+    if (isHistogramAvailable(cs)) {
+      final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(cs.getHistogram()));
+      double rawSelectivity = rangedSelectivity(kll, boundaries);
+      if (inverseBool) {
+        // when inverseBool == true, this is a NOT_BETWEEN and selectivity 
must be inverted
+        // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
+        double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
+        rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      }
+      return scaleSelectivityToNullableValues(kll, rawSelectivity, scan);
+    }
+
+    if (isUniformWithinRangeEnabled() && hasUsableMinMax(cs)) {
+      RelDataType columnType = 
scan.getRowType().getFieldList().get(inputRefIndex).getType();
+      Double uniformSelectivity = computeUniformRangeSelectivity(cs, 
boundaries, scan, inverseBool, typeRange,
+          columnType);
+      if (uniformSelectivity != null) {
+        return uniformSelectivity;
+      }
+    }
+
+    return defaultSelectivity.get();
+  }
+
+  private boolean isUniformWithinRangeEnabled() {
+    HiveConfPlannerContext ctx =
+        
childRel.getCluster().getPlanner().getContext().unwrap(HiveConfPlannerContext.class);
+    return ctx == null || ctx.isUniformWithinRange();
+  }
+
+  private static boolean hasUsableMinMax(ColStatistics cs) {
+    ColStatistics.Range range = cs.getRange();
+    return range != null && range.minValue != null && range.maxValue != null;
+  }
+
+  /**
+   * Converts column MIN/MAX statistics into the same numeric space used by 
{@link #extractLiteral}.
+   * DATE column stats from HMS are stored as days since epoch; literals use 
epoch seconds.
+   */
+  private static Optional<float[]> convertColRangeToFloatBounds(ColStatistics 
cs, RelDataType columnType) {
+    ColStatistics.Range range = cs.getRange();
+    if (range == null || range.minValue == null || range.maxValue == null) {
+      return Optional.empty();
+    }
+    final float min;
+    final float max;
+    switch (columnType.getSqlTypeName()) {
+    case DATE:
+      min = range.minValue.longValue() * 86400L;
+      max = range.maxValue.longValue() * 86400L;
+      break;
+    case TIMESTAMP:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case TINYINT:
+      min = range.minValue.byteValue();
+      max = range.maxValue.byteValue();
+      break;
+    case SMALLINT:
+      min = range.minValue.shortValue();
+      max = range.maxValue.shortValue();
+      break;
+    case INTEGER:
+      min = range.minValue.intValue();
+      max = range.maxValue.intValue();
+      break;
+    case BIGINT:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case FLOAT:
+      min = range.minValue.floatValue();
+      max = range.maxValue.floatValue();
+      break;
+    case DOUBLE:
+      min = (float) range.minValue.doubleValue();
+      max = (float) range.maxValue.doubleValue();
+      break;
+    case DECIMAL:
+      min = new BigDecimal(range.minValue.toString()).floatValue();
+      max = new BigDecimal(range.maxValue.toString()).floatValue();
+      break;
+    default:
+      return Optional.empty();
+    }
+    return Optional.of(new float[] { min, max });
+  }
+
+  private Double computeUniformRangeSelectivity(ColStatistics cs, Range<Float> 
boundaries, HiveTableScan scan,
+      boolean inverseBool, Range<Float> typeRange, RelDataType columnType) {
+    Optional<float[]> minMax = convertColRangeToFloatBounds(cs, columnType);
+    if (minMax.isEmpty()) {
+      return null;
+    }
+    float min = minMax.get()[0];
+    float max = minMax.get()[1];
+
+    float lowerInfinite = Float.NEGATIVE_INFINITY;
+    float upperInfinite = Float.POSITIVE_INFINITY;
+    boolean isOneSidedUpper = Float.compare(boundaries.lowerEndpoint(), 
lowerInfinite) == 0
+        && Float.compare(boundaries.upperEndpoint(), upperInfinite) != 0;
+    boolean isOneSidedLower = Float.compare(boundaries.upperEndpoint(), 
upperInfinite) == 0
+        && Float.compare(boundaries.lowerEndpoint(), lowerInfinite) != 0;
+
+    double rawSelectivity;
+    if (isOneSidedUpper) {
+      rawSelectivity = computeOneSidedUniformSelectivity(min, max, 
boundaries.upperEndpoint(), true,
+          BoundType.CLOSED.equals(boundaries.upperBoundType()));
+    } else if (isOneSidedLower) {
+      rawSelectivity = computeOneSidedUniformSelectivity(min, max, 
boundaries.lowerEndpoint(), false,
+          BoundType.CLOSED.equals(boundaries.lowerBoundType()));
+    } else {
+      rawSelectivity = computeTwoSidedUniformSelectivity(min, max, boundaries, 
inverseBool, typeRange);
+    }
+
+    if (rawSelectivity < 0 || Double.isNaN(rawSelectivity) || 
Double.isInfinite(rawSelectivity)) {
+      return null;
+    }
+    return scaleSelectivityForNulls(cs, Math.min(1.0, Math.max(0.0, 
rawSelectivity)), scan);
+  }
+
+  /**
+   * Mirrors {@code StatsRulesProcFactory.EvaluateComparatorWithRange} 
semantics for one-sided predicates.
+   */
+  private static double computeOneSidedUniformSelectivity(float min, float 
max, float value, boolean upperBound,
+      boolean closedBound) {
+    Optional<Double> earlyReturn = applyOneSidedEarlyReturn(min, max, value, 
upperBound, closedBound);
+    if (earlyReturn.isPresent()) {
+      return earlyReturn.get();
+    }
+    float domainWidth = max - min;
+    if (domainWidth <= 0) {
+      return 0;
+    }
+    if (upperBound) {
+      return (value - min) / domainWidth;
+    }
+    return (max - value) / domainWidth;
+  }
+
+  private static Optional<Double> applyOneSidedEarlyReturn(float min, float 
max, float value, boolean upperBound,
+      boolean closedBound) {
+    if (upperBound) {
+      if (max < value || (Float.compare(max, value) == 0 && closedBound)) {
+        return Optional.of(1.0);
+      }
+      if (min > value || (Float.compare(min, value) == 0 && !closedBound)) {
+        return Optional.of(0.0);
+      }
+    } else {
+      if (min > value || (Float.compare(min, value) == 0 && closedBound)) {
+        return Optional.of(1.0);
+      }
+      if (max < value || (Float.compare(max, value) == 0 && !closedBound)) {
+        return Optional.of(0.0);
+      }
+    }
+    return Optional.empty();
+  }
+
+  private static double computeTwoSidedUniformSelectivity(float min, float 
max, Range<Float> boundaries,
+      boolean inverseBool, Range<Float> typeRange) {
+    if (Float.compare(min, max) == 0) {
+      double betweenSelectivity = isPointInClosedRange(boundaries, min) ? 1.0 
: 0.0;
+      return inverseBool ? 1.0 - betweenSelectivity : betweenSelectivity;
+    }
+
+    Range<Float> domain = Range.closedOpen(min, Math.nextUp(max));
+    Range<Float> predicateRange = convertRangeToClosedOpen(boundaries);
+
     if (inverseBool) {
-      // when inverseBool == true, this is a NOT_BETWEEN and selectivity must 
be inverted
-      // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
-      double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
-      rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      Range<Float> universe = domain;
+      if (typeRange != null) {
+        Range<Float> typeRangeClosedOpen = convertRangeToClosedOpen(typeRange);
+        universe = intersectClosedOpenRanges(domain, typeRangeClosedOpen);
+        if (universe == null) {
+          return 0;
+        }
+      }
+      float universeWidth = rangeWidth(universe);
+      if (universeWidth <= 0) {
+        return 0;
+      }
+      Range<Float> betweenIntersect = intersectClosedOpenRanges(universe, 
predicateRange);
+      float betweenWidth = betweenIntersect == null ? 0 : 
rangeWidth(betweenIntersect);
+      return 1.0 - betweenWidth / universeWidth;
+    }
+
+    Range<Float> intersect = intersectClosedOpenRanges(domain, predicateRange);
+    float overlapWidth = intersect == null ? 0 : rangeWidth(intersect);
+    float domainWidth = max - min;
+    if (domainWidth <= 0) {
+      return 0;
+    }
+    return overlapWidth / domainWidth;
+  }
+
+  private static boolean isPointInClosedRange(Range<Float> boundaries, float 
point) {
+    if (boundaries.isEmpty()) {
+      return false;
+    }
+    float lower = boundaries.lowerEndpoint();
+    float upper = boundaries.upperEndpoint();
+    boolean lowerOk = BoundType.CLOSED.equals(boundaries.lowerBoundType())
+        ? Float.compare(point, lower) >= 0
+        : Float.compare(point, lower) > 0;
+    boolean upperOk = BoundType.CLOSED.equals(boundaries.upperBoundType())
+        ? Float.compare(point, upper) <= 0
+        : Float.compare(point, upper) < 0;
+    return lowerOk && upperOk;
+  }
+
+  private static Range<Float> intersectClosedOpenRanges(Range<Float> left, 
Range<Float> right) {

Review Comment:
   How about calling this method "intersectRanges"? I don't see why it should 
be limited to closed-open ranges.



##########
ql/src/java/org/apache/hadoop/hive/ql/optimizer/calcite/stats/FilterSelectivityEstimator.java:
##########
@@ -484,19 +484,261 @@ private Double 
computeRangePredicateSelectivity(Supplier<Double> defaultSelectiv
     }
 
     final List<ColStatistics> colStats = 
scan.getColStat(Collections.singletonList(inputRefIndex));
-    if (colStats.isEmpty() || !isHistogramAvailable(colStats.get(0))) {
+    if (colStats.isEmpty()) {
       return defaultSelectivity.get();
     }
 
-    final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(colStats.get(0).getHistogram()));
-    double rawSelectivity = rangedSelectivity(kll, boundaries);
+    final ColStatistics cs = colStats.get(0);
+    if (isHistogramAvailable(cs)) {
+      final KllFloatsSketch kll = 
KllFloatsSketch.heapify(Memory.wrap(cs.getHistogram()));
+      double rawSelectivity = rangedSelectivity(kll, boundaries);
+      if (inverseBool) {
+        // when inverseBool == true, this is a NOT_BETWEEN and selectivity 
must be inverted
+        // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
+        double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
+        rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      }
+      return scaleSelectivityToNullableValues(kll, rawSelectivity, scan);
+    }
+
+    if (isUniformWithinRangeEnabled() && hasUsableMinMax(cs)) {
+      RelDataType columnType = 
scan.getRowType().getFieldList().get(inputRefIndex).getType();
+      Double uniformSelectivity = computeUniformRangeSelectivity(cs, 
boundaries, scan, inverseBool, typeRange,
+          columnType);
+      if (uniformSelectivity != null) {
+        return uniformSelectivity;
+      }
+    }
+
+    return defaultSelectivity.get();
+  }
+
+  private boolean isUniformWithinRangeEnabled() {
+    HiveConfPlannerContext ctx =
+        
childRel.getCluster().getPlanner().getContext().unwrap(HiveConfPlannerContext.class);
+    return ctx == null || ctx.isUniformWithinRange();
+  }
+
+  private static boolean hasUsableMinMax(ColStatistics cs) {
+    ColStatistics.Range range = cs.getRange();
+    return range != null && range.minValue != null && range.maxValue != null;
+  }
+
+  /**
+   * Converts column MIN/MAX statistics into the same numeric space used by 
{@link #extractLiteral}.
+   * DATE column stats from HMS are stored as days since epoch; literals use 
epoch seconds.
+   */
+  private static Optional<float[]> convertColRangeToFloatBounds(ColStatistics 
cs, RelDataType columnType) {
+    ColStatistics.Range range = cs.getRange();
+    if (range == null || range.minValue == null || range.maxValue == null) {
+      return Optional.empty();
+    }
+    final float min;
+    final float max;
+    switch (columnType.getSqlTypeName()) {
+    case DATE:
+      min = range.minValue.longValue() * 86400L;
+      max = range.maxValue.longValue() * 86400L;
+      break;
+    case TIMESTAMP:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case TINYINT:
+      min = range.minValue.byteValue();
+      max = range.maxValue.byteValue();
+      break;
+    case SMALLINT:
+      min = range.minValue.shortValue();
+      max = range.maxValue.shortValue();
+      break;
+    case INTEGER:
+      min = range.minValue.intValue();
+      max = range.maxValue.intValue();
+      break;
+    case BIGINT:
+      min = range.minValue.longValue();
+      max = range.maxValue.longValue();
+      break;
+    case FLOAT:
+      min = range.minValue.floatValue();
+      max = range.maxValue.floatValue();
+      break;
+    case DOUBLE:
+      min = (float) range.minValue.doubleValue();
+      max = (float) range.maxValue.doubleValue();
+      break;
+    case DECIMAL:
+      min = new BigDecimal(range.minValue.toString()).floatValue();
+      max = new BigDecimal(range.maxValue.toString()).floatValue();
+      break;
+    default:
+      return Optional.empty();
+    }
+    return Optional.of(new float[] { min, max });
+  }
+
+  private Double computeUniformRangeSelectivity(ColStatistics cs, Range<Float> 
boundaries, HiveTableScan scan,
+      boolean inverseBool, Range<Float> typeRange, RelDataType columnType) {
+    Optional<float[]> minMax = convertColRangeToFloatBounds(cs, columnType);
+    if (minMax.isEmpty()) {
+      return null;
+    }
+    float min = minMax.get()[0];
+    float max = minMax.get()[1];
+
+    float lowerInfinite = Float.NEGATIVE_INFINITY;
+    float upperInfinite = Float.POSITIVE_INFINITY;
+    boolean isOneSidedUpper = Float.compare(boundaries.lowerEndpoint(), 
lowerInfinite) == 0
+        && Float.compare(boundaries.upperEndpoint(), upperInfinite) != 0;
+    boolean isOneSidedLower = Float.compare(boundaries.upperEndpoint(), 
upperInfinite) == 0
+        && Float.compare(boundaries.lowerEndpoint(), lowerInfinite) != 0;
+
+    double rawSelectivity;
+    if (isOneSidedUpper) {
+      rawSelectivity = computeOneSidedUniformSelectivity(min, max, 
boundaries.upperEndpoint(), true,
+          BoundType.CLOSED.equals(boundaries.upperBoundType()));
+    } else if (isOneSidedLower) {
+      rawSelectivity = computeOneSidedUniformSelectivity(min, max, 
boundaries.lowerEndpoint(), false,
+          BoundType.CLOSED.equals(boundaries.lowerBoundType()));
+    } else {
+      rawSelectivity = computeTwoSidedUniformSelectivity(min, max, boundaries, 
inverseBool, typeRange);
+    }
+
+    if (rawSelectivity < 0 || Double.isNaN(rawSelectivity) || 
Double.isInfinite(rawSelectivity)) {
+      return null;
+    }
+    return scaleSelectivityForNulls(cs, Math.min(1.0, Math.max(0.0, 
rawSelectivity)), scan);
+  }
+
+  /**
+   * Mirrors {@code StatsRulesProcFactory.EvaluateComparatorWithRange} 
semantics for one-sided predicates.
+   */
+  private static double computeOneSidedUniformSelectivity(float min, float 
max, float value, boolean upperBound,
+      boolean closedBound) {
+    Optional<Double> earlyReturn = applyOneSidedEarlyReturn(min, max, value, 
upperBound, closedBound);
+    if (earlyReturn.isPresent()) {
+      return earlyReturn.get();
+    }
+    float domainWidth = max - min;
+    if (domainWidth <= 0) {
+      return 0;
+    }
+    if (upperBound) {
+      return (value - min) / domainWidth;
+    }
+    return (max - value) / domainWidth;
+  }
+
+  private static Optional<Double> applyOneSidedEarlyReturn(float min, float 
max, float value, boolean upperBound,
+      boolean closedBound) {
+    if (upperBound) {
+      if (max < value || (Float.compare(max, value) == 0 && closedBound)) {
+        return Optional.of(1.0);
+      }
+      if (min > value || (Float.compare(min, value) == 0 && !closedBound)) {
+        return Optional.of(0.0);
+      }
+    } else {
+      if (min > value || (Float.compare(min, value) == 0 && closedBound)) {
+        return Optional.of(1.0);
+      }
+      if (max < value || (Float.compare(max, value) == 0 && !closedBound)) {
+        return Optional.of(0.0);
+      }
+    }
+    return Optional.empty();
+  }
+
+  private static double computeTwoSidedUniformSelectivity(float min, float 
max, Range<Float> boundaries,
+      boolean inverseBool, Range<Float> typeRange) {
+    if (Float.compare(min, max) == 0) {
+      double betweenSelectivity = isPointInClosedRange(boundaries, min) ? 1.0 
: 0.0;
+      return inverseBool ? 1.0 - betweenSelectivity : betweenSelectivity;
+    }
+
+    Range<Float> domain = Range.closedOpen(min, Math.nextUp(max));
+    Range<Float> predicateRange = convertRangeToClosedOpen(boundaries);
+
     if (inverseBool) {
-      // when inverseBool == true, this is a NOT_BETWEEN and selectivity must 
be inverted
-      // if there's a cast, the inversion is with respect to its codomain 
(range of the values of the cast)
-      double typeRangeSelectivity = rangedSelectivity(kll, typeRange);
-      rawSelectivity = typeRangeSelectivity - rawSelectivity;
+      Range<Float> universe = domain;
+      if (typeRange != null) {
+        Range<Float> typeRangeClosedOpen = convertRangeToClosedOpen(typeRange);
+        universe = intersectClosedOpenRanges(domain, typeRangeClosedOpen);
+        if (universe == null) {
+          return 0;
+        }
+      }
+      float universeWidth = rangeWidth(universe);
+      if (universeWidth <= 0) {
+        return 0;
+      }
+      Range<Float> betweenIntersect = intersectClosedOpenRanges(universe, 
predicateRange);
+      float betweenWidth = betweenIntersect == null ? 0 : 
rangeWidth(betweenIntersect);
+      return 1.0 - betweenWidth / universeWidth;
+    }
+
+    Range<Float> intersect = intersectClosedOpenRanges(domain, predicateRange);
+    float overlapWidth = intersect == null ? 0 : rangeWidth(intersect);
+    float domainWidth = max - min;
+    if (domainWidth <= 0) {
+      return 0;
+    }
+    return overlapWidth / domainWidth;
+  }
+
+  private static boolean isPointInClosedRange(Range<Float> boundaries, float 
point) {

Review Comment:
   Please use com.google.common.collect.Range#contains.



##########
ql/src/test/org/apache/hadoop/hive/ql/optimizer/calcite/stats/TestFilterSelectivityEstimator.java:
##########
@@ -1202,4 +1210,184 @@ private static long timestampMillis(String timestamp) {
   private static long timestamp(String timestamp) {
     return timestampMillis(timestamp) / 1000;
   }
+
+  private static final int INTEGER_FIELD_INDEX = 6; // f_integer
+  private static final int DATE_FIELD_INDEX = 9; // f_date
+
+  private void setupMinMaxNoHistogram(float min, float max) {
+    setupMinMaxNoHistogram(min, max, 0);
+  }
+
+  private void setupMinMaxNoHistogram(float min, float max, long numNulls) {
+    stats = new ColStatistics();
+    stats.setHistogram(null);
+    stats.setRange(min, max);
+    stats.setNumNulls(numNulls);
+    currentInputRef = REX_BUILDER.makeInputRef(scan, INTEGER_FIELD_INDEX);
+    doReturn(Collections.singletonList(stats)).when(tableMock)
+        .getColStat(Collections.singletonList(INTEGER_FIELD_INDEX));
+  }
+
+  private RelNode createScanWithPlanner(HiveConf conf) {
+    RelOptPlanner planner = CalcitePlanner.createPlanner(conf);
+    RelOptCluster cluster = RelOptCluster.create(planner, REX_BUILDER);
+    RelBuilder relBuilder = HiveRelFactories.HIVE_BUILDER.create(cluster, 
schemaMock);
+    HiveTableScan tableScan =
+        new HiveTableScan(cluster, cluster.traitSetOf(HiveRelNode.CONVENTION), 
tableMock, "table", null, false, false);
+    return relBuilder.push(tableScan).build();
+  }
+
+  @Test
+  public void testComparisonMinMaxNoHistogram() {
+    setupMinMaxNoHistogram(0, 100);
+    RexNode int50 = REX_BUILDER.makeLiteral(50, 
TYPE_FACTORY.createSqlType(INTEGER), true);
+    RexNode filter = 
REX_BUILDER.makeCall(SqlStdOperatorTable.LESS_THAN_OR_EQUAL, currentInputRef, 
int50);
+    FilterSelectivityEstimator estimator = new 
FilterSelectivityEstimator(scan, mq);
+    Assert.assertEquals(0.5, estimator.estimateSelectivity(filter), DELTA);

Review Comment:
   Shouldn't the expected result be 51/101? There are 101 integers in the 
interval [0, 100] and 51 integers from that interval that fulfill the predicate 
`x <= 50`, i.e., [0, 50]. So I would have expected the selectivity to be 51/101 
or about 0.5049505.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to