This is an automated email from the ASF dual-hosted git repository.
JingsongLi pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/paimon.git
The following commit(s) were added to refs/heads/master by this push:
new 345526e265 [iceberg] Omit unknown null counts and bounds in
IcebergDataFileMeta (#8732)
345526e265 is described below
commit 345526e265a7533645b42565c6a931b017f5037d
Author: Jiajia Li <[email protected]>
AuthorDate: Mon Jul 20 21:35:42 2026 +0800
[iceberg] Omit unknown null counts and bounds in IcebergDataFileMeta (#8732)
---
.../iceberg/manifest/IcebergDataFileMeta.java | 39 +++--
.../iceberg/manifest/IcebergDataFileMetaTest.java | 183 +++++++++++++++++++++
2 files changed, 208 insertions(+), 14 deletions(-)
diff --git
a/paimon-core/src/main/java/org/apache/paimon/iceberg/manifest/IcebergDataFileMeta.java
b/paimon-core/src/main/java/org/apache/paimon/iceberg/manifest/IcebergDataFileMeta.java
index 07543e10a0..7334f7528a 100644
---
a/paimon-core/src/main/java/org/apache/paimon/iceberg/manifest/IcebergDataFileMeta.java
+++
b/paimon-core/src/main/java/org/apache/paimon/iceberg/manifest/IcebergDataFileMeta.java
@@ -174,25 +174,36 @@ public class IcebergDataFileMeta {
}
int idx = indexMap.get(field.name());
- nullValueCounts.put(field.id(), stats.nullCounts().getLong(idx));
+ // an unknown null count must be omitted, not published as 0
+ if (!stats.nullCounts().isNullAt(idx)) {
+ nullValueCounts.put(field.id(),
stats.nullCounts().getLong(idx));
+ }
+
+ // these types have no bounds; skip before reading the stats slots
+ DataTypeRoot typeRoot = field.dataType().getTypeRoot();
+ if (typeRoot == DataTypeRoot.ARRAY
+ || typeRoot == DataTypeRoot.MAP
+ || typeRoot == DataTypeRoot.ROW
+ || typeRoot == DataTypeRoot.MULTISET
+ || typeRoot == DataTypeRoot.VARIANT
+ || typeRoot == DataTypeRoot.VECTOR
+ || typeRoot == DataTypeRoot.BLOB) {
+ continue;
+ }
+ // use the nullable copy of the type, so that an unknown (null)
min/max slot
+ // of a required field reads as null instead of garbage
InternalRow.FieldGetter fieldGetter =
- InternalRow.createFieldGetter(field.dataType(), idx);
+ InternalRow.createFieldGetter(field.dataType().nullable(),
idx);
Object minValue = fieldGetter.getFieldOrNull(stats.minValues());
Object maxValue = fieldGetter.getFieldOrNull(stats.maxValues());
if (minValue != null && maxValue != null) {
- DataTypeRoot typeRoot = field.dataType().getTypeRoot();
- if (typeRoot != DataTypeRoot.ARRAY
- && typeRoot != DataTypeRoot.MAP
- && typeRoot != DataTypeRoot.ROW
- && typeRoot != DataTypeRoot.MULTISET) {
- lowerBounds.put(
- field.id(),
- IcebergConversions.toByteBuffer(field.dataType(),
minValue).array());
- upperBounds.put(
- field.id(),
- IcebergConversions.toByteBuffer(field.dataType(),
maxValue).array());
- }
+ lowerBounds.put(
+ field.id(),
+ IcebergConversions.toByteBuffer(field.dataType(),
minValue).array());
+ upperBounds.put(
+ field.id(),
+ IcebergConversions.toByteBuffer(field.dataType(),
maxValue).array());
}
}
diff --git
a/paimon-core/src/test/java/org/apache/paimon/iceberg/manifest/IcebergDataFileMetaTest.java
b/paimon-core/src/test/java/org/apache/paimon/iceberg/manifest/IcebergDataFileMetaTest.java
index 9f9c3b9621..c4888312b0 100644
---
a/paimon-core/src/test/java/org/apache/paimon/iceberg/manifest/IcebergDataFileMetaTest.java
+++
b/paimon-core/src/test/java/org/apache/paimon/iceberg/manifest/IcebergDataFileMetaTest.java
@@ -19,12 +19,22 @@
package org.apache.paimon.iceberg.manifest;
import org.apache.paimon.TestKeyValueGenerator;
+import org.apache.paimon.data.BinaryArray;
+import org.apache.paimon.data.BinaryArrayWriter;
+import org.apache.paimon.data.BinaryRow;
+import org.apache.paimon.data.BinaryRowWriter;
+import org.apache.paimon.data.GenericMap;
+import org.apache.paimon.iceberg.metadata.IcebergDataField;
+import org.apache.paimon.iceberg.metadata.IcebergSchema;
+import org.apache.paimon.stats.SimpleStats;
import org.apache.paimon.types.DataField;
+import org.apache.paimon.types.DataTypes;
import org.apache.paimon.types.RowType;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
+import java.util.Arrays;
import java.util.List;
import java.util.Optional;
@@ -45,4 +55,177 @@ class IcebergDataFileMetaTest {
assertThat(partitionField.get().name()).isEqualTo("partition");
assertThat(partitionField.get().type()).isEqualTo(partitionType.notNull());
}
+
+ @Test
+ @DisplayName("Test unknown null count is omitted instead of published as
0")
+ void testUnknownNullCountOmitted() {
+ IcebergSchema icebergSchema =
+ new IcebergSchema(
+ 0,
+ Arrays.asList(
+ new IcebergDataField(1, "a", false, "int",
null),
+ new IcebergDataField(2, "b", false, "int",
null)));
+
+ BinaryRow values = new BinaryRow(2);
+ BinaryRowWriter rowWriter = new BinaryRowWriter(values);
+ rowWriter.setNullAt(0);
+ rowWriter.setNullAt(1);
+ rowWriter.complete();
+
+ // "a" has an unknown null count, "b" has a known null count of 0
+ BinaryArray nullCounts = new BinaryArray();
+ BinaryArrayWriter arrayWriter = new BinaryArrayWriter(nullCounts, 2,
8);
+ arrayWriter.setNullLong(0);
+ arrayWriter.writeLong(1, 0L);
+ arrayWriter.complete();
+
+ IcebergDataFileMeta meta =
+ IcebergDataFileMeta.create(
+ IcebergDataFileMeta.Content.DATA,
+ "path",
+ "parquet",
+ BinaryRow.EMPTY_ROW,
+ 10,
+ 100,
+ icebergSchema,
+ new SimpleStats(values, values, nullCounts),
+ null);
+
+ assertThat(meta.nullValueCounts().size()).isEqualTo(1);
+ assertThat(((GenericMap) meta.nullValueCounts()).get(1)).isNull();
+ assertThat(((GenericMap) meta.nullValueCounts()).get(2)).isEqualTo(0L);
+ }
+
+ @Test
+ @DisplayName("Test required field with unknown stats does not publish
garbage bounds")
+ void testRequiredFieldWithUnknownStats() {
+ IcebergSchema icebergSchema =
+ new IcebergSchema(
+ 0,
+ Arrays.asList(
+ new IcebergDataField(1, "id", true, "int",
null),
+ new IcebergDataField(2, "cnt", true, "long",
null)));
+
+ BinaryRow values = new BinaryRow(2);
+ BinaryRowWriter rowWriter = new BinaryRowWriter(values);
+ rowWriter.writeInt(0, 1);
+ rowWriter.setNullAt(1);
+ rowWriter.complete();
+
+ BinaryArray nullCounts = new BinaryArray();
+ BinaryArrayWriter arrayWriter = new BinaryArrayWriter(nullCounts, 2,
8);
+ arrayWriter.writeLong(0, 0L);
+ arrayWriter.setNullLong(1);
+ arrayWriter.complete();
+
+ IcebergDataFileMeta meta =
+ IcebergDataFileMeta.create(
+ IcebergDataFileMeta.Content.DATA,
+ "path",
+ "parquet",
+ BinaryRow.EMPTY_ROW,
+ 10,
+ 100,
+ icebergSchema,
+ new SimpleStats(values, values, nullCounts),
+ null);
+
+ assertThat(meta.nullValueCounts().size()).isEqualTo(1);
+ assertThat(((GenericMap) meta.nullValueCounts()).get(1)).isEqualTo(0L);
+ assertThat(meta.lowerBounds().size()).isEqualTo(1);
+ assertThat(meta.upperBounds().size()).isEqualTo(1);
+ }
+
+ @Test
+ @DisplayName("Test stats of a statsColumns subset are read by stats index,
not field ordinal")
+ void testStatsColumnsSubsetAlignment() {
+ IcebergSchema icebergSchema =
+ new IcebergSchema(
+ 0,
+ Arrays.asList(
+ new IcebergDataField(1, "a", false, "int",
null),
+ new IcebergDataField(2, "b", false, "long",
null)));
+
+ // stats only cover "b": slot 0 in the stats, ordinal 1 in the schema
+ BinaryRow values = new BinaryRow(1);
+ BinaryRowWriter rowWriter = new BinaryRowWriter(values);
+ rowWriter.writeLong(0, 7L);
+ rowWriter.complete();
+
+ BinaryArray nullCounts = new BinaryArray();
+ BinaryArrayWriter arrayWriter = new BinaryArrayWriter(nullCounts, 1,
8);
+ arrayWriter.writeLong(0, 2L);
+ arrayWriter.complete();
+
+ IcebergDataFileMeta meta =
+ IcebergDataFileMeta.create(
+ IcebergDataFileMeta.Content.DATA,
+ "path",
+ "parquet",
+ BinaryRow.EMPTY_ROW,
+ 10,
+ 100,
+ icebergSchema,
+ new SimpleStats(values, values, nullCounts),
+ Arrays.asList("b"));
+
+ assertThat(meta.nullValueCounts().size()).isEqualTo(1);
+ assertThat(((GenericMap) meta.nullValueCounts()).get(2)).isEqualTo(2L);
+ assertThat(meta.lowerBounds().size()).isEqualTo(1);
+ byte[] expectedLong7 = {7, 0, 0, 0, 0, 0, 0, 0};
+ assertThat((byte[]) ((GenericMap)
meta.lowerBounds()).get(2)).isEqualTo(expectedLong7);
+ assertThat((byte[]) ((GenericMap)
meta.upperBounds()).get(2)).isEqualTo(expectedLong7);
+ }
+
+ @Test
+ @DisplayName("Test required nested field with unknown stats is skipped
before reading slots")
+ void testRequiredNestedFieldSkipped() {
+ IcebergSchema icebergSchema =
+ new IcebergSchema(
+ 0,
+ Arrays.asList(
+ new IcebergDataField(1, "id", true, "int",
null),
+ new IcebergDataField(
+ new DataField(
+ 2,
+ "arr",
+
DataTypes.ARRAY(DataTypes.INT()).notNull()))));
+
+ BinaryRow minValues = new BinaryRow(2);
+ BinaryRowWriter minWriter = new BinaryRowWriter(minValues);
+ minWriter.writeInt(0, 1);
+ minWriter.setNullAt(1);
+ minWriter.complete();
+
+ BinaryRow maxValues = new BinaryRow(2);
+ BinaryRowWriter maxWriter = new BinaryRowWriter(maxValues);
+ maxWriter.writeInt(0, 5);
+ maxWriter.setNullAt(1);
+ maxWriter.complete();
+
+ BinaryArray nullCounts = new BinaryArray();
+ BinaryArrayWriter arrayWriter = new BinaryArrayWriter(nullCounts, 2,
8);
+ arrayWriter.writeLong(0, 0L);
+ arrayWriter.setNullLong(1);
+ arrayWriter.complete();
+
+ IcebergDataFileMeta meta =
+ IcebergDataFileMeta.create(
+ IcebergDataFileMeta.Content.DATA,
+ "path",
+ "parquet",
+ BinaryRow.EMPTY_ROW,
+ 10,
+ 100,
+ icebergSchema,
+ new SimpleStats(minValues, maxValues, nullCounts),
+ null);
+
+ assertThat(meta.lowerBounds().size()).isEqualTo(1);
+ assertThat(meta.upperBounds().size()).isEqualTo(1);
+ assertThat((byte[]) ((GenericMap) meta.lowerBounds()).get(1))
+ .isEqualTo(new byte[] {1, 0, 0, 0});
+ assertThat((byte[]) ((GenericMap) meta.upperBounds()).get(1))
+ .isEqualTo(new byte[] {5, 0, 0, 0});
+ }
}