This is an automated email from the ASF dual-hosted git repository.
etseidl pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow-rs.git
The following commit(s) were added to refs/heads/main by this push:
new e2db103480 Feat: support ndv for View types (#11163)
e2db103480 is described below
commit e2db1034808c13a6252d485922a9734ae742bd51
Author: RIchard Baah <[email protected]>
AuthorDate: Tue Sep 22 19:03:28 2026 -0400
Feat: support ndv for View types (#11163)
# Which issue does this PR close?
- Closes #10805.
# Rationale for this change
Utf8View and BinaryView were skipped in `update_distinct_values_seen`,
so distinct_count was never written for those column types even when
set_write_row_group_number_distinct_values(true) was set.
# What changes are included in this PR?
see PR
# Are these changes tested?
yes 1 round trip test
# Are there any user-facing changes?
not directly, new feature
---
parquet/src/arrow/arrow_writer/mod.rs | 54 ++++++++++++++++++++++++++++++++++-
1 file changed, 53 insertions(+), 1 deletion(-)
diff --git a/parquet/src/arrow/arrow_writer/mod.rs
b/parquet/src/arrow/arrow_writer/mod.rs
index 2056922dd1..4bee015d9c 100644
--- a/parquet/src/arrow/arrow_writer/mod.rs
+++ b/parquet/src/arrow/arrow_writer/mod.rs
@@ -2066,6 +2066,18 @@ fn update_distinct_values_seen(
seen.insert(hash_bytes(&buffer[start..start + byte_width]));
}
}
+ ArrowDataType::Utf8View => {
+ let string_view_array = array.as_string_view();
+ for &row in non_null_indices {
+
seen.insert(hash_bytes(string_view_array.value(row).as_bytes()));
+ }
+ }
+ ArrowDataType::BinaryView => {
+ let binary_view_array = array.as_binary_view();
+ for &row in non_null_indices {
+ seen.insert(hash_bytes(binary_view_array.value(row)));
+ }
+ }
data_type => {
if let Some(width) = fixed_byte_width(data_type) {
let buffer = data.buffers()[0].as_slice();
@@ -2074,7 +2086,7 @@ fn update_distinct_values_seen(
seen.insert(hash_bytes(&buffer[pos..pos + width]));
}
}
- // Utf8View, BinaryView, nested types: skip
+ // nested types (List, LargeList, etc.) are Parquet groups, not
leaf columns: skip
}
}
}
@@ -6495,6 +6507,46 @@ mod tests {
assert_eq!(count, cardinality as u64);
}
+ #[test]
+ fn test_number_distinct_values_view_types() {
+ // 5 distinct values repeated across 30 rows, with every 4th row null.
+ // Verifies Utf8View is counted correctly (BinaryView shares the same
code path).
+ let cardinality = 5u32;
+ let distinct_strings = ["alpha", "beta", "gamma", "delta", "epsilon"];
+
+ let string_view_col: ArrayRef =
Arc::new(StringViewArray::from_iter((0..30u32).map(|i| {
+ if i % 4 == 0 {
+ None
+ } else {
+ Some(distinct_strings[(i % cardinality) as usize])
+ }
+ })));
+
+ let schema = Arc::new(Schema::new(vec![Field::new(
+ "string_view_col",
+ DataType::Utf8View,
+ true,
+ )]));
+ let batch = RecordBatch::try_new(schema,
vec![string_view_col]).unwrap();
+
+ let props = WriterProperties::builder()
+ .set_write_row_group_number_distinct_values(true)
+ .build();
+ let mut parquet_bytes = Vec::new();
+ let mut writer =
+ ArrowWriter::try_new(&mut parquet_bytes, batch.schema(),
Some(props)).unwrap();
+ writer.write(&batch).unwrap();
+ let metadata = writer.close().unwrap();
+
+ let distinct_count = metadata
+ .row_group(0)
+ .column(0)
+ .statistics()
+ .and_then(|s| s.distinct_count_opt())
+ .expect("distinct_count should be set for Utf8View column");
+ assert_eq!(distinct_count, cardinality as u64);
+ }
+
#[test]
fn test_number_distinct_values_not_written_by_default() {
let array: ArrayRef = Arc::new(Int32Array::from_iter_values(0..100));