Jefffrey commented on code in PR #11173:
URL: https://github.com/apache/arrow-rs/pull/11173#discussion_r4079029308
##########
parquet/src/arrow/arrow_writer/mod.rs:
##########
@@ -1145,21 +1145,12 @@ impl ArrowColumnWriter {
let non_null = levels.non_null_indices();
match array.as_any_dictionary_opt() {
Some(dict) => {
- // For dictionary arrays, hash the integer keys rather
than the actual values.
- // Key cardinality equals value cardinality, so
distinct-value counting stays
- // correct while avoiding the cost of hashing
arbitrary-length values.
- let keys = dict.keys();
- let key_data = keys.to_data();
- let offset = key_data.offset();
- let width = arrow_key_byte_width(keys.data_type());
- if width > 0 {
- let buffer = key_data.buffers()[0].as_slice();
- // Only visit non-null rows to avoid counting nulls as
a distinct value.
- for &row in non_null {
- let pos = (offset + row) * width;
- seen.insert(hash_bytes(&buffer[pos..pos + width]));
- }
- }
+ // Hash values, not key indices: the same key index can
map to different
+ // values across batches, causing undercounting.
+ let values = dict.values();
+ let non_null_value_indices: Vec<usize> =
+ (0..values.len()).filter(|&i|
values.is_valid(i)).collect();
Review Comment:
is it valid to use
[`valid_indices`](https://docs.rs/arrow/latest/arrow/buffer/struct.NullBuffer.html#method.valid_indices)
here too? or doesnt really improve anything (quality wise)
##########
parquet/src/arrow/arrow_writer/mod.rs:
##########
@@ -1145,21 +1145,12 @@ impl ArrowColumnWriter {
let non_null = levels.non_null_indices();
match array.as_any_dictionary_opt() {
Some(dict) => {
- // For dictionary arrays, hash the integer keys rather
than the actual values.
- // Key cardinality equals value cardinality, so
distinct-value counting stays
- // correct while avoiding the cost of hashing
arbitrary-length values.
- let keys = dict.keys();
- let key_data = keys.to_data();
- let offset = key_data.offset();
- let width = arrow_key_byte_width(keys.data_type());
- if width > 0 {
- let buffer = key_data.buffers()[0].as_slice();
- // Only visit non-null rows to avoid counting nulls as
a distinct value.
- for &row in non_null {
- let pos = (offset + row) * width;
- seen.insert(hash_bytes(&buffer[pos..pos + width]));
- }
- }
+ // Hash values, not key indices: the same key index can
map to different
+ // values across batches, causing undercounting.
+ let values = dict.values();
Review Comment:
what happens if theres a value not referenced by a key? or thats not a
reachable scenario?
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]