mkleen commented on code in PR #11285:
URL: https://github.com/apache/arrow-rs/pull/11285#discussion_r4146270465


##########
parquet/src/arrow/arrow_reader/statistics/page_index.rs:
##########
@@ -0,0 +1,1594 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! Reads the page statistics stored in a Parquet `ColumnIndex` and puts them
+//! straight into Arrow arrays.
+//!
+//! The usual route first builds a [`ColumnIndexMetaData`] and then copies its
+//! values into Arrow arrays one by one. This module skips the middle step: it
+//! reads the stored bytes once and writes each value directly into the memory
+//! that will become the final Arrow array.
+//!
+//! [`ColumnIndexMetaData`]: 
crate::file::page_index::column_index::ColumnIndexMetaData
+
+use super::{DataPageStatistics, from_bytes_to_i256};
+use super::{from_bytes_to_f16, from_bytes_to_i32, from_bytes_to_i64, 
from_bytes_to_i128};
+use crate::basic::{BoundaryOrder, Type as PhysicalType};
+use crate::errors::{ParquetError, Result};
+use crate::parquet_thrift::{
+    ElementType, FieldType, ReadThrift, ThriftCompactInputProtocol, 
ThriftSliceInputProtocol,
+    validate_list_type,
+};
+use arrow_array::types::{
+    Date32Type, Date64Type, Decimal32Type, Decimal64Type, Decimal128Type, 
Decimal256Type, Int8Type,
+    Int16Type, Time32MillisecondType, Time32SecondType, Time64MicrosecondType,
+    Time64NanosecondType, TimestampMicrosecondType, TimestampMillisecondType,
+    TimestampNanosecondType, TimestampSecondType, UInt8Type, UInt16Type, 
UInt32Type, UInt64Type,
+};
+use arrow_array::{
+    Array, ArrayRef, BinaryArray, BinaryViewArray, BooleanArray, 
Decimal32Array, Decimal64Array,
+    Decimal128Array, Decimal256Array, FixedSizeBinaryArray, Float16Array, 
Float32Array,
+    Float64Array, Int32Array, Int64Array, LargeBinaryArray, LargeStringArray, 
StringArray,
+    StringViewArray, UInt64Array, new_null_array,
+};
+use arrow_buffer::{
+    BooleanBuffer, BooleanBufferBuilder, NullBuffer, OffsetBuffer, 
ScalarBuffer, i256,
+};
+use arrow_schema::{DataType, TimeUnit};
+use std::sync::Arc;
+
+/// The min or max values of every page, kept in the form Parquet stores them 
in
+/// (its "physical type"), before they are turned into the Arrow type the
+/// caller asked for.
+#[derive(Debug)]
+enum PhysicalValues {
+    Boolean(Vec<bool>),
+    Int32(Vec<i32>),
+    Int64(Vec<i64>),
+    Float(Vec<f32>),
+    Double(Vec<f64>),
+    /// Used for both `BYTE_ARRAY` and `FIXED_LEN_BYTE_ARRAY`. Fixed length
+    /// values are kept here too, because a stored min or max may have been
+    /// shortened and so may not have the expected length.
+    Bytes {
+        offsets: Vec<i32>,
+        values: Vec<u8>,
+        /// `true` for `FIXED_LEN_BYTE_ARRAY`
+        fixed_len: bool,
+    },
+    /// `INT96` values are only counted, not kept. No Arrow type is read
+    /// from them, so their mins and maxes always come out as nulls. They are
+    /// still checked for length, as the older decoder does.
+    Int96(Vec<()>),
+    /// Decimals stored as bytes (in either kind of byte column), turned
+    /// into numbers while reading, with their precision and scale. This skips
+    /// keeping a copy of the bytes and converting them afterwards, which is
+    /// much faster.
+    Decimal32(Vec<i32>, u8, i8),
+    Decimal64(Vec<i64>, u8, i8),
+    Decimal128(Vec<i128>, u8, i8),
+    Decimal256(Vec<i256>, u8, i8),
+}
+
+impl PhysicalValues {
+    /// `data_type` is the Arrow type the caller wants. It only matters for
+    /// decimals stored as bytes, which are turned into numbers right away.
+    fn new(physical_type: PhysicalType, data_type: &DataType, capacity: usize) 
-> Self {

Review Comment:
   I guess we could simplify this with some macros but this would it only make 
it harder to review.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to