zeroshade commented on code in PR #38367:
URL: https://github.com/apache/arrow/pull/38367#discussion_r1376434496
##########
go/parquet/internal/encoding/boolean_decoder.go:
##########
@@ -109,3 +114,76 @@ func (dec *PlainBooleanDecoder) DecodeSpaced(out []bool,
nullCount int, validBit
}
return dec.Decode(out)
}
+
+type RleBooleanDecoder struct {
+ decoder
+
+ rleDec *utils.RleDecoder
+}
+
+func (RleBooleanDecoder) Type() parquet.Type {
+ return parquet.Types.Boolean
+}
+
+func (dec *RleBooleanDecoder) SetData(nvals int, data []byte) error {
+ dec.nvals = nvals
+
+ if len(data) < 4 {
+ return fmt.Errorf("invalid length - %d (corrupt data page?)",
len(data))
+ }
+
+ // load the first 4 bytes in little-endian which indicates the length
+ nbytes := binary.LittleEndian.Uint32(data[:4])
+ if nbytes > uint32(len(data)-4) {
+ return fmt.Errorf("received invalid number of bytes - %d
(corrupt data page?)", nbytes)
+ }
+
+ dec.data = data[4:]
+ if dec.rleDec == nil {
+ dec.rleDec = utils.NewRleDecoder(bytes.NewReader(dec.data), 1)
+ } else {
+ dec.rleDec.Reset(bytes.NewReader(dec.data), 1)
+ }
+ return nil
+}
+
+func (dec *RleBooleanDecoder) Decode(out []bool) (int, error) {
+ max := shared_utils.MinInt(len(out), dec.nvals)
+
+ var (
+ buf [1024]uint64
+ n = max
+ )
+
+ for n > 0 {
+ batch := shared_utils.MinInt(len(buf), n)
+ decoded := dec.rleDec.GetBatch(buf[:batch])
+ if decoded != batch {
+ return max - n, io.ErrUnexpectedEOF
Review Comment:
Hmm, you have a point though I think that would be a greater issue to think
about attempting to fix.
For DataPageV2Header it contains the number of nulls so we can easily handle
changing `dec.nvals` to the correct number, but for DataPageV1 you have a point
that *technically* this might be slightly off or is at least a bug waiting to
happen that should be addressed.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]