This is an automated email from the ASF dual-hosted git repository.

jenniferdai pushed a commit to branch orc
in repository https://gitbox.apache.org/repos/asf/incubator-pinot.git


The following commit(s) were added to refs/heads/orc by this push:
     new 91767a1  Adding ORC Reader Test
91767a1 is described below

commit 91767a15e7f02bc7277dc8acd144bf490414ece4
Author: Jennifer Dai <[email protected]>
AuthorDate: Wed Mar 20 16:23:51 2019 -0700

    Adding ORC Reader Test
---
 .../pinot/orc/data/readers/ORCRecordReader.java    |  10 +-
 .../orc/data/readers/ORCRecordReaderTest.java      | 115 +++++++++++++++++++++
 2 files changed, 121 insertions(+), 4 deletions(-)

diff --git 
a/pinot-orc/src/main/java/org/apache/pinot/orc/data/readers/ORCRecordReader.java
 
b/pinot-orc/src/main/java/org/apache/pinot/orc/data/readers/ORCRecordReader.java
index b42262a..02dac23 100644
--- 
a/pinot-orc/src/main/java/org/apache/pinot/orc/data/readers/ORCRecordReader.java
+++ 
b/pinot-orc/src/main/java/org/apache/pinot/orc/data/readers/ORCRecordReader.java
@@ -64,13 +64,15 @@ public class ORCRecordReader implements RecordReader {
   org.apache.orc.RecordReader _recordReader;
   VectorizedRowBatch _reusableVectorizedRowBatch;
 
+  public static final String LOCAL_FS_PREFIX = "file://";
+
   private static final Logger LOGGER = 
LoggerFactory.getLogger(ORCRecordReader.class);
 
   private void init(String inputPath, Schema schema) {
     Configuration conf = new Configuration();
     LOGGER.info("Creating segment for {}", inputPath);
     try {
-      Path orcReaderPath = new Path("file://" + inputPath);
+      Path orcReaderPath = new Path(LOCAL_FS_PREFIX + inputPath);
       LOGGER.info("orc reader path is {}", orcReaderPath);
       _reader = OrcFile.createReader(orcReaderPath, 
OrcFile.readerOptions(conf));
       _orcSchema = _reader.getSchema();
@@ -119,7 +121,7 @@ public class ORCRecordReader implements RecordReader {
     return reuse;
   }
 
-  private void fillGenericRow(GenericRow genericRow, VectorizedRowBatch 
rowBatch) throws IOException {
+  private void fillGenericRow(GenericRow genericRow, VectorizedRowBatch 
rowBatch) {
     // ORC's TypeDescription is the equivalent of a schema. The way we will 
support ORC in Pinot
     // will be to get the top level struct that contains all our fields and 
look through its
     // children to determine the fields in our schemas.
@@ -127,7 +129,7 @@ public class ORCRecordReader implements RecordReader {
       for (int i = 0; i < _orcSchema.getChildren().size(); i++) {
         // Get current column in schema
         TypeDescription currColumn = _orcSchema.getChildren().get(i);
-        String currColumnName = currColumn.getFieldNames().get(0);
+        String currColumnName = _orcSchema.getFieldNames().get(i);
         if (!_pinotSchema.getColumnNames().contains(currColumnName)) {
           LOGGER.warn("Skipping column {} because it is not in pinot schema", 
currColumnName);
           continue;
@@ -135,7 +137,7 @@ public class ORCRecordReader implements RecordReader {
         int currColRowIndex = currColumn.getId();
         ColumnVector vector = rowBatch.cols[currColRowIndex];
         // Previous value set to null, not used except to save allocation 
memory in OrcMapredRecordReader
-        WritableComparable writableComparable = 
OrcMapredRecordReader.nextValue(vector, currColRowIndex, _orcSchema, null);
+        WritableComparable writableComparable = 
OrcMapredRecordReader.nextValue(vector, currColRowIndex, currColumn, null);
         genericRow.putField(currColumnName, getBaseObject(writableComparable));
       }
     } else {
diff --git 
a/pinot-orc/src/test/java/org/apache/pinot/orc/data/readers/ORCRecordReaderTest.java
 
b/pinot-orc/src/test/java/org/apache/pinot/orc/data/readers/ORCRecordReaderTest.java
new file mode 100644
index 0000000..6bea742
--- /dev/null
+++ 
b/pinot-orc/src/test/java/org/apache/pinot/orc/data/readers/ORCRecordReaderTest.java
@@ -0,0 +1,115 @@
+package org.apache.pinot.orc.data.readers;
+
+/**
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements.  See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership.  The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License.  You may obtain a copy of the License at
+ *
+ *   http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied.  See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+import java.io.File;
+import java.io.IOException;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
+import java.util.List;
+import org.apache.commons.io.FileUtils;
+import org.apache.hadoop.conf.Configuration;
+import org.apache.hadoop.fs.Path;
+import org.apache.hadoop.hive.ql.exec.vector.BytesColumnVector;
+import org.apache.hadoop.hive.ql.exec.vector.LongColumnVector;
+import org.apache.hadoop.hive.ql.exec.vector.VectorizedRowBatch;
+import org.apache.orc.OrcFile;
+import org.apache.orc.TypeDescription;
+import org.apache.orc.Writer;
+import org.apache.pinot.common.data.DimensionFieldSpec;
+import org.apache.pinot.common.data.FieldSpec;
+import org.apache.pinot.common.data.Schema;
+import org.apache.pinot.core.data.GenericRow;
+import org.apache.pinot.core.indexsegment.generator.SegmentGeneratorConfig;
+import org.testng.Assert;
+import org.testng.annotations.AfterClass;
+import org.testng.annotations.BeforeClass;
+import org.testng.annotations.Test;
+
+
+public class ORCRecordReaderTest {
+  private static final File TEMP_DIR = FileUtils.getTempDirectory();
+  private static final File ORC_FILE = new File(TEMP_DIR.getAbsolutePath(), 
"my-file.orc");
+
+  @BeforeClass
+  public void setUp()
+      throws Exception {
+    FileUtils.deleteQuietly(TEMP_DIR);
+    TypeDescription schema =
+        TypeDescription.fromString("struct<x:int,y:string>");
+    Writer writer = OrcFile.createWriter(new Path(ORC_FILE.getAbsolutePath()),
+        OrcFile.writerOptions(new Configuration())
+            .setSchema(schema));
+
+    VectorizedRowBatch batch = schema.createRowBatch();
+    LongColumnVector x = (LongColumnVector) batch.cols[0];
+    BytesColumnVector y = (BytesColumnVector) batch.cols[1];
+    for(int r=0; r < 5; ++r) {
+      int row = batch.size++;
+      x.vector[row] = r;
+      byte[] buffer = ("Last-" + (r * 3)).getBytes(StandardCharsets.UTF_8);
+      y.setRef(row, buffer, 0, buffer.length);
+      // If the batch is full, write it out and start over.
+      if (batch.size == batch.getMaxSize()) {
+        writer.addRowBatch(batch);
+        batch.reset();
+      }
+    }
+    if (batch.size != 0) {
+      writer.addRowBatch(batch);
+    }
+    writer.close();
+  }
+
+  @Test
+  public void testReadData()
+      throws IOException {
+
+    ORCRecordReader orcRecordReader = new ORCRecordReader();
+
+    SegmentGeneratorConfig segmentGeneratorConfig = new 
SegmentGeneratorConfig();
+    segmentGeneratorConfig.setInputFilePath(ORC_FILE.getAbsolutePath());
+    Schema schema = new Schema();
+    FieldSpec xFieldSpec = new DimensionFieldSpec("x", 
FieldSpec.DataType.LONG, true);
+    schema.addField(xFieldSpec);
+    FieldSpec yFieldSpec = new DimensionFieldSpec("y", 
FieldSpec.DataType.BYTES, true);
+    schema.addField(yFieldSpec);
+    segmentGeneratorConfig.setSchema(schema);
+    orcRecordReader.init(segmentGeneratorConfig);
+
+    List<GenericRow> genericRows = new ArrayList<>();
+    while (orcRecordReader.hasNext()) {
+      genericRows.add(orcRecordReader.next());
+    }
+    orcRecordReader.close();
+    Assert.assertEquals(genericRows.size(), 5, "Generic row size must be 5");
+
+    for (int i = 0; i < genericRows.size(); i++) {
+      Assert.assertEquals(genericRows.get(i).getValue("x"), i);
+      Assert.assertEquals(genericRows.get(i).getValue("y"), ("Last-" + (i * 
3)).getBytes(StandardCharsets.UTF_8));
+    }
+  }
+
+  @AfterClass
+  public void tearDown() {
+    FileUtils.deleteQuietly(TEMP_DIR);
+  }
+
+}


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to