deniskuzZ commented on code in PR #6793:
URL: https://github.com/apache/hive/pull/6793#discussion_r4154456235


##########
llap-server/src/java/org/apache/hadoop/hive/llap/io/api/impl/LlapInputFormat.java:
##########
@@ -123,19 +125,15 @@ public RecordReader<NullWritable, VectorizedRowBatch> 
getRecordReader(
           cvp, executor, sourceInputFormat, sourceSerDe, reporter, daemonConf);
       if (rr == null) {
         // Reader-specific incompatibility like SMB or schema evolution.
+        if (sourceInputFormat instanceof LlapCacheOnlyInputFormatInterface) {
+          LlapProxy.getIo().initCacheOnlyInputFormat(sourceInputFormat);
+        }
         return sourceInputFormat.getRecordReader(split, job, reporter);
       }
-      // For non-vectorized operator case, wrap the reader if possible.
-      RecordReader<NullWritable, VectorizedRowBatch> result = rr;
-      if (!Utilities.getIsVectorized(job)) {
-        result = null;
-        if (HiveConf.getBoolVar(job, ConfVars.LLAP_IO_ROW_WRAPPER_ENABLED)) {
-          result = wrapLlapReader(tableIncludedCols, rr, split);
-        }
-        if (result == null) {
-          // Cannot wrap a reader for non-vectorized pipeline.
-          return sourceInputFormat.getRecordReader(split, job, reporter);
-        }
+      RecordReader<NullWritable, VectorizedRowBatch> result =

Review Comment:
   wrapOrFallback never falls back!
   
   what if we refactor
   ````
   public RecordReader<NullWritable, VectorizedRowBatch> getRecordReader(       
 
   
       boolean vectorized = Utilities.getIsVectorized(job);                     
                                                                                
                                           
       if (!vectorized && !canWrapToRows(job)) {                                
                                                                                
                                           
         // The pipeline wants rows and no batch-to-row wrapper exists for this 
format; decide here,                                                            
                                           
         // before a reader (and its read pipeline, footer fetch, counters) is 
built for nothing.                                                              
                                            
         return fallback(split, job, reporter);                                 
                                                                                
                                           
       }
   ....
   LlapRecordReader rr = LlapRecordReader.create(job, fileSplit, 
tableIncludedCols, hostName,                                                    
                                                    
             cvp, executor, sourceInputFormat, sourceSerDe, reporter, 
daemonConf);                                                                    
                                                     
         if (rr == null) {                                                      
                                                                                
                                           
           // Reader-specific incompatibility: SMB, schema evolution, or the 
producer declined the split.                                                    
                                              
           return fallback(split, job, reporter);                               
                                                                                
                                           
         }                                                                      
                                                                                
                                           
         RecordReader<NullWritable, VectorizedRowBatch> result =                
                                                                                
                                           
             vectorized ? rr : wrapToRows(rr, tableIncludedCols, split);      
   ....
   }
   
     /** The non-vectorized pipeline can take LLAP batches only through the 
source format's row wrapper. */                                                 
                                               
     private boolean canWrapToRows(JobConf job) {                               
                                                                                
                                           
       return HiveConf.getBoolVar(job, ConfVars.LLAP_IO_ROW_WRAPPER_ENABLED)    
                                                                                
                                           
           && sourceInputFormat instanceof BatchToRowInputFormat;               
                                                                                
                                           
     }
   
     /** Only called when {@link #canWrapToRows} held, so the cast is safe. */  
                                                                                
                                           
     private RecordReader<NullWritable, VectorizedRowBatch> wrapToRows(         
                                                                                
                                           
         LlapRecordReader rr, List<Integer> includedCols, InputSplit split) 
throws IOException {                                                            
                                               
       LlapIoImpl.LOG.info("Using batch-to-row converter for split: " + split); 
                                                                                
                                           
       return bogusCast(((BatchToRowInputFormat) sourceInputFormat).getWrapper( 
                                                                                
                                           
           rr, rr.getVectorizedRowBatchCtx(), includedCols));                   
                                                                                
                                           
     }
   
     /**                                                                        
                                                                                
                                           
      * Serves the split with the source format. A cache-only format still gets 
the LLAP caches                                                                 
                                           
      * injected, so a declined split reads through the same cache entries as 
the native path.                                                                
                                             
      */                                                                        
                                                                                
                                           
     private RecordReader<NullWritable, VectorizedRowBatch> fallback(           
                                                                                
                                           
         InputSplit split, JobConf job, Reporter reporter) throws IOException { 
                                                                                
                                           
       if (sourceInputFormat instanceof LlapCacheOnlyInputFormatInterface) {    
                                                                                
                                           
         HiveInputFormat.injectLlapCaches(sourceInputFormat, LlapProxy.getIo(), 
job);                                                                           
                                           
       }                                                                        
                                                                                
                                           
       return sourceInputFormat.getRecordReader(split, job, reporter);          
                                                                                
                                           
     }  
   ````



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to