He-Pin commented on code in PR #1816:
URL: https://github.com/apache/pekko-connectors/pull/1816#discussion_r3746396454


##########
google-cloud-bigquery-storage/src/main/scala/org/apache/pekko/stream/connectors/googlecloud/bigquery/storage/impl/ArrowSource.scala:
##########
@@ -32,18 +32,24 @@ import scala.jdk.CollectionConverters._
 
 object ArrowSource {
 
-  def readRecordsMerged(client: BigQueryReadClient, readSession: ReadSession): 
Source[List[BigQueryRecord], NotUsed] =
+  def readRecordsMerged(client: BigQueryReadClient,
+      readSession: ReadSession,
+      allocatorBytes: Long): Source[List[BigQueryRecord], NotUsed] =
     readMerged(client, readSession)
-      .map(a => new 
SimpleRowReader(readSession.schema.arrowSchema.get).read(a))
+      .map { a =>
+        val reader = new SimpleRowReader(readSession.schema.arrowSchema.get, 
allocatorBytes)
+        try reader.read(a)
+        finally reader.close()
+      }
 
   def readMerged(client: BigQueryReadClient, session: ReadSession): 
Source[ArrowRecordBatch, NotUsed] =
-    read(client, session)
-      .reduce((a, b) => a.merge(b))
+    read(client, session).reduce((a, b) => a.merge(b))
 
-  def readRecords(client: BigQueryReadClient, session: ReadSession): 
Seq[Source[BigQueryRecord, NotUsed]] =
+  def readRecords(client: BigQueryReadClient, session: ReadSession,
+      allocatorBytes: Long): Seq[Source[BigQueryRecord, NotUsed]] =
     read(client, session)
       .map { a =>
-        a.map(new SimpleRowReader(session.schema.arrowSchema.get).read(_))
+        a.map(new SimpleRowReader(session.schema.arrowSchema.get, 
allocatorBytes).read(_))

Review Comment:
   The `SimpleRowReader` created here is never closed — each batch allocates a 
fresh `RootAllocator` that is never released, leaking native memory over time.
   
   `readRecordsMerged` now correctly wraps the reader in `try/finally 
reader.close()`, but this path does not. Consider mirroring the same pattern:
   
   ```scala
   a.map { batch =>
     val reader = new SimpleRowReader(session.schema.arrowSchema.get, 
allocatorBytes)
     try reader.read(batch)
     finally reader.close()
   }.mapConcat(c => c)
   ```



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to