This is an automated email from the ASF dual-hosted git repository.
philo-he pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/gluten.git
The following commit(s) were added to refs/heads/main by this push:
new 8fa03c0fed [GLUTEN-12743][CI] Stabilize Delta predicate-pushdown
fixture (#13058)
8fa03c0fed is described below
commit 8fa03c0fedf6cd7305526d4dfa759b0264eb99b0
Author: Felipe Pessoto <[email protected]>
AuthorDate: Wed Sep 23 09:56:42 2026 -0700
[GLUTEN-12743][CI] Stabilize Delta predicate-pushdown fixture (#13058)
---
.../delta-spark-ut/apply-delta-test-patches.sh | 61 ++++++++++++++++++++++
.../util/delta-spark-ut/known-failures.txt | 22 +++++++-
2 files changed, 81 insertions(+), 2 deletions(-)
diff --git a/.github/workflows/util/delta-spark-ut/apply-delta-test-patches.sh
b/.github/workflows/util/delta-spark-ut/apply-delta-test-patches.sh
index 4fd9c874a6..ea1587bdce 100755
--- a/.github/workflows/util/delta-spark-ut/apply-delta-test-patches.sh
+++ b/.github/workflows/util/delta-spark-ut/apply-delta-test-patches.sh
@@ -135,6 +135,67 @@ git -C "$DELTA_DIR" --no-pager diff -- \
"spark/src/test/scala/org/apache/spark/sql/delta/DeltaParquetFileFormatSuite.scala"
|| true
echo "::endgroup::"
+echo "::group::Capping predicate-pushdown DV fixture row groups by row count"
+# DeletionVectorsWithPredicatePushdownSuite writes one 1,000,000-row Parquet
+# file and expects its 2 MiB block size to produce two row groups. The rows can
+# arrive at Velox in one Arrow batch, so the native writer cannot evaluate its
+# buffered-byte flush threshold until the whole batch is already in one group.
+# Cap this fixture at 500,000 rows per group so Arrow splits the batch while
+# retaining the native write path and the existing Hadoop block-size setting.
+DV_SUITE="$DELTA_DIR/spark/src/test/scala/org/apache/spark/sql/delta/deletionvectors/DeletionVectorsSuite.scala"
+if [ ! -f "$DV_SUITE" ]; then
+ echo "Expected file not found in Delta clone: $DV_SUITE" >&2
+ echo "The Delta directory layout for ref '${DELTA_REF}' may have changed."
>&2
+ exit 1
+fi
+if ! sed 's/^__BLANK_CONTEXT__$/ /' <<'PATCH' | git -C "$DELTA_DIR" apply -
+diff --git
a/spark/src/test/scala/org/apache/spark/sql/delta/deletionvectors/DeletionVectorsSuite.scala
b/spark/src/test/scala/org/apache/spark/sql/delta/deletionvectors/DeletionVectorsSuite.scala
+---
a/spark/src/test/scala/org/apache/spark/sql/delta/deletionvectors/DeletionVectorsSuite.scala
++++
b/spark/src/test/scala/org/apache/spark/sql/delta/deletionvectors/DeletionVectorsSuite.scala
+@@ -913,12 +913,14 @@ class DeletionVectorsWithPredicatePushdownSuite extends
DeletionVectorsSuite {
+ super.beforeAll()
+__BLANK_CONTEXT__
+ // 2MB rowgroups.
+ hadoopConf().set("parquet.block.size", (2 * 1024 * 1024).toString)
+__BLANK_CONTEXT__
+- spark.range(0, multiRowgroupTableRowsNum, 1, 1).toDF("id")
+- .write
+- .option(DeltaConfigs.ENABLE_DELETION_VECTORS_CREATION.key,
true.toString)
+- .format("delta")
+- .saveAsTable(multiRowgroupTable)
++ withSQLConf("spark.gluten.sql.native.parquet.write.blockRows" ->
"500000") {
++ spark.range(0, multiRowgroupTableRowsNum, 1, 1).toDF("id")
++ .write
++ .option(DeltaConfigs.ENABLE_DELETION_VECTORS_CREATION.key,
true.toString)
++ .format("delta")
++ .saveAsTable(multiRowgroupTable)
++ }
+__BLANK_CONTEXT__
+ val deltaLog = DeltaLog.forTable(spark,
TableIdentifier(multiRowgroupTable))
+ val files = deltaLog.update().allFiles.collect()
+PATCH
+then
+ echo "ERROR: predicate-pushdown DV fixture patch did not apply." >&2
+ echo "The patch expects the Delta v4.2.0 beforeAll fixture shape;" \
+ "ref '${DELTA_REF}' must remain source-compatible." >&2
+ exit 1
+fi
+DV_ROW_CAP_SCOPES=$(
+ grep -Fxc \
+ ' withSQLConf("spark.gluten.sql.native.parquet.write.blockRows" ->
"500000") {' \
+ "$DV_SUITE" || true
+)
+if [ "$DV_ROW_CAP_SCOPES" -ne 1 ]; then
+ echo "ERROR: expected exactly one predicate-pushdown DV row-count scope;" \
+ "found ${DV_ROW_CAP_SCOPES}." >&2
+ echo "DeletionVectorsSuite may have changed in Delta ref '${DELTA_REF}'." >&2
+ exit 1
+fi
+echo "Capped predicate-pushdown DV fixture row groups at 500,000 rows."
+git -C "$DELTA_DIR" --no-pager diff -- \
+
"spark/src/test/scala/org/apache/spark/sql/delta/deletionvectors/DeletionVectorsSuite.scala"
|| true
+echo "::endgroup::"
+
echo "::group::Force-failing memory-hog DeletionVectorsSuite 2B-row tests"
# Two DeletionVectorsSuite tests read from / delete from a 2-billion-row table.
# Under the Gluten Velox bundle they balloon the forked test JVM to ~13G of
diff --git a/.github/workflows/util/delta-spark-ut/known-failures.txt
b/.github/workflows/util/delta-spark-ut/known-failures.txt
index 2e3ede435b..faaec020ab 100644
--- a/.github/workflows/util/delta-spark-ut/known-failures.txt
+++ b/.github/workflows/util/delta-spark-ut/known-failures.txt
@@ -446,8 +446,26 @@
org.apache.spark.sql.delta.coordinatedcommits.CoordinatedCommitsSuite#Incomplete
org.apache.spark.sql.delta.deletionvectors.DeletionVectorsSuite#DELETE with
DVs with column mapping mode=id
org.apache.spark.sql.delta.deletionvectors.DeletionVectorsSuite#huge table:
delete a small number of rows from tables of 2B rows with DVs
org.apache.spark.sql.delta.deletionvectors.DeletionVectorsSuite#huge table:
read from tables of 2B rows with existing DV of many zeros
-org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#(It
is not a test it is a sbt.testing.SuiteSelector)
-org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#<suite
aborted>
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Multiple delete statements. vectorizedReaderEnabled: false
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Multiple delete statements. vectorizedReaderEnabled: true
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Multiple delete statements. vectorizedReaderEnabled: true
readColumnarBatchAsRows: true
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Scan with predicates - no deletes. vectorizedReaderEnabled: false
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Scan with predicates - no deletes. vectorizedReaderEnabled: true
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Scan with predicates - no deletes. vectorizedReaderEnabled: true
readColumnarBatchAsRows: true
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Scan with predicates. vectorizedReaderEnabled: false readColumnarBatchAsRows:
false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Scan with predicates. vectorizedReaderEnabled: true readColumnarBatchAsRows:
false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Scan with predicates. vectorizedReaderEnabled: true readColumnarBatchAsRows:
true
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single delete statement with multiple ids. vectorizedReaderEnabled: false
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single delete statement with multiple ids. vectorizedReaderEnabled: true
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single delete statement with multiple ids. vectorizedReaderEnabled: true
readColumnarBatchAsRows: true
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single deletion at the first row group. vectorizedReaderEnabled: false
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single deletion at the first row group. vectorizedReaderEnabled: true
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single deletion at the first row group. vectorizedReaderEnabled: true
readColumnarBatchAsRows: true
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single deletion at the second row group. vectorizedReaderEnabled: false
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single deletion at the second row group. vectorizedReaderEnabled: true
readColumnarBatchAsRows: false
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#PredicatePushdown:
Single deletion at the second row group. vectorizedReaderEnabled: true
readColumnarBatchAsRows: true
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#huge
table: delete a small number of rows from tables of 2B rows with DVs
+org.apache.spark.sql.delta.deletionvectors.DeletionVectorsWithPredicatePushdownSuite#huge
table: read from tables of 2B rows with existing DV of many zeros
org.apache.spark.sql.delta.generatedsuites.DeleteTempViewSQLNameBasedSuite#test
delete on temp view - nontrivial projection - Dataset TempView
org.apache.spark.sql.delta.generatedsuites.DeleteTempViewSQLNameBasedSuite#test
delete on temp view - nontrivial projection - SQL TempView
org.apache.spark.sql.delta.generatedsuites.DeleteTempViewSQLPathBasedCDCOnSuite#test
delete on temp view - nontrivial projection - Dataset TempView
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]