This is an automated email from the ASF dual-hosted git repository.

JingsongLi pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/paimon.git


The following commit(s) were added to refs/heads/master by this push:
     new 645ce6f8de [core] Add regression test for rewrite_file_index over an 
arity-shrinking column drop (#10092)
645ce6f8de is described below

commit 645ce6f8de0053a05ad2d746d9deed1ee23b6b47
Author: YangJie <[email protected]>
AuthorDate: Fri Sep 25 10:20:26 2026 -0400

    [core] Add regression test for rewrite_file_index over an arity-shrinking 
column drop (#10092)
---
 .../paimon/index/FileIndexProcessorTest.java       | 67 ++++++++++++++++++++++
 1 file changed, 67 insertions(+)

diff --git 
a/paimon-core/src/test/java/org/apache/paimon/index/FileIndexProcessorTest.java 
b/paimon-core/src/test/java/org/apache/paimon/index/FileIndexProcessorTest.java
index 19a5c899e5..bfdf0f45b2 100644
--- 
a/paimon-core/src/test/java/org/apache/paimon/index/FileIndexProcessorTest.java
+++ 
b/paimon-core/src/test/java/org/apache/paimon/index/FileIndexProcessorTest.java
@@ -250,4 +250,71 @@ public class FileIndexProcessorTest {
             }
         }
     }
+
+    @Test
+    public void testProcessAfterDroppingColumnBeforeIndexedColumn() throws 
Exception {
+        // The file is written at schema 0 [k, a, v] with a bloom filter on v. 
Dropping "a"
+        // shrinks the current schema to [k, v] while the file keeps v at file 
position 2; the
+        // rewrite must still build the index for v (a file-schema position 
past the current
+        // arity must not throw or index the wrong column).
+        LocalFileIO fileIO = LocalFileIO.create();
+        Path warehouse = new Path(tempDir.toString());
+        Map<String, String> options = new HashMap<>();
+        options.put(CoreOptions.BUCKET.key(), "1");
+        options.put(CoreOptions.FILE_FORMAT.key(), "parquet");
+        options.put(CoreOptions.FILE_INDEX + ".bloom-filter.columns", "v");
+        options.put(CoreOptions.FILE_INDEX_IN_MANIFEST_THRESHOLD.key(), "0 B");
+        RowType rowType =
+                RowType.of(
+                        new DataType[] {DataTypes.INT(), DataTypes.STRING(), 
DataTypes.INT()},
+                        new String[] {"k", "a", "v"});
+        Identifier identifier = Identifier.create("mydb", "t");
+        FileStoreTable table;
+        try (FileSystemCatalog catalog = new FileSystemCatalog(fileIO, 
warehouse)) {
+            catalog.createDatabase("mydb", false);
+            catalog.createTable(
+                    identifier,
+                    new Schema(
+                            rowType.getFields(),
+                            Collections.emptyList(),
+                            Collections.singletonList("k"),
+                            options,
+                            ""),
+                    false);
+            table = (FileStoreTable) catalog.getTable(identifier);
+        }
+
+        String commitUser = UUID.randomUUID().toString();
+        try (TableWriteImpl<?> write = table.newWrite(commitUser);
+                TableCommitImpl commit = table.newCommit(commitUser)) {
+            write.write(GenericRow.of(1, BinaryString.fromString("x"), 10));
+            write.write(GenericRow.of(2, BinaryString.fromString("y"), 20));
+            commit.commit(1, write.prepareCommit(false, 1));
+        }
+
+        table.schemaManager().commitChanges(SchemaChange.dropColumn("a"));
+        table = table.copyWithLatestSchema();
+
+        List<ManifestEntry> entries = table.store().newScan().plan().files();
+        assertThat(entries).isNotEmpty();
+        ManifestEntry entry = entries.get(0);
+        assertThat(entry.file().schemaId()).isEqualTo(0L);
+
+        FileIndexProcessor processor = new FileIndexProcessor(table);
+        DataFileMeta processed = processor.process(entry.partition(), 
entry.bucket(), entry);
+
+        String indexFile =
+                processed.extraFiles().stream()
+                        .filter(name -> 
name.endsWith(DataFilePathFactory.INDEX_PATH_SUFFIX))
+                        .findFirst()
+                        .orElseThrow(() -> new AssertionError("no file index 
was written"));
+        Path indexPath =
+                new Path(
+                        
table.store().pathFactory().bucketPath(entry.partition(), entry.bucket()),
+                        indexFile);
+        try (FileIndexFormat.Reader reader =
+                FileIndexFormat.createReader(fileIO.newInputStream(indexPath), 
rowType)) {
+            assertThat(reader.readAll().keySet()).containsExactly("v");
+        }
+    }
 }

Reply via email to