This is an automated email from the ASF dual-hosted git repository.
JingsongLi pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/paimon.git
The following commit(s) were added to refs/heads/master by this push:
new 645ce6f8de [core] Add regression test for rewrite_file_index over an
arity-shrinking column drop (#10092)
645ce6f8de is described below
commit 645ce6f8de0053a05ad2d746d9deed1ee23b6b47
Author: YangJie <[email protected]>
AuthorDate: Fri Sep 25 10:20:26 2026 -0400
[core] Add regression test for rewrite_file_index over an arity-shrinking
column drop (#10092)
---
.../paimon/index/FileIndexProcessorTest.java | 67 ++++++++++++++++++++++
1 file changed, 67 insertions(+)
diff --git
a/paimon-core/src/test/java/org/apache/paimon/index/FileIndexProcessorTest.java
b/paimon-core/src/test/java/org/apache/paimon/index/FileIndexProcessorTest.java
index 19a5c899e5..bfdf0f45b2 100644
---
a/paimon-core/src/test/java/org/apache/paimon/index/FileIndexProcessorTest.java
+++
b/paimon-core/src/test/java/org/apache/paimon/index/FileIndexProcessorTest.java
@@ -250,4 +250,71 @@ public class FileIndexProcessorTest {
}
}
}
+
+ @Test
+ public void testProcessAfterDroppingColumnBeforeIndexedColumn() throws
Exception {
+ // The file is written at schema 0 [k, a, v] with a bloom filter on v.
Dropping "a"
+ // shrinks the current schema to [k, v] while the file keeps v at file
position 2; the
+ // rewrite must still build the index for v (a file-schema position
past the current
+ // arity must not throw or index the wrong column).
+ LocalFileIO fileIO = LocalFileIO.create();
+ Path warehouse = new Path(tempDir.toString());
+ Map<String, String> options = new HashMap<>();
+ options.put(CoreOptions.BUCKET.key(), "1");
+ options.put(CoreOptions.FILE_FORMAT.key(), "parquet");
+ options.put(CoreOptions.FILE_INDEX + ".bloom-filter.columns", "v");
+ options.put(CoreOptions.FILE_INDEX_IN_MANIFEST_THRESHOLD.key(), "0 B");
+ RowType rowType =
+ RowType.of(
+ new DataType[] {DataTypes.INT(), DataTypes.STRING(),
DataTypes.INT()},
+ new String[] {"k", "a", "v"});
+ Identifier identifier = Identifier.create("mydb", "t");
+ FileStoreTable table;
+ try (FileSystemCatalog catalog = new FileSystemCatalog(fileIO,
warehouse)) {
+ catalog.createDatabase("mydb", false);
+ catalog.createTable(
+ identifier,
+ new Schema(
+ rowType.getFields(),
+ Collections.emptyList(),
+ Collections.singletonList("k"),
+ options,
+ ""),
+ false);
+ table = (FileStoreTable) catalog.getTable(identifier);
+ }
+
+ String commitUser = UUID.randomUUID().toString();
+ try (TableWriteImpl<?> write = table.newWrite(commitUser);
+ TableCommitImpl commit = table.newCommit(commitUser)) {
+ write.write(GenericRow.of(1, BinaryString.fromString("x"), 10));
+ write.write(GenericRow.of(2, BinaryString.fromString("y"), 20));
+ commit.commit(1, write.prepareCommit(false, 1));
+ }
+
+ table.schemaManager().commitChanges(SchemaChange.dropColumn("a"));
+ table = table.copyWithLatestSchema();
+
+ List<ManifestEntry> entries = table.store().newScan().plan().files();
+ assertThat(entries).isNotEmpty();
+ ManifestEntry entry = entries.get(0);
+ assertThat(entry.file().schemaId()).isEqualTo(0L);
+
+ FileIndexProcessor processor = new FileIndexProcessor(table);
+ DataFileMeta processed = processor.process(entry.partition(),
entry.bucket(), entry);
+
+ String indexFile =
+ processed.extraFiles().stream()
+ .filter(name ->
name.endsWith(DataFilePathFactory.INDEX_PATH_SUFFIX))
+ .findFirst()
+ .orElseThrow(() -> new AssertionError("no file index
was written"));
+ Path indexPath =
+ new Path(
+
table.store().pathFactory().bucketPath(entry.partition(), entry.bucket()),
+ indexFile);
+ try (FileIndexFormat.Reader reader =
+ FileIndexFormat.createReader(fileIO.newInputStream(indexPath),
rowType)) {
+ assertThat(reader.readAll().keySet()).containsExactly("v");
+ }
+ }
}