This is an automated email from the ASF dual-hosted git repository.
jrmccluskey pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/beam.git
The following commit(s) were added to refs/heads/master by this push:
new 9dbf3e716a9 Expose Public API for Iceberg Side Input Cache (#40140)
9dbf3e716a9 is described below
commit 9dbf3e716a96d0b003708b1fab7eca1c4f59106c
Author: Jack McCluskey <[email protected]>
AuthorDate: Mon Sep 21 16:55:41 2026 -0400
Expose Public API for Iceberg Side Input Cache (#40140)
* Expose Public API for Iceberg Side Input Cache
* Update
sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergIO.java
Co-authored-by: Ahmed Abualsaud
<[email protected]>
* Streamline API, remove redundant checks
* Check polled table metric
* Evolve Iceberg schema mid-execution in streaming side input cache test
* Throw if sub-options are enabled without side input cache enabled
* test with partition spec evolution
---------
Co-authored-by: Ahmed Abualsaud
<[email protected]>
---
.../org/apache/beam/sdk/io/iceberg/IcebergIO.java | 125 ++++-
.../IcebergWriteSchemaTransformProvider.java | 57 ++
.../beam/sdk/io/iceberg/TableMetadataDriver.java | 2 +-
.../iceberg/IcebergIOSideInputTableCacheTest.java | 615 +++++++++++++++++++++
.../IcebergWriteSchemaTransformProviderTest.java | 185 +++++++
.../sdk/io/iceberg/TableMetadataDriverTest.java | 4 +-
6 files changed, 979 insertions(+), 9 deletions(-)
diff --git
a/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergIO.java
b/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergIO.java
index 080c31ff912..3aeddb41303 100644
---
a/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergIO.java
+++
b/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergIO.java
@@ -29,8 +29,10 @@ import
org.apache.beam.sdk.io.iceberg.cdc.IncrementalChangelogSource;
import org.apache.beam.sdk.options.StreamingOptions;
import org.apache.beam.sdk.schemas.Schema;
import org.apache.beam.sdk.transforms.PTransform;
+import org.apache.beam.sdk.transforms.display.DisplayData;
import org.apache.beam.sdk.values.PBegin;
import org.apache.beam.sdk.values.PCollection;
+import org.apache.beam.sdk.values.PCollectionView;
import org.apache.beam.sdk.values.Row;
import
org.apache.beam.vendor.guava.v32_1_2_jre.com.google.common.base.Preconditions;
import
org.apache.beam.vendor.guava.v32_1_2_jre.com.google.common.base.Predicates;
@@ -390,6 +392,7 @@ public class IcebergIO {
.setCatalogConfig(catalog)
.setDistributionMode(DistributionMode.NONE)
.setAutoSharding(false)
+ .setUsingSideInputTableCache(false)
.build();
}
@@ -416,6 +419,14 @@ public class IcebergIO {
abstract @Nullable List<String> getSortFields();
+ abstract boolean getUsingSideInputTableCache();
+
+ abstract @Nullable Integer getMaximumCacheSize();
+
+ abstract @Nullable Duration getTableRefreshInterval();
+
+ abstract @Nullable Integer getPollingBuckets();
+
abstract Builder toBuilder();
@AutoValue.Builder
@@ -440,6 +451,14 @@ public class IcebergIO {
abstract Builder setSortFields(List<String> sortFields);
+ abstract Builder setUsingSideInputTableCache(boolean
usingSideInputTableCache);
+
+ abstract Builder setMaximumCacheSize(@Nullable Integer maximumCacheSize);
+
+ abstract Builder setTableRefreshInterval(@Nullable Duration
refreshInterval);
+
+ abstract Builder setPollingBuckets(@Nullable Integer pollingBuckets);
+
abstract WriteRows build();
}
@@ -476,11 +495,11 @@ public class IcebergIO {
* Defines distribution of write data. Supported distributions:
*
* <ol>
- * <li>{@link DistributionMode.NONE}: don't shuffle rows (default)
- * <li>{@link DistributionMode.HASH}: shuffle rows by partition key
before writing data
+ * <li>{@link DistributionMode#NONE}: don't shuffle rows (default)
+ * <li>{@link DistributionMode#HASH}: shuffle rows by partition key
before writing data
* </ol>
*
- * {@link DistributionMode.RANGE} is not supported yet
+ * {@link DistributionMode#RANGE} is not supported yet
*/
public WriteRows withDistributionMode(DistributionMode mode) {
return toBuilder().setDistributionMode(mode).build();
@@ -524,6 +543,68 @@ public class IcebergIO {
return toBuilder().setSortFields(sortFields).build();
}
+ /**
+ * Enables expirable side-input caching of Iceberg table metadata across
workers.
+ *
+ * <p>When enabled, a driver transform periodically polls the Iceberg
catalog and broadcasts
+ * lightweight table specifications as a side input. Workers construct
in-memory {@link Table}
+ * representations without issuing remote catalog RPCs, drastically
reducing catalog load.
+ */
+ public WriteRows withSideInputTableCache() {
+ return toBuilder().setUsingSideInputTableCache(true).build();
+ }
+
+ /**
+ * Sets the maximum number of distinct table metadata specifications to
broadcast in the
+ * side-input cache. Any tables exceeding this limit fall back to
worker-local catalog loading.
+ *
+ * <p><b>Note:</b> This option is only supported for bounded (batch)
pipelines. Calling this on
+ * an unbounded streaming pipeline will throw an exception at pipeline
construction.
+ */
+ public WriteRows withMaximumCacheSize(int maximumCacheSize) {
+ Preconditions.checkArgument(maximumCacheSize > 0, "maximumCacheSize must
be greater than 0");
+ return toBuilder().setMaximumCacheSize(maximumCacheSize).build();
+ }
+
+ /**
+ * Sets the interval at which table metadata is refreshed from the Iceberg
catalog.
+ *
+ * <p>Applicable for unbounded streaming pipelines. Defaults to 5 minutes.
+ */
+ public WriteRows withTableRefreshInterval(Duration refreshInterval) {
+ Preconditions.checkNotNull(refreshInterval, "refreshInterval must not be
null");
+ Preconditions.checkArgument(
+ refreshInterval.isLongerThan(Duration.ZERO), "refreshInterval must
be greater than 0");
+ return toBuilder().setTableRefreshInterval(refreshInterval).build();
+ }
+
+ /**
+ * Sets the number of parallel buckets/workers used to query the Iceberg
catalog during
+ * refreshes. Defaults to 1 to serialize catalog queries and protect
catalogs from connection
+ * spikes.
+ */
+ public WriteRows withPollingBuckets(int pollingBuckets) {
+ Preconditions.checkArgument(pollingBuckets > 0, "pollingBuckets must be
greater than 0");
+ return toBuilder().setPollingBuckets(pollingBuckets).build();
+ }
+
+ @Override
+ public void populateDisplayData(DisplayData.Builder builder) {
+ super.populateDisplayData(builder);
+ builder.add(
+ DisplayData.item("usingSideInputTableCache",
getUsingSideInputTableCache())
+ .withLabel("Using Side-Input Table Cache"));
+ builder.addIfNotNull(
+ DisplayData.item("maximumCacheSize", getMaximumCacheSize())
+ .withLabel("Maximum Cache Size"));
+ builder.addIfNotNull(
+ DisplayData.item("tableRefreshInterval", getTableRefreshInterval())
+ .withLabel("Table Refresh Interval"));
+ builder.addIfNotNull(
+ DisplayData.item("pollingBuckets", getPollingBuckets())
+ .withLabel("Catalog Polling Buckets"));
+ }
+
@Override
public IcebergWriteResult expand(PCollection<Row> input) {
List<?> allToArgs = Arrays.asList(getTableIdentifier(),
getDynamicDestinations());
@@ -549,6 +630,35 @@ public class IcebergIO {
"Must only provide direct write limit for unbounded pipelines.");
}
+ boolean hasSideInputOptions =
+ getMaximumCacheSize() != null
+ || getTableRefreshInterval() != null
+ || getPollingBuckets() != null;
+ Preconditions.checkArgument(
+ getUsingSideInputTableCache() || !hasSideInputOptions,
+ "Cannot specify side-input cache sub-options (maximumCacheSize, "
+ + "tableRefreshInterval, pollingBuckets) without enabling
side-input table cache via withSideInputTableCache().");
+
+ PCollectionView<Map<String, SerializableTableSpec>> metadataView = null;
+ if (getUsingSideInputTableCache()) {
+ TableMetadataDriver.Builder driverBuilder =
+ TableMetadataDriver.builder()
+ .setCatalogConfig(getCatalogConfig())
+ .setDynamicDestinations(destinations);
+
+ if (getMaximumCacheSize() != null) {
+ driverBuilder.setMaximumCacheSize(getMaximumCacheSize());
+ }
+ if (getTableRefreshInterval() != null) {
+ driverBuilder.setRefreshInterval(getTableRefreshInterval());
+ }
+ if (getPollingBuckets() != null) {
+ driverBuilder.setPollingBuckets(getPollingBuckets());
+ }
+
+ metadataView = input.apply("GenerateTableMetadataView",
driverBuilder.build().asView());
+ }
+
switch (getDistributionMode()) {
case NONE:
Preconditions.checkArgument(
@@ -563,12 +673,14 @@ public class IcebergIO {
destinations,
getTriggeringFrequency(),
getDirectWriteByteLimit(),
- getWriteProperties()));
+ getWriteProperties(),
+ metadataView));
case HASH:
return input
.apply(
"AssignDestinationAndPartition",
- new AssignDestinationsAndPartitions(destinations,
getCatalogConfig()))
+ new AssignDestinationsAndPartitions(
+ destinations, getCatalogConfig(), metadataView))
.apply(
"Write Rows to Partitions",
new WriteToPartitions(
@@ -576,7 +688,8 @@ public class IcebergIO {
destinations,
getTriggeringFrequency(),
getAutoSharding(),
- getWriteProperties()));
+ getWriteProperties(),
+ metadataView));
default:
throw new UnsupportedOperationException(
"Unsupported distribution mode: " + getDistributionMode());
diff --git
a/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergWriteSchemaTransformProvider.java
b/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergWriteSchemaTransformProvider.java
index 0ae9d5fb0ec..e5a81e41a8b 100644
---
a/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergWriteSchemaTransformProvider.java
+++
b/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/IcebergWriteSchemaTransformProvider.java
@@ -163,6 +163,23 @@ public class IcebergWriteSchemaTransformProvider
+ "'write.parquet.bloom-filter-enabled.column.<col>').")
public abstract @Nullable Map<String, String> getWriteProperties();
+ @SchemaFieldDescription(
+ "Enables expirable side-input caching of Iceberg table metadata across
workers to reduce catalog load.")
+ public abstract @Nullable Boolean getUsingSideInputTableCache();
+
+ @SchemaFieldDescription(
+ "For a streaming pipeline, sets the interval in seconds at which table
metadata is refreshed from the catalog.")
+ public abstract @Nullable Integer getTableRefreshIntervalSeconds();
+
+ @SchemaFieldDescription(
+ "For a batch pipeline, sets the maximum number of table metadata specs
to cache in memory. "
+ + "Tables exceeding this limit fall back to worker-local catalog
loading.")
+ public abstract @Nullable Integer getMaximumCacheSize();
+
+ @SchemaFieldDescription(
+ "Sets the number of parallel buckets/workers used to query the Iceberg
catalog during refreshes. Defaults to 1.")
+ public abstract @Nullable Integer getPollingBuckets();
+
@AutoValue.Builder
public abstract static class Builder {
public abstract Builder setTable(String table);
@@ -195,6 +212,14 @@ public class IcebergWriteSchemaTransformProvider
public abstract Builder setWriteProperties(Map<String, String>
writeProperties);
+ public abstract Builder setUsingSideInputTableCache(Boolean
usingSideInputTableCache);
+
+ public abstract Builder setTableRefreshIntervalSeconds(Integer
tableRefreshIntervalSeconds);
+
+ public abstract Builder setMaximumCacheSize(Integer maximumCacheSize);
+
+ public abstract Builder setPollingBuckets(Integer pollingBuckets);
+
public abstract Configuration build();
}
@@ -291,6 +316,38 @@ public class IcebergWriteSchemaTransformProvider
writeTransform = writeTransform.withWriteProperties(writeProperties);
}
+ boolean hasSideInputOptions =
+ configuration.getTableRefreshIntervalSeconds() != null
+ || configuration.getMaximumCacheSize() != null
+ || configuration.getPollingBuckets() != null;
+
+ if (!Boolean.TRUE.equals(configuration.getUsingSideInputTableCache())
+ && hasSideInputOptions) {
+ throw new IllegalArgumentException(
+ "Cannot specify side-input cache sub-options
(table_refresh_interval_seconds, "
+ + "maximum_cache_size, polling_buckets) without explicitly
setting using_side_input_table_cache to true.");
+ }
+
+ boolean enableSideInputCache =
+ Boolean.TRUE.equals(configuration.getUsingSideInputTableCache());
+
+ if (enableSideInputCache) {
+ writeTransform = writeTransform.withSideInputTableCache();
+ @Nullable Integer refreshSec =
configuration.getTableRefreshIntervalSeconds();
+ if (refreshSec != null) {
+ writeTransform =
+
writeTransform.withTableRefreshInterval(Duration.standardSeconds(refreshSec));
+ }
+ @Nullable Integer maxCacheSize = configuration.getMaximumCacheSize();
+ if (maxCacheSize != null) {
+ writeTransform = writeTransform.withMaximumCacheSize(maxCacheSize);
+ }
+ @Nullable Integer pollingBuckets = configuration.getPollingBuckets();
+ if (pollingBuckets != null) {
+ writeTransform = writeTransform.withPollingBuckets(pollingBuckets);
+ }
+ }
+
// TODO: support dynamic destinations
IcebergWriteResult result = rows.apply(writeTransform);
diff --git
a/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/TableMetadataDriver.java
b/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/TableMetadataDriver.java
index cf5d6310a1d..679463ab5db 100644
---
a/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/TableMetadataDriver.java
+++
b/sdks/java/io/iceberg/src/main/java/org/apache/beam/sdk/io/iceberg/TableMetadataDriver.java
@@ -276,7 +276,7 @@ public abstract class TableMetadataDriver
Integer maxCacheSize = getMaximumCacheSize();
if (maxCacheSize != null) {
if (isStreaming) {
- throw new UnsupportedOperationException(
+ throw new IllegalArgumentException(
"maximumCacheSize is currently not supported for unbounded
streaming pipelines.");
}
cachedTableIds = distinctTableIds.apply("CapCacheSize",
Sample.any(maxCacheSize));
diff --git
a/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/IcebergIOSideInputTableCacheTest.java
b/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/IcebergIOSideInputTableCacheTest.java
new file mode 100644
index 00000000000..246f8410ec3
--- /dev/null
+++
b/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/IcebergIOSideInputTableCacheTest.java
@@ -0,0 +1,615 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.beam.sdk.io.iceberg;
+
+import static java.util.Arrays.asList;
+import static org.hamcrest.MatcherAssert.assertThat;
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertNotNull;
+import static org.junit.Assert.assertThrows;
+import static org.junit.Assert.assertTrue;
+
+import java.io.Serializable;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.UUID;
+import org.apache.beam.sdk.Pipeline;
+import org.apache.beam.sdk.PipelineResult;
+import org.apache.beam.sdk.metrics.MetricNameFilter;
+import org.apache.beam.sdk.metrics.MetricQueryResults;
+import org.apache.beam.sdk.metrics.MetricResult;
+import org.apache.beam.sdk.metrics.MetricsFilter;
+import org.apache.beam.sdk.schemas.Schema;
+import org.apache.beam.sdk.testing.TestPipeline;
+import org.apache.beam.sdk.testing.TestStream;
+import org.apache.beam.sdk.transforms.Create;
+import org.apache.beam.sdk.transforms.DoFn;
+import org.apache.beam.sdk.transforms.ParDo;
+import org.apache.beam.sdk.transforms.display.DisplayData;
+import org.apache.beam.sdk.values.PCollection;
+import org.apache.beam.sdk.values.Row;
+import org.apache.beam.sdk.values.ValueInSingleWindow;
+import
org.apache.beam.vendor.guava.v32_1_2_jre.com.google.common.collect.ImmutableList;
+import
org.apache.beam.vendor.guava.v32_1_2_jre.com.google.common.collect.ImmutableMap;
+import org.apache.iceberg.CatalogUtil;
+import org.apache.iceberg.DataFile;
+import org.apache.iceberg.DistributionMode;
+import org.apache.iceberg.FileFormat;
+import org.apache.iceberg.PartitionSpec;
+import org.apache.iceberg.Snapshot;
+import org.apache.iceberg.SnapshotChanges;
+import org.apache.iceberg.Table;
+import org.apache.iceberg.catalog.TableIdentifier;
+import org.apache.iceberg.data.IcebergGenerics;
+import org.apache.iceberg.data.Record;
+import org.apache.iceberg.types.Types;
+import org.hamcrest.Matchers;
+import org.joda.time.Duration;
+import org.junit.Before;
+import org.junit.ClassRule;
+import org.junit.Rule;
+import org.junit.Test;
+import org.junit.rules.TemporaryFolder;
+import org.junit.runner.RunWith;
+import org.junit.runners.Parameterized;
+
+/** Tests for {@link IcebergIO.WriteRows} with side-input table caching
enabled. */
+@RunWith(Parameterized.class)
+public class IcebergIOSideInputTableCacheTest implements Serializable {
+
+ private static final String NONE = "none";
+ private static final String HASH = "hash";
+ private static final String HASH_WITH_AUTOSHARDING = "hashWithAutoSharding";
+
+ @Parameterized.Parameters(name = "distributionMode={0}")
+ public static Iterable<Object[]> data() {
+ return asList(new Object[][] {{NONE}, {HASH}, {HASH_WITH_AUTOSHARDING}});
+ }
+
+ @Parameterized.Parameter(0)
+ public String distributionMode;
+
+ @ClassRule public static final TemporaryFolder TEMPORARY_FOLDER = new
TemporaryFolder();
+
+ @Rule
+ public transient TestDataWarehouse warehouse =
+ new TestDataWarehouse(TEMPORARY_FOLDER, "default_side_input");
+
+ @Rule public transient TestPipeline testPipeline = TestPipeline.create();
+
+ private IcebergCatalogConfig catalogConfig;
+
+ @Before
+ public void setUp() {
+ Map<String, String> catalogProps =
+ ImmutableMap.<String, String>builder()
+ .put("type", CatalogUtil.ICEBERG_CATALOG_TYPE_HADOOP)
+ .put("warehouse", warehouse.location)
+ .build();
+
+ catalogConfig =
+ IcebergCatalogConfig.builder()
+ .setCatalogName("hadoop")
+ .setCatalogProperties(catalogProps)
+ .build();
+ TableCache.invalidateAll();
+ }
+
+ private IcebergIO.WriteRows applyDistribution(IcebergIO.WriteRows write) {
+ if (distributionMode.contains(HASH)) {
+ write = write.withDistributionMode(DistributionMode.HASH);
+ }
+ if (distributionMode.equals(HASH_WITH_AUTOSHARDING)) {
+ write = write.withAutosharding();
+ }
+ return write;
+ }
+
+ private long getTablesPolledCount(PipelineResult result) {
+ MetricQueryResults metrics =
+ result
+ .metrics()
+ .queryMetrics(
+ MetricsFilter.builder()
+ .addNameFilter(
+ MetricNameFilter.named(TableMetadataDriver.class,
"tablesPolled"))
+ .build());
+ long total = 0;
+ for (MetricResult<Long> counter : metrics.getCounters()) {
+ Long val = counter.getCommitted() != null ? counter.getCommitted() :
counter.getAttempted();
+ if (val != null) {
+ total += val;
+ }
+ }
+ return total;
+ }
+
+ @Test
+ public void testBatchSingleTableWithSideInputCache() throws Exception {
+ TableIdentifier tableId =
+ TableIdentifier.of(
+ "default_side_input",
+ "single_table_" + Long.toString(UUID.randomUUID().hashCode(), 16));
+
+ warehouse.createTable(tableId, TestFixtures.SCHEMA);
+
+ PCollection<Row> input =
+ testPipeline
+ .apply("CreateRecords",
Create.of(TestFixtures.asRows(TestFixtures.FILE1SNAPSHOT1)))
+
.setRowSchema(IcebergUtils.icebergSchemaToBeamSchema(TestFixtures.SCHEMA));
+
+ IcebergIO.WriteRows write =
+ IcebergIO.writeRows(catalogConfig)
+ .to(tableId)
+ .withSideInputTableCache()
+ .withPollingBuckets(1);
+
+ input.apply("WriteToTable", applyDistribution(write));
+ PipelineResult result = testPipeline.run();
+ result.waitUntilFinish();
+
+ assertEquals(1L, getTablesPolledCount(result));
+
+ Table table = warehouse.loadTable(tableId);
+ List<Record> writtenRecords =
ImmutableList.copyOf(IcebergGenerics.read(table).build());
+ assertThat(writtenRecords,
Matchers.containsInAnyOrder(TestFixtures.FILE1SNAPSHOT1.toArray()));
+ assertNotNull(table.currentSnapshot());
+ }
+
+ @Test
+ public void testBatchDynamicDestinationsWithSideInputCache() throws
Exception {
+ final String salt = Long.toString(UUID.randomUUID().hashCode(), 16);
+ final TableIdentifier table1Id = TableIdentifier.of("default_side_input",
"dyn_table1_" + salt);
+ final TableIdentifier table2Id = TableIdentifier.of("default_side_input",
"dyn_table2_" + salt);
+ final TableIdentifier table3Id = TableIdentifier.of("default_side_input",
"dyn_table3_" + salt);
+
+ warehouse.createTable(table1Id, TestFixtures.SCHEMA);
+ warehouse.createTable(table2Id, TestFixtures.SCHEMA);
+ warehouse.createTable(table3Id, TestFixtures.SCHEMA);
+
+ Schema beamSchema =
IcebergUtils.icebergSchemaToBeamSchema(TestFixtures.SCHEMA);
+ Schema inputSchema =
+
Schema.builder().addStringField("dest").addFields(beamSchema.getFields()).build();
+
+ List<Row> rows = new ArrayList<>();
+ for (Record record : TestFixtures.FILE1SNAPSHOT1) {
+ Row beamRow = IcebergUtils.icebergRecordToBeamRow(beamSchema, record);
+ rows.add(
+ Row.withSchema(inputSchema)
+ .addValue(IcebergUtils.tableIdentifierToString(table1Id))
+ .addValues(beamRow.getValues())
+ .build());
+ }
+ for (Record record : TestFixtures.FILE2SNAPSHOT1) {
+ Row beamRow = IcebergUtils.icebergRecordToBeamRow(beamSchema, record);
+ rows.add(
+ Row.withSchema(inputSchema)
+ .addValue(IcebergUtils.tableIdentifierToString(table2Id))
+ .addValues(beamRow.getValues())
+ .build());
+ }
+ for (Record record : TestFixtures.FILE3SNAPSHOT1) {
+ Row beamRow = IcebergUtils.icebergRecordToBeamRow(beamSchema, record);
+ rows.add(
+ Row.withSchema(inputSchema)
+ .addValue(IcebergUtils.tableIdentifierToString(table3Id))
+ .addValues(beamRow.getValues())
+ .build());
+ }
+
+ DynamicDestinations dynamicDestinations =
+ new DynamicDestinations() {
+ @Override
+ public Schema getDataSchema() {
+ return beamSchema;
+ }
+
+ @Override
+ public Row getData(Row element) {
+ Row.Builder builder = Row.withSchema(beamSchema);
+ for (Schema.Field field : beamSchema.getFields()) {
+ builder.addValue(element.getValue(field.getName()));
+ }
+ return builder.build();
+ }
+
+ @Override
+ public IcebergDestination instantiateDestination(String destination)
{
+ return IcebergDestination.builder()
+
.setTableIdentifier(IcebergUtils.parseTableIdentifier(destination))
+ .setFileFormat(FileFormat.PARQUET)
+ .build();
+ }
+
+ @Override
+ public String getTableStringIdentifier(ValueInSingleWindow<Row>
element) {
+ return element.getValue().getString("dest");
+ }
+ };
+
+ PCollection<Row> input =
+ testPipeline.apply("CreateRows",
Create.of(rows)).setRowSchema(inputSchema);
+
+ IcebergIO.WriteRows write =
+ IcebergIO.writeRows(catalogConfig)
+ .to(dynamicDestinations)
+ .withSideInputTableCache()
+ .withPollingBuckets(1);
+
+ input.apply("WriteDynamic", applyDistribution(write));
+ PipelineResult result = testPipeline.run();
+ result.waitUntilFinish();
+
+ assertEquals(3L, getTablesPolledCount(result));
+
+ Table table1 = warehouse.loadTable(table1Id);
+ Table table2 = warehouse.loadTable(table2Id);
+ Table table3 = warehouse.loadTable(table3Id);
+
+ List<Record> records1 =
ImmutableList.copyOf(IcebergGenerics.read(table1).build());
+ List<Record> records2 =
ImmutableList.copyOf(IcebergGenerics.read(table2).build());
+ List<Record> records3 =
ImmutableList.copyOf(IcebergGenerics.read(table3).build());
+
+ assertThat(records1,
Matchers.containsInAnyOrder(TestFixtures.FILE1SNAPSHOT1.toArray()));
+ assertThat(records2,
Matchers.containsInAnyOrder(TestFixtures.FILE2SNAPSHOT1.toArray()));
+ assertThat(records3,
Matchers.containsInAnyOrder(TestFixtures.FILE3SNAPSHOT1.toArray()));
+ }
+
+ @Test
+ public void testBatchMaximumCacheSizeSamplingAndFallback() throws Exception {
+ final String salt = Long.toString(UUID.randomUUID().hashCode(), 16);
+ final TableIdentifier table1Id = TableIdentifier.of("default_side_input",
"sample_t1_" + salt);
+ final TableIdentifier table2Id = TableIdentifier.of("default_side_input",
"sample_t2_" + salt);
+ final TableIdentifier table3Id = TableIdentifier.of("default_side_input",
"sample_t3_" + salt);
+ final TableIdentifier table4Id = TableIdentifier.of("default_side_input",
"sample_t4_" + salt);
+
+ warehouse.createTable(table1Id, TestFixtures.SCHEMA);
+ warehouse.createTable(table2Id, TestFixtures.SCHEMA);
+ warehouse.createTable(table3Id, TestFixtures.SCHEMA);
+ warehouse.createTable(table4Id, TestFixtures.SCHEMA);
+
+ Schema beamSchema =
IcebergUtils.icebergSchemaToBeamSchema(TestFixtures.SCHEMA);
+ Schema inputSchema =
+
Schema.builder().addStringField("dest").addFields(beamSchema.getFields()).build();
+
+ List<Row> rows = new ArrayList<>();
+ for (Record record : TestFixtures.FILE1SNAPSHOT1) {
+ Row beamRow = IcebergUtils.icebergRecordToBeamRow(beamSchema, record);
+ rows.add(
+ Row.withSchema(inputSchema)
+ .addValue(IcebergUtils.tableIdentifierToString(table1Id))
+ .addValues(beamRow.getValues())
+ .build());
+ rows.add(
+ Row.withSchema(inputSchema)
+ .addValue(IcebergUtils.tableIdentifierToString(table2Id))
+ .addValues(beamRow.getValues())
+ .build());
+ rows.add(
+ Row.withSchema(inputSchema)
+ .addValue(IcebergUtils.tableIdentifierToString(table3Id))
+ .addValues(beamRow.getValues())
+ .build());
+ rows.add(
+ Row.withSchema(inputSchema)
+ .addValue(IcebergUtils.tableIdentifierToString(table4Id))
+ .addValues(beamRow.getValues())
+ .build());
+ }
+
+ DynamicDestinations dynamicDestinations =
+ new DynamicDestinations() {
+ @Override
+ public Schema getDataSchema() {
+ return beamSchema;
+ }
+
+ @Override
+ public Row getData(Row element) {
+ Row.Builder builder = Row.withSchema(beamSchema);
+ for (Schema.Field field : beamSchema.getFields()) {
+ builder.addValue(element.getValue(field.getName()));
+ }
+ return builder.build();
+ }
+
+ @Override
+ public IcebergDestination instantiateDestination(String destination)
{
+ return IcebergDestination.builder()
+
.setTableIdentifier(IcebergUtils.parseTableIdentifier(destination))
+ .setFileFormat(FileFormat.PARQUET)
+ .build();
+ }
+
+ @Override
+ public String getTableStringIdentifier(ValueInSingleWindow<Row>
element) {
+ return element.getValue().getString("dest");
+ }
+ };
+
+ PCollection<Row> input =
+ testPipeline.apply("CreateSampleRows",
Create.of(rows)).setRowSchema(inputSchema);
+
+ // Maximum cache size of 2 forces 2 tables to be cached and 2 to fall back
to TableCache
+ IcebergIO.WriteRows write =
+ IcebergIO.writeRows(catalogConfig)
+ .to(dynamicDestinations)
+ .withSideInputTableCache()
+ .withMaximumCacheSize(2)
+ .withPollingBuckets(1);
+
+ input.apply("WriteWithSampleCap", applyDistribution(write));
+ PipelineResult result = testPipeline.run();
+ result.waitUntilFinish();
+
+ assertEquals(2L, getTablesPolledCount(result));
+
+ // Verify all 4 tables received data successfully
+ for (TableIdentifier tId : asList(table1Id, table2Id, table3Id, table4Id))
{
+ Table table = warehouse.loadTable(tId);
+ List<Record> written =
ImmutableList.copyOf(IcebergGenerics.read(table).build());
+ assertEquals(TestFixtures.FILE1SNAPSHOT1.size(), written.size());
+ }
+ }
+
+ @Test
+ public void testStreamingWithSideInputCache() throws Exception {
+ TableIdentifier tableId =
+ TableIdentifier.of(
+ "default_side_input",
+ "streaming_table_" + Long.toString(UUID.randomUUID().hashCode(),
16));
+
+ warehouse.createTable(tableId, TestFixtures.SCHEMA);
+
+ Schema beamSchema =
IcebergUtils.icebergSchemaToBeamSchema(TestFixtures.SCHEMA);
+ List<Row> rows1 = TestFixtures.asRows(TestFixtures.FILE1SNAPSHOT1);
+ List<Row> rows2 = TestFixtures.asRows(TestFixtures.FILE2SNAPSHOT1);
+
+ TestStream<Row> testStream =
+ TestStream.create(beamSchema)
+ .addElements(rows1.get(0), rows1.subList(1,
rows1.size()).toArray(new Row[0]))
+ .advanceProcessingTime(Duration.standardSeconds(2))
+ .addElements(rows2.get(0), rows2.subList(1,
rows2.size()).toArray(new Row[0]))
+ .advanceProcessingTime(Duration.standardSeconds(2))
+ .advanceWatermarkToInfinity();
+
+ PCollection<Row> input = testPipeline.apply("StreamingInput", testStream);
+
+ IcebergIO.WriteRows write =
+ IcebergIO.writeRows(catalogConfig)
+ .to(tableId)
+ .withSideInputTableCache()
+ .withTriggeringFrequency(Duration.standardSeconds(1))
+ .withTableRefreshInterval(Duration.standardSeconds(2))
+ .withPollingBuckets(1);
+
+ input.apply("StreamingWrite", applyDistribution(write));
+ PipelineResult result = testPipeline.run();
+ result.waitUntilFinish();
+
+ assertThat(getTablesPolledCount(result),
Matchers.greaterThanOrEqualTo(1L));
+
+ Table table = warehouse.loadTable(tableId);
+ List<Record> written =
ImmutableList.copyOf(IcebergGenerics.read(table).build());
+ List<Record> expected = new ArrayList<>();
+ expected.addAll(TestFixtures.FILE1SNAPSHOT1);
+ expected.addAll(TestFixtures.FILE2SNAPSHOT1);
+ assertThat(written, Matchers.containsInAnyOrder(expected.toArray()));
+ assertNotNull(table.currentSnapshot());
+ }
+
+ @Test
+ public void testStreamingSpecEvolutionWithoutPipelineRestart() throws
Exception {
+ TableIdentifier tableId =
+ TableIdentifier.of(
+ "default_side_input", "spec_evolve_" +
Long.toString(UUID.randomUUID().hashCode(), 16));
+
+ // Initial table schema: id (long), name (string), city (string)
+ org.apache.iceberg.Schema icebergSchema =
+ new org.apache.iceberg.Schema(
+ Types.NestedField.required(1, "id", Types.LongType.get()),
+ Types.NestedField.optional(2, "name", Types.StringType.get()),
+ Types.NestedField.optional(3, "city", Types.StringType.get()));
+
+ // Create table unpartitioned with format-version 2 to support spec
evolution
+ Table realTable =
+ warehouse.createTable(
+ tableId,
+ icebergSchema,
+ PartitionSpec.unpartitioned(),
+ ImmutableMap.of("format-version", "2"));
+
+ Schema beamSchema =
+ Schema.builder()
+ .addInt64Field("id")
+ .addNullableStringField("name")
+ .addNullableStringField("city")
+ .build();
+
+ Row row1 = Row.withSchema(beamSchema).addValues(1L, "alice", "New
York").build();
+ Row row2 = Row.withSchema(beamSchema).addValues(2L, "bob", "San
Francisco").build();
+
+ TestStream<Row> testStream =
+ TestStream.create(beamSchema)
+ .addElements(row1)
+ .advanceProcessingTime(Duration.standardSeconds(2))
+ .addElements(row2)
+ .advanceProcessingTime(Duration.standardSeconds(2))
+ .advanceWatermarkToInfinity();
+
+ PCollection<Row> input =
+ testPipeline
+ .apply("StreamingEvolvedInput", testStream)
+ .apply(
+ "EvolveSpecMidExecution",
+ ParDo.of(new EvolveSpecMidExecutionDoFn(catalogConfig,
tableId.toString())))
+ .setRowSchema(beamSchema);
+
+ IcebergIO.WriteRows write =
+ IcebergIO.writeRows(catalogConfig)
+ .to(tableId)
+ .withSideInputTableCache()
+ .withTriggeringFrequency(Duration.standardSeconds(1))
+ .withTableRefreshInterval(Duration.standardSeconds(1))
+ .withPollingBuckets(1);
+
+ input.apply("StreamingWriteEvolved", write);
+ PipelineResult result = testPipeline.run();
+ result.waitUntilFinish();
+
+ assertThat(getTablesPolledCount(result),
Matchers.greaterThanOrEqualTo(1L));
+
+ realTable.refresh();
+ List<DataFile> addedFiles =
+
ImmutableList.copyOf(SnapshotChanges.builderFor(realTable).build().addedDataFiles());
+ if (addedFiles.size() < 2) {
+ List<DataFile> allAddedFiles = new ArrayList<>();
+ for (Snapshot s : realTable.snapshots()) {
+ for (DataFile df :
+
SnapshotChanges.builderFor(realTable).snapshot(s).build().addedDataFiles()) {
+ allAddedFiles.add(df);
+ }
+ }
+ addedFiles = allAddedFiles;
+ }
+
+ assertEquals(2, addedFiles.size());
+
+ DataFile firstFile = addedFiles.get(0);
+ DataFile secondFile = addedFiles.get(1);
+ DataFile unpartitionedFile;
+ DataFile partitionedFile;
+ if (realTable.specs().get(firstFile.specId()).isUnpartitioned()) {
+ unpartitionedFile = firstFile;
+ partitionedFile = secondFile;
+ } else {
+ unpartitionedFile = secondFile;
+ partitionedFile = firstFile;
+ }
+
+ PartitionSpec spec1 = realTable.specs().get(unpartitionedFile.specId());
+ assertNotNull(spec1);
+ assertTrue("First DataFile must have unpartitioned spec",
spec1.isUnpartitioned());
+
+ PartitionSpec spec2 = realTable.specs().get(partitionedFile.specId());
+ assertNotNull(spec2);
+ assertEquals(1, spec2.fields().size());
+ assertEquals("city", spec2.fields().get(0).name());
+
+ List<Record> records =
ImmutableList.copyOf(IcebergGenerics.read(realTable).build());
+ assertEquals(2, records.size());
+ }
+
+ @Test
+ public void testPreconditionsAndValidation() {
+ TableIdentifier tableId = TableIdentifier.of("default_side_input",
"validation_table");
+ IcebergIO.WriteRows write = IcebergIO.writeRows(catalogConfig).to(tableId);
+
+ assertThrows(IllegalArgumentException.class, () ->
write.withMaximumCacheSize(0));
+ assertThrows(IllegalArgumentException.class, () ->
write.withMaximumCacheSize(-1));
+
+ assertThrows(
+ IllegalArgumentException.class, () ->
write.withTableRefreshInterval(Duration.ZERO));
+
+ assertThrows(IllegalArgumentException.class, () ->
write.withPollingBuckets(0));
+ assertThrows(IllegalArgumentException.class, () ->
write.withPollingBuckets(-1));
+
+ // Unbounded streaming pipeline with maximumCacheSize must fail at expand
+ Pipeline p = Pipeline.create();
+ Schema schema = Schema.builder().addInt64Field("id").build();
+ TestStream<Row> testStream =
+ TestStream.create(schema)
+ .addElements(Row.withSchema(schema).addValues(1L).build())
+ .advanceWatermarkToInfinity();
+
+ PCollection<Row> streamInput = p.apply("StreamForValidation", testStream);
+
+ // Sub-options specified without withSideInputTableCache() must fail at
expand
+ IcebergIO.WriteRows writeWithMaxCacheOnly =
+ IcebergIO.writeRows(catalogConfig).to(tableId).withMaximumCacheSize(5);
+ assertThrows(IllegalArgumentException.class, () ->
streamInput.apply(writeWithMaxCacheOnly));
+
+ IcebergIO.WriteRows writeWithRefreshIntervalOnly =
+ IcebergIO.writeRows(catalogConfig)
+ .to(tableId)
+ .withTableRefreshInterval(Duration.standardMinutes(1));
+ assertThrows(
+ IllegalArgumentException.class, () ->
streamInput.apply(writeWithRefreshIntervalOnly));
+
+ IcebergIO.WriteRows writeWithPollingBucketsOnly =
+ IcebergIO.writeRows(catalogConfig).to(tableId).withPollingBuckets(2);
+ assertThrows(
+ IllegalArgumentException.class, () ->
streamInput.apply(writeWithPollingBucketsOnly));
+
+ IcebergIO.WriteRows streamWrite =
+ IcebergIO.writeRows(catalogConfig)
+ .to(tableId)
+ .withSideInputTableCache()
+ .withMaximumCacheSize(5);
+
+ assertThrows(IllegalArgumentException.class, () ->
streamInput.apply(streamWrite));
+ }
+
+ @Test
+ public void testDisplayData() {
+ TableIdentifier tableId = TableIdentifier.of("default_side_input",
"display_data_table");
+ IcebergIO.WriteRows write =
+ IcebergIO.writeRows(catalogConfig)
+ .to(tableId)
+ .withSideInputTableCache()
+ .withMaximumCacheSize(100)
+ .withTableRefreshInterval(Duration.standardMinutes(10))
+ .withPollingBuckets(3);
+
+ DisplayData displayData = DisplayData.from(write);
+ Map<String, String> items = new HashMap<>();
+ for (DisplayData.Item item : displayData.items()) {
+ items.put(item.getKey(), item.getValue() != null ?
item.getValue().toString() : "");
+ }
+
+ assertEquals("true", items.get("usingSideInputTableCache"));
+ assertEquals("100", items.get("maximumCacheSize"));
+ assertEquals("600000", items.get("tableRefreshInterval"));
+ assertEquals("3", items.get("pollingBuckets"));
+ }
+
+ private static class EvolveSpecMidExecutionDoFn extends DoFn<Row, Row> {
+ private final IcebergCatalogConfig catalogConfig;
+ private final String tableIdString;
+
+ EvolveSpecMidExecutionDoFn(IcebergCatalogConfig catalogConfig, String
tableIdString) {
+ this.catalogConfig = catalogConfig;
+ this.tableIdString = tableIdString;
+ }
+
+ @ProcessElement
+ public void processElement(@Element Row row, OutputReceiver<Row> out) {
+ Long id = row.getInt64("id");
+ if (id != null && id == 2L) {
+ Table table =
+
catalogConfig.catalog().loadTable(IcebergUtils.parseTableIdentifier(tableIdString));
+ if (table.spec().isUnpartitioned()) {
+ table.updateSpec().addField("city").commit();
+ }
+ }
+ out.output(row);
+ }
+ }
+}
diff --git
a/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/IcebergWriteSchemaTransformProviderTest.java
b/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/IcebergWriteSchemaTransformProviderTest.java
index bfb762f8e89..c4ac1e22cbf 100644
---
a/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/IcebergWriteSchemaTransformProviderTest.java
+++
b/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/IcebergWriteSchemaTransformProviderTest.java
@@ -25,6 +25,8 @@ import static
org.apache.iceberg.util.DateTimeUtil.dateFromDays;
import static org.apache.iceberg.util.DateTimeUtil.timestampFromMicros;
import static org.hamcrest.MatcherAssert.assertThat;
import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertNotNull;
+import static org.junit.Assert.assertThrows;
import static org.junit.Assert.assertTrue;
import static org.junit.Assume.assumeTrue;
@@ -117,6 +119,26 @@ public class IcebergWriteSchemaTransformProviderTest {
new IcebergWriteSchemaTransformProvider().from(transformConfigRow);
}
+ @Test
+ public void testBuildTransformWithRowAndSideInputCache() {
+ Map<String, String> properties = new HashMap<>();
+ properties.put("type", CatalogUtil.ICEBERG_CATALOG_TYPE_HADOOP);
+ properties.put("warehouse", "test_location");
+
+ Row transformConfigRow =
+ Row.withSchema(new
IcebergWriteSchemaTransformProvider().configurationSchema())
+ .withFieldValue("table", "test_table_identifier")
+ .withFieldValue("catalog_name", "test-name")
+ .withFieldValue("catalog_properties", properties)
+ .withFieldValue("using_side_input_table_cache", true)
+ .withFieldValue("table_refresh_interval_seconds", 60)
+ .withFieldValue("maximum_cache_size", 100)
+ .withFieldValue("polling_buckets", 2)
+ .build();
+
+ new IcebergWriteSchemaTransformProvider().from(transformConfigRow);
+ }
+
@Test
public void testSimpleAppend() {
String identifier = "default.table_" +
Long.toString(UUID.randomUUID().hashCode(), 16);
@@ -159,6 +181,54 @@ public class IcebergWriteSchemaTransformProviderTest {
assertThat(writtenRecords,
Matchers.containsInAnyOrder(TestFixtures.FILE1SNAPSHOT1.toArray()));
}
+ @Test
+ public void testSimpleAppendWithSideInputCache() {
+ String identifier =
+ "default.table_side_input_" +
Long.toString(UUID.randomUUID().hashCode(), 16);
+
+ Map<String, String> properties = new HashMap<>();
+ properties.put("type", CatalogUtil.ICEBERG_CATALOG_TYPE_HADOOP);
+ properties.put("warehouse", warehouse.location);
+
+ Configuration config =
+ Configuration.builder()
+ .setTable(identifier)
+ .setCatalogName("name")
+ .setCatalogProperties(properties)
+ .setDistributionMode(distributionMode.name())
+ .setUsingSideInputTableCache(true)
+ .setTableRefreshIntervalSeconds(60)
+ .setPollingBuckets(1)
+ .build();
+
+ PCollectionRowTuple input =
+ PCollectionRowTuple.of(
+ INPUT_TAG,
+ testPipeline
+ .apply(
+ "Records To Add",
Create.of(TestFixtures.asRows(TestFixtures.FILE1SNAPSHOT1)))
+
.setRowSchema(IcebergUtils.icebergSchemaToBeamSchema(TestFixtures.SCHEMA)));
+
+ PCollection<Row> result =
+ input
+ .apply(
+ "Append To Table With Cache",
+ new IcebergWriteSchemaTransformProvider().from(config))
+ .get(SNAPSHOTS_TAG);
+
+ PAssert.that(result)
+ .satisfies(new VerifyOutputs(Collections.singletonList(identifier),
"append"));
+
+ testPipeline.run().waitUntilFinish();
+
+ TableIdentifier tableId = TableIdentifier.parse(identifier);
+ Table table = warehouse.loadTable(tableId);
+
+ List<Record> writtenRecords =
ImmutableList.copyOf(IcebergGenerics.read(table).build());
+
+ assertThat(writtenRecords,
Matchers.containsInAnyOrder(TestFixtures.FILE1SNAPSHOT1.toArray()));
+ }
+
@Test
public void testWriteUsingManagedTransform() {
String identifier = "default.table_" +
Long.toString(UUID.randomUUID().hashCode(), 16);
@@ -194,6 +264,121 @@ public class IcebergWriteSchemaTransformProviderTest {
assertThat(writtenRecords,
Matchers.containsInAnyOrder(TestFixtures.FILE1SNAPSHOT1.toArray()));
}
+ @Test
+ public void testWriteUsingManagedTransformWithSideInputCache() {
+ String identifier =
+ "default.table_managed_cache_" +
Long.toString(UUID.randomUUID().hashCode(), 16);
+
+ String yamlConfig =
+ String.format(
+ "table: %s\n"
+ + "catalog_name: test-name\n"
+ + "distribution_mode: %s\n"
+ + "using_side_input_table_cache: true\n"
+ + "table_refresh_interval_seconds: 60\n"
+ + "maximum_cache_size: 50\n"
+ + "polling_buckets: 1\n"
+ + "catalog_properties: \n"
+ + " type: %s\n"
+ + " warehouse: %s",
+ identifier,
+ distributionMode.name(),
+ CatalogUtil.ICEBERG_CATALOG_TYPE_HADOOP,
+ warehouse.location);
+ Map<String, Object> configMap = new Yaml().load(yamlConfig);
+
+ PCollection<Row> inputRows =
+ testPipeline
+ .apply("Records To Add",
Create.of(TestFixtures.asRows(TestFixtures.FILE1SNAPSHOT1)))
+
.setRowSchema(IcebergUtils.icebergSchemaToBeamSchema(TestFixtures.SCHEMA));
+
+ Managed.ManagedTransform writeTransform =
Managed.write(Managed.ICEBERG).withConfig(configMap);
+ PCollectionRowTuple output = PCollectionRowTuple.of(INPUT_TAG,
inputRows).apply(writeTransform);
+
+ PAssert.that(output.get(SNAPSHOTS_TAG))
+ .satisfies(new VerifyOutputs(Collections.singletonList(identifier),
"append"));
+
+ testPipeline.run().waitUntilFinish();
+
+ Table table = warehouse.loadTable(TableIdentifier.parse(identifier));
+ List<Record> writtenRecords =
ImmutableList.copyOf(IcebergGenerics.read(table).build());
+
+ assertThat(writtenRecords,
Matchers.containsInAnyOrder(TestFixtures.FILE1SNAPSHOT1.toArray()));
+ }
+
+ @Test
+ public void testSideInputCacheSubOptionsRequireExplicitEnablement() {
+ IcebergWriteSchemaTransformProvider provider = new
IcebergWriteSchemaTransformProvider();
+ Pipeline p = Pipeline.create();
+ PCollectionRowTuple dummyInput =
+ PCollectionRowTuple.of(
+ INPUT_TAG,
+ p.apply("DummyInput",
Create.of(TestFixtures.asRows(TestFixtures.FILE1SNAPSHOT1)))
+
.setRowSchema(IcebergUtils.icebergSchemaToBeamSchema(TestFixtures.SCHEMA)));
+
+ // Setting sub-options when using_side_input_table_cache is not set (null)
must throw
+ // IllegalArgumentException
+ Configuration configWithMaxCacheSizeOnly =
+ Configuration.builder()
+ .setTable("default.table_max_cache")
+ .setCatalogName("name")
+ .setCatalogProperties(Collections.singletonMap("type", "hadoop"))
+ .setMaximumCacheSize(50)
+ .build();
+ assertThrows(
+ IllegalArgumentException.class,
+ () -> dummyInput.apply(provider.from(configWithMaxCacheSizeOnly)));
+
+ Configuration configWithRefreshIntervalOnly =
+ Configuration.builder()
+ .setTable("default.table_refresh")
+ .setCatalogName("name")
+ .setCatalogProperties(Collections.singletonMap("type", "hadoop"))
+ .setTableRefreshIntervalSeconds(60)
+ .build();
+ assertThrows(
+ IllegalArgumentException.class,
+ () -> dummyInput.apply(provider.from(configWithRefreshIntervalOnly)));
+
+ Configuration configWithPollingBucketsOnly =
+ Configuration.builder()
+ .setTable("default.table_buckets")
+ .setCatalogName("name")
+ .setCatalogProperties(Collections.singletonMap("type", "hadoop"))
+ .setPollingBuckets(2)
+ .build();
+ assertThrows(
+ IllegalArgumentException.class,
+ () -> dummyInput.apply(provider.from(configWithPollingBucketsOnly)));
+
+ // Setting using_side_input_table_cache to false while setting sub-options
must throw
+ // IllegalArgumentException
+ Configuration invalidConfigWithFalse =
+ Configuration.builder()
+ .setTable("default.table_invalid")
+ .setCatalogName("name")
+ .setCatalogProperties(Collections.singletonMap("type", "hadoop"))
+ .setUsingSideInputTableCache(false)
+ .setMaximumCacheSize(50)
+ .build();
+ assertThrows(
+ IllegalArgumentException.class,
+ () -> dummyInput.apply(provider.from(invalidConfigWithFalse)));
+
+ // Explicitly setting using_side_input_table_cache to true with
sub-options succeeds
+ Configuration validConfig =
+ Configuration.builder()
+ .setTable("default.table_valid")
+ .setCatalogName("name")
+ .setCatalogProperties(Collections.singletonMap("type", "hadoop"))
+ .setUsingSideInputTableCache(true)
+ .setMaximumCacheSize(50)
+ .setTableRefreshIntervalSeconds(60)
+ .setPollingBuckets(2)
+ .build();
+ assertNotNull(dummyInput.apply(provider.from(validConfig)));
+ }
+
/**
* @param operation if null, just perform a normal dynamic destination write
test; otherwise,
* performs a simple filter on the record before writing. Valid options
are "keep", "drop",
diff --git
a/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/TableMetadataDriverTest.java
b/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/TableMetadataDriverTest.java
index c46b155921b..cb627f15063 100644
---
a/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/TableMetadataDriverTest.java
+++
b/sdks/java/io/iceberg/src/test/java/org/apache/beam/sdk/io/iceberg/TableMetadataDriverTest.java
@@ -706,7 +706,7 @@ public class TableMetadataDriverTest implements
Serializable {
}
@Test
- public void
testMaximumCacheSizeInStreamingThrowsUnsupportedOperationException() {
+ public void testMaximumCacheSizeInStreamingThrowsIllegalArgumentException() {
pipeline.enableAbandonedNodeEnforcement(false);
Row row = Row.withSchema(BEAM_SCHEMA).addValues(1L, "v1",
"default.test_table").build();
TestStream<Row> stream =
@@ -718,7 +718,7 @@ public class TableMetadataDriverTest implements
Serializable {
PCollection<Row> input = pipeline.apply("StreamInput", stream);
assertThrows(
- UnsupportedOperationException.class,
+ IllegalArgumentException.class,
() ->
input.apply(
TableMetadataDriver.builder()