Akash3121 commented on code in PR #10121:
URL: https://github.com/apache/paimon/pull/10121#discussion_r4078994061
##########
paimon-format/src/test/java/org/apache/paimon/format/blob/BlobFileFormatTest.java:
##########
@@ -667,6 +667,63 @@ private void assertDuplicateMapBlobKeyRejected(
.hasMessage("Invalid MAP<X, BLOB> payload: duplicate key.");
}
+ @Test
+ public void testMapDescriptorReadsCoalesceMetadata() throws IOException {
+ for (int entryCount : new int[] {32, 4097}) {
+ TrackingLocalFileIO trackingIO = new TrackingLocalFileIO();
+ RowType rowType = RowType.of(DataTypes.MAP(DataTypes.INT(),
DataTypes.BLOB()));
+ Map<Object, Object> entries = new LinkedHashMap<>();
+ entries.put(null, null);
+ for (int i = 0; i < entryCount; i++) {
+ entries.put(i, new BlobData(i == 0 ? new byte[0] : new byte[]
{1, 2, 3}));
+ }
+ Path mapFile = new Path(parent, UUID.randomUUID().toString());
+ BlobFileFormat format =
+ new BlobFileFormat(true,
BlobFormatWriter.DEFAULT_COPY_BUFFER_SIZE);
+ try (PositionOutputStream out =
trackingIO.newOutputStream(mapFile, false)) {
+ FormatWriter writer =
format.createWriterFactory(rowType).create(out, null);
+ writer.addElement(GenericRow.of(new GenericMap(entries)));
+ writer.close();
+ }
+
+ FormatReaderContext context =
+ new FormatReaderContext(
+ trackingIO, mapFile,
trackingIO.getFileSize(mapFile), null, null);
+ List<InternalRow> rows = new ArrayList<>();
+ try (FileRecordReader<InternalRow> reader =
+ format.createReaderFactory(null, rowType,
null).createReader(context)) {
+ reader.forEachRemaining(rows::add);
+ }
+ GenericMap result = (GenericMap) rows.get(0).getMap(0);
+ assertThat(result.size()).isEqualTo(entryCount + 1);
+ assertThat(result.get(null)).isNull();
+ long valueStart = 4 + 9 + (long) entryCount * Integer.BYTES;
+ long valueEnd = valueStart + (entryCount - 1L) * 3;
+ for (int i = 0; i < entryCount; i++) {
+ Blob blob = (Blob) result.get(i);
+ assertThat(blob).isInstanceOf(BlobRef.class);
+ assertThat(blob.toDescriptor().offset())
+ .isEqualTo(valueStart + Math.max(0, i - 1L) * 3);
+ assertThat(blob.toDescriptor().length()).isEqualTo(i == 0 ? 0
: 3);
+ }
+ List<long[]> ranges = trackingIO.lastInputStream.readRanges;
+ if (entryCount == 32) {
+ // File footer/index plus map header, lengths, combined
indexes and keys.
+ assertThat(ranges).hasSize(6);
Review Comment:
Nit: avoid coupling this regression to the readers total req count
This exact count also includes unrelated file-footer and row-index reads, so
a future optimization or refactor in those readers could break this test even
if the map metadata remains correctly coalesced. Could we assert the relevant
map index/key ranges directly, or use an upper bound here as in the
large-metadata case? That would preserve the optimization guarantee without
coupling the test to the complete reader request sequence.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]