From f2ce938f4435c3b4b6e61d3133ad749999f461f3 Mon Sep 17 00:00:00 2001 From: Costas Zarifis Date: Mon, 28 Sep 2026 07:51:51 +0000 Subject: [PATCH] GH-3828: Preserve column path identity in dictionary lookup --- .../parquet/hadoop/DictionaryPageReader.java | 15 ++--- .../parquet/hadoop/TestParquetWriter.java | 58 +++++++++++++++++++ 2 files changed, 66 insertions(+), 7 deletions(-) diff --git a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/DictionaryPageReader.java b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/DictionaryPageReader.java index e1118b6076..1fdaab53fa 100644 --- a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/DictionaryPageReader.java +++ b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/DictionaryPageReader.java @@ -31,6 +31,7 @@ import org.apache.parquet.column.page.DictionaryPageReadStore; import org.apache.parquet.hadoop.metadata.BlockMetaData; import org.apache.parquet.hadoop.metadata.ColumnChunkMetaData; +import org.apache.parquet.hadoop.metadata.ColumnPath; import org.apache.parquet.io.ParquetDecodingException; /** @@ -44,8 +45,8 @@ class DictionaryPageReader implements DictionaryPageReadStore { private final ParquetFileReader reader; - private final Map columns; - private final Map> dictionaryPageCache; + private final Map columns; + private final Map> dictionaryPageCache; private ColumnChunkPageReadStore rowGroup = null; private ByteBufferReleaser releaser; @@ -64,7 +65,7 @@ class DictionaryPageReader implements DictionaryPageReadStore { releaser = new ByteBufferReleaser(allocator); for (ColumnChunkMetaData column : block.getColumns()) { - columns.put(column.getPath().toDotString(), column); + columns.put(column.getPath(), column); } } @@ -87,14 +88,14 @@ public DictionaryPage readDictionaryPage(ColumnDescriptor descriptor) { return rowGroup.readDictionaryPage(descriptor); } - String dotPath = String.join(".", descriptor.getPath()); - ColumnChunkMetaData column = columns.get(dotPath); + ColumnPath path = ColumnPath.get(descriptor.getPath()); + ColumnChunkMetaData column = columns.get(path); if (column == null) { - throw new ParquetDecodingException("Failed to load dictionary, unknown column: " + dotPath); + throw new ParquetDecodingException("Failed to load dictionary, unknown column: " + path); } return dictionaryPageCache - .computeIfAbsent(dotPath, key -> { + .computeIfAbsent(path, key -> { try { final DictionaryPage dict = column.hasDictionaryPage() ? reader.readDictionary(column) : null; diff --git a/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestParquetWriter.java b/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestParquetWriter.java index c501298e76..e561f762ef 100644 --- a/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestParquetWriter.java +++ b/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestParquetWriter.java @@ -61,11 +61,14 @@ import org.apache.parquet.bytes.HeapByteBufferAllocator; import org.apache.parquet.bytes.TrackingByteBufferAllocator; import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.column.Dictionary; import org.apache.parquet.column.Encoding; import org.apache.parquet.column.ParquetProperties; import org.apache.parquet.column.ParquetProperties.WriterVersion; import org.apache.parquet.column.page.DataPage; import org.apache.parquet.column.page.DataPageV2; +import org.apache.parquet.column.page.DictionaryPage; +import org.apache.parquet.column.page.DictionaryPageReadStore; import org.apache.parquet.column.page.PageReadStore; import org.apache.parquet.column.page.PageReader; import org.apache.parquet.column.values.bloomfilter.BloomFilter; @@ -335,6 +338,61 @@ public void testParquetFileWithBloomFilter() throws IOException { } } + @Test + public void testDictionaryReaderKeepsCollidingDotStringPathsDistinct() throws Exception { + MessageType schema = Types.buildMessage() + .required(BINARY) + .as(stringType()) + .named("a.b") + .requiredGroup() + .required(BINARY) + .as(stringType()) + .named("b") + .named("a") + .named("msg"); + Configuration conf = new Configuration(); + GroupWriteSupport.setSchema(schema, conf); + GroupFactory factory = new SimpleGroupFactory(schema); + + Path path = newTempPath(); + try (ParquetWriter writer = ExampleParquetWriter.builder(path) + .withAllocator(allocator) + .withConf(conf) + .withDictionaryEncoding(true) + .build()) { + for (int i = 0; i < 100; i++) { + String suffix = (i % 2 == 0) ? "one" : "two"; + Group group = factory.newGroup().append("a.b", suffix); + group.addGroup("a").append("b", "nested-" + suffix); + writer.write(group); + } + } + + try (ParquetFileReader reader = ParquetFileReader.open(HadoopInputFile.fromPath(path, conf))) { + ColumnDescriptor topLevelDescriptor = schema.getColumnDescription(new String[] {"a.b"}); + ColumnDescriptor nestedDescriptor = schema.getColumnDescription(new String[] {"a", "b"}); + DictionaryPageReadStore dictionaryReader = reader.getNextDictionaryReader(); + DictionaryPage topLevelPage = dictionaryReader.readDictionaryPage(topLevelDescriptor); + DictionaryPage nestedPage = dictionaryReader.readDictionaryPage(nestedDescriptor); + assertThat(topLevelPage).isNotNull(); + assertThat(nestedPage).isNotNull(); + + Dictionary topLevelDictionary = topLevelPage.getEncoding().initDictionary(topLevelDescriptor, topLevelPage); + Set topLevelValues = new HashSet<>(); + for (int id = 0; id <= topLevelDictionary.getMaxId(); id++) { + topLevelValues.add(topLevelDictionary.decodeToBinary(id).toStringUsingUTF8()); + } + assertThat(topLevelValues).containsExactlyInAnyOrder("one", "two"); + + Dictionary nestedDictionary = nestedPage.getEncoding().initDictionary(nestedDescriptor, nestedPage); + Set nestedValues = new HashSet<>(); + for (int id = 0; id <= nestedDictionary.getMaxId(); id++) { + nestedValues.add(nestedDictionary.decodeToBinary(id).toStringUsingUTF8()); + } + assertThat(nestedValues).containsExactlyInAnyOrder("nested-one", "nested-two"); + } + } + @Test public void testParquetFileWithBloomFilterWithFpp() throws IOException { int buildBloomFilterCount = 100000;