diff --git a/parquet-column/src/main/java/org/apache/parquet/schema/Types.java b/parquet-column/src/main/java/org/apache/parquet/schema/Types.java index 2f12991ab0..6c1c855846 100644 --- a/parquet-column/src/main/java/org/apache/parquet/schema/Types.java +++ b/parquet-column/src/main/java/org/apache/parquet/schema/Types.java @@ -344,6 +344,9 @@ public abstract static class BasePrimitiveBuilder parquetSchema = Lists.newArrayList(new SchemaElement("Message").setNum_children(1), leaf); + List columnOrders = + Lists.newArrayList(new org.apache.parquet.format.ColumnOrder()); + columnOrders.get(0).setTYPE_ORDER(new TypeDefinedOrder()); + + MessageType schema = converter.fromParquetSchema(parquetSchema, columnOrders); + + PrimitiveType result = schema.getType("bool_ts").asPrimitiveType(); + assertThat(result.getPrimitiveTypeName()).isEqualTo(PrimitiveTypeName.BOOLEAN); + assertThat(result.getLogicalTypeAnnotation()).isNull(); + assertThat(result.columnOrder().getColumnOrderName()).isEqualTo(ColumnOrder.ColumnOrderName.UNDEFINED); + } + + private static PrimitiveType droppedAnnotationInt32() { + return Types.optional(PrimitiveTypeName.INT32) + .columnOrder(ColumnOrder.undefined()) + .named("ts_int32"); + } + + @Test + public void testDroppedAnnotationIgnoresStats() { + ParquetMetadataConverter converter = new ParquetMetadataConverter(); + org.apache.parquet.format.Statistics stats = new org.apache.parquet.format.Statistics(); + stats.setMin_value(new byte[] {1, 2, 3, 4}); + stats.setMax_value(new byte[] {0, 1, 2, 3}); + stats.setNull_count(3L); + + Statistics result = converter.fromParquetStatistics(Version.FULL_VERSION, stats, droppedAnnotationInt32()); + + assertThat(result.hasNonNullValue()).isFalse(); + assertThat(result.isNumNullsSet()).isTrue(); + assertThat(result.getNumNulls()).isEqualTo(3L); + } + + @Test + public void testDroppedAnnotationColumnIndexIsNull() { + PrimitiveType int32Type = Types.required(PrimitiveTypeName.INT32).named("i32"); + ColumnIndexBuilder cb = ColumnIndexBuilder.getBuilder(int32Type, Integer.MAX_VALUE); + Statistics stats = Statistics.createStats(int32Type); + stats.updateStats(-100); + stats.updateStats(100); + cb.add(stats, null); + org.apache.parquet.format.ColumnIndex parquetColumnIndex = + ParquetMetadataConverter.toParquetColumnIndex(int32Type, cb.build()); + + assertThat(ParquetMetadataConverter.fromParquetColumnIndex(droppedAnnotationInt32(), parquetColumnIndex)) + .isNull(); + } } diff --git a/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestReadInvalidTypeCombination.java b/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestReadInvalidTypeCombination.java new file mode 100644 index 0000000000..6726ce6a06 --- /dev/null +++ b/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestReadInvalidTypeCombination.java @@ -0,0 +1,77 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.hadoop; + +import static org.assertj.core.api.Assertions.assertThat; + +import java.net.URISyntaxException; +import org.apache.hadoop.conf.Configuration; +import org.apache.hadoop.fs.Path; +import org.apache.parquet.example.data.Group; +import org.apache.parquet.hadoop.example.GroupReadSupport; +import org.apache.parquet.hadoop.metadata.ParquetMetadata; +import org.apache.parquet.hadoop.util.HadoopInputFile; +import org.apache.parquet.schema.ColumnOrder.ColumnOrderName; +import org.apache.parquet.schema.PrimitiveType; +import org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName; +import org.junit.jupiter.api.Test; + +public class TestReadInvalidTypeCombination { + + // Path to a Parquet file that contains an invalid logical/physical type combination. + private static final String FILE_PATH = "/invalid_type_combination.parquet"; + + private static Path getFilePath() { + try { + return new Path(TestReadInvalidTypeCombination.class.getResource(FILE_PATH).toURI()); + } catch (URISyntaxException e) { + throw new RuntimeException(e); + } + } + + @Test + public void testReadInvalidTypeCombinationSucceeds() throws Exception { + Configuration conf = new Configuration(); + Path file = getFilePath(); + + // The footer parse should succeed and drop the annotation and stats for the column. + try (ParquetFileReader reader = ParquetFileReader.open(HadoopInputFile.fromPath(file, conf))) { + ParquetMetadata footer = reader.getFooter(); + PrimitiveType column = + footer.getFileMetaData().getSchema().getType("int32_uuid").asPrimitiveType(); + + assertThat(column.getPrimitiveTypeName()).isEqualTo(PrimitiveTypeName.INT32); + assertThat(column.getLogicalTypeAnnotation()).isNull(); + assertThat(column.columnOrder().getColumnOrderName()).isEqualTo(ColumnOrderName.UNDEFINED); + } + + // The physical values are still fully readable. + int rows = 0; + try (ParquetReader reader = ParquetReader.builder(new GroupReadSupport(), file) + .withConf(conf) + .build()) { + Group g; + while ((g = reader.read()) != null) { + assertThat(g.getInteger("int32_uuid", 0)).isEqualTo(rows); + rows++; + } + } + assertThat(rows).isEqualTo(10); + } +} diff --git a/parquet-hadoop/src/test/resources/invalid_type_combination.parquet b/parquet-hadoop/src/test/resources/invalid_type_combination.parquet new file mode 100644 index 0000000000..400136461a Binary files /dev/null and b/parquet-hadoop/src/test/resources/invalid_type_combination.parquet differ