Search in sources :

Example 1 with GroupReadSupport

use of org.apache.parquet.hadoop.example.GroupReadSupport in project parquet-mr by apache.

the class TestParquetWriterNewPage method test.

@Test
public void test() throws Exception {
    Configuration conf = new Configuration();
    Path root = new Path("target/tests/TestParquetWriter/");
    FileSystem fs = root.getFileSystem(conf);
    if (fs.exists(root)) {
        fs.delete(root, true);
    }
    fs.mkdirs(root);
    MessageType schema = parseMessageType("message test { " + "required binary binary_field; " + "required int32 int32_field; " + "required int64 int64_field; " + "required boolean boolean_field; " + "required float float_field; " + "required double double_field; " + "required fixed_len_byte_array(3) flba_field; " + "required int96 int96_field; " + "optional binary null_field; " + "} ");
    GroupWriteSupport.setSchema(schema, conf);
    SimpleGroupFactory f = new SimpleGroupFactory(schema);
    Map<String, Encoding> expected = new HashMap<String, Encoding>();
    expected.put("10-" + PARQUET_1_0, PLAIN_DICTIONARY);
    expected.put("1000-" + PARQUET_1_0, PLAIN);
    expected.put("10-" + PARQUET_2_0, RLE_DICTIONARY);
    expected.put("1000-" + PARQUET_2_0, DELTA_BYTE_ARRAY);
    for (int modulo : asList(10, 1000)) {
        for (WriterVersion version : WriterVersion.values()) {
            Path file = new Path(root, version.name() + "_" + modulo);
            ParquetWriter<Group> writer = new ParquetWriter<Group>(file, new GroupWriteSupport(), UNCOMPRESSED, 1024, 1024, 512, true, false, version, conf);
            for (int i = 0; i < 1000; i++) {
                writer.write(f.newGroup().append("binary_field", "test" + (i % modulo)).append("int32_field", 32).append("int64_field", 64l).append("boolean_field", true).append("float_field", 1.0f).append("double_field", 2.0d).append("flba_field", "foo").append("int96_field", Binary.fromConstantByteArray(new byte[12])));
            }
            writer.close();
            ParquetReader<Group> reader = ParquetReader.builder(new GroupReadSupport(), file).withConf(conf).build();
            for (int i = 0; i < 1000; i++) {
                Group group = reader.read();
                assertEquals("test" + (i % modulo), group.getBinary("binary_field", 0).toStringUsingUTF8());
                assertEquals(32, group.getInteger("int32_field", 0));
                assertEquals(64l, group.getLong("int64_field", 0));
                assertEquals(true, group.getBoolean("boolean_field", 0));
                assertEquals(1.0f, group.getFloat("float_field", 0), 0.001);
                assertEquals(2.0d, group.getDouble("double_field", 0), 0.001);
                assertEquals("foo", group.getBinary("flba_field", 0).toStringUsingUTF8());
                assertEquals(Binary.fromConstantByteArray(new byte[12]), group.getInt96("int96_field", 0));
                assertEquals(0, group.getFieldRepetitionCount("null_field"));
            }
            reader.close();
            ParquetMetadata footer = readFooter(conf, file, NO_FILTER);
            for (BlockMetaData blockMetaData : footer.getBlocks()) {
                for (ColumnChunkMetaData column : blockMetaData.getColumns()) {
                    if (column.getPath().toDotString().equals("binary_field")) {
                        String key = modulo + "-" + version;
                        Encoding expectedEncoding = expected.get(key);
                        assertTrue(key + ":" + column.getEncodings() + " should contain " + expectedEncoding, column.getEncodings().contains(expectedEncoding));
                    }
                }
            }
        }
    }
}
Also used : Path(org.apache.hadoop.fs.Path) Group(org.apache.parquet.example.data.Group) BlockMetaData(org.apache.parquet.hadoop.metadata.BlockMetaData) GroupReadSupport(org.apache.parquet.hadoop.example.GroupReadSupport) Configuration(org.apache.hadoop.conf.Configuration) HashMap(java.util.HashMap) ParquetMetadata(org.apache.parquet.hadoop.metadata.ParquetMetadata) ColumnChunkMetaData(org.apache.parquet.hadoop.metadata.ColumnChunkMetaData) SimpleGroupFactory(org.apache.parquet.example.data.simple.SimpleGroupFactory) Encoding(org.apache.parquet.column.Encoding) WriterVersion(org.apache.parquet.column.ParquetProperties.WriterVersion) GroupWriteSupport(org.apache.parquet.hadoop.example.GroupWriteSupport) FileSystem(org.apache.hadoop.fs.FileSystem) MessageTypeParser.parseMessageType(org.apache.parquet.schema.MessageTypeParser.parseMessageType) MessageType(org.apache.parquet.schema.MessageType) Test(org.junit.Test)

Example 2 with GroupReadSupport

use of org.apache.parquet.hadoop.example.GroupReadSupport in project parquet-mr by apache.

the class TestParquetWriterAppendBlocks method testBasicBehavior.

@Test
public void testBasicBehavior() throws IOException {
    Path combinedFile = newTemp();
    ParquetFileWriter writer = new ParquetFileWriter(CONF, FILE_SCHEMA, combinedFile);
    writer.start();
    writer.appendFile(CONF, file1);
    writer.appendFile(CONF, file2);
    writer.end(EMPTY_METADATA);
    LinkedList<Group> expected = new LinkedList<Group>();
    expected.addAll(file1content);
    expected.addAll(file2content);
    ParquetReader<Group> reader = ParquetReader.builder(new GroupReadSupport(), combinedFile).build();
    Group next;
    while ((next = reader.read()) != null) {
        Group expectedNext = expected.removeFirst();
        // check each value; equals is not supported for simple records
        Assert.assertEquals("Each id should match", expectedNext.getInteger("id", 0), next.getInteger("id", 0));
        Assert.assertEquals("Each string should match", expectedNext.getString("string", 0), next.getString("string", 0));
    }
    Assert.assertEquals("All records should be present", 0, expected.size());
}
Also used : Path(org.apache.hadoop.fs.Path) Group(org.apache.parquet.example.data.Group) GroupReadSupport(org.apache.parquet.hadoop.example.GroupReadSupport) LinkedList(java.util.LinkedList) Test(org.junit.Test)

Example 3 with GroupReadSupport

use of org.apache.parquet.hadoop.example.GroupReadSupport in project parquet-mr by apache.

the class TestParquetWriterAppendBlocks method testAllowDroppingColumns.

@Test
public void testAllowDroppingColumns() throws IOException {
    MessageType droppedColumnSchema = Types.buildMessage().required(BINARY).as(UTF8).named("string").named("AppendTest");
    Path droppedColumnFile = newTemp();
    ParquetFileWriter writer = new ParquetFileWriter(CONF, droppedColumnSchema, droppedColumnFile);
    writer.start();
    writer.appendFile(CONF, file1);
    writer.appendFile(CONF, file2);
    writer.end(EMPTY_METADATA);
    LinkedList<Group> expected = new LinkedList<Group>();
    expected.addAll(file1content);
    expected.addAll(file2content);
    ParquetMetadata footer = ParquetFileReader.readFooter(CONF, droppedColumnFile, NO_FILTER);
    for (BlockMetaData rowGroup : footer.getBlocks()) {
        Assert.assertEquals("Should have only the string column", 1, rowGroup.getColumns().size());
    }
    ParquetReader<Group> reader = ParquetReader.builder(new GroupReadSupport(), droppedColumnFile).build();
    Group next;
    while ((next = reader.read()) != null) {
        Group expectedNext = expected.removeFirst();
        Assert.assertEquals("Each string should match", expectedNext.getString("string", 0), next.getString("string", 0));
    }
    Assert.assertEquals("All records should be present", 0, expected.size());
}
Also used : Path(org.apache.hadoop.fs.Path) Group(org.apache.parquet.example.data.Group) BlockMetaData(org.apache.parquet.hadoop.metadata.BlockMetaData) GroupReadSupport(org.apache.parquet.hadoop.example.GroupReadSupport) ParquetMetadata(org.apache.parquet.hadoop.metadata.ParquetMetadata) MessageType(org.apache.parquet.schema.MessageType) LinkedList(java.util.LinkedList) Test(org.junit.Test)

Example 4 with GroupReadSupport

use of org.apache.parquet.hadoop.example.GroupReadSupport in project parquet-mr by apache.

the class TestFiltersWithMissingColumns method countFilteredRecords.

public static long countFilteredRecords(Path path, FilterPredicate pred) throws IOException {
    ParquetReader<Group> reader = ParquetReader.builder(new GroupReadSupport(), path).withFilter(FilterCompat.get(pred)).build();
    long count = 0;
    try {
        while (reader.read() != null) {
            count += 1;
        }
    } finally {
        reader.close();
    }
    return count;
}
Also used : Group(org.apache.parquet.example.data.Group) GroupReadSupport(org.apache.parquet.hadoop.example.GroupReadSupport)

Example 5 with GroupReadSupport

use of org.apache.parquet.hadoop.example.GroupReadSupport in project apex-malhar by apache.

the class AbstractParquetFileReader method openFile.

/**
 * Opens the file to read using GroupReadSupport
 */
@Override
protected InputStream openFile(Path path) throws IOException {
    InputStream is = super.openFile(path);
    GroupReadSupport readSupport = new GroupReadSupport();
    readSupport.init(configuration, null, schema);
    reader = new ParquetReader<>(path, readSupport);
    return is;
}
Also used : GroupReadSupport(org.apache.parquet.hadoop.example.GroupReadSupport) InputStream(java.io.InputStream)

Aggregations

GroupReadSupport (org.apache.parquet.hadoop.example.GroupReadSupport)9 Group (org.apache.parquet.example.data.Group)7 Path (org.apache.hadoop.fs.Path)5 Configuration (org.apache.hadoop.conf.Configuration)4 ParquetMetadata (org.apache.parquet.hadoop.metadata.ParquetMetadata)4 MessageType (org.apache.parquet.schema.MessageType)4 Test (org.junit.Test)4 BlockMetaData (org.apache.parquet.hadoop.metadata.BlockMetaData)3 HashMap (java.util.HashMap)2 LinkedList (java.util.LinkedList)2 Encoding (org.apache.parquet.column.Encoding)2 WriterVersion (org.apache.parquet.column.ParquetProperties.WriterVersion)2 SimpleGroupFactory (org.apache.parquet.example.data.simple.SimpleGroupFactory)2 GroupWriteSupport (org.apache.parquet.hadoop.example.GroupWriteSupport)2 ColumnChunkMetaData (org.apache.parquet.hadoop.metadata.ColumnChunkMetaData)2 MessageTypeParser.parseMessageType (org.apache.parquet.schema.MessageTypeParser.parseMessageType)2 InputStream (java.io.InputStream)1 ArrayList (java.util.ArrayList)1 FileSystem (org.apache.hadoop.fs.FileSystem)1 SimpleGroup (org.apache.parquet.example.data.simple.SimpleGroup)1