Skip to content
This repository has been archived by the owner on Jan 11, 2021. It is now read-only.

Commit

Permalink
Add benchmark for compression codecs (#96)
Browse files Browse the repository at this point in the history
  • Loading branch information
sunchao committed Apr 25, 2018
1 parent f0bcc82 commit 8ef823d
Show file tree
Hide file tree
Showing 3 changed files with 169 additions and 1 deletion.
5 changes: 4 additions & 1 deletion Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,10 @@ byteorder = "1"
snap = "0.2"
brotli = "1.1.2"
flate2 = "0.2"
rand = "0.4"
thrift = "0.0.4"
x86intrin = "0.4.3"
chrono = "0.4"

[dev-dependencies]
lazy_static = "1"
rand = "0.4"
165 changes: 165 additions & 0 deletions benches/codec.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,165 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.

#![feature(test)]
extern crate parquet;
#[macro_use]
extern crate lazy_static;
extern crate test;
use test::Bencher;

use std::env;
use std::fs::File;

use parquet::basic::Compression;
use parquet::file::reader::*;
use parquet::compression::*;

// 10k rows written in page v2 with type:
//
// message test {
// required binary binary_field,
// required int32 int32_field,
// required int64 int64_field,
// required boolean boolean_field,
// required float float_field,
// required double double_field,
// required fixed_len_byte_array(1024) flba_field,
// required int96 int96_field
// }
//
// filled with random values.
const TEST_FILE: &str = "10k-v2.parquet";

fn get_rg_reader() -> Box<RowGroupReader> {
let mut path_buf = env::current_dir().unwrap();
path_buf.push("data");
path_buf.push(TEST_FILE);
let file = File::open(path_buf.as_path()).unwrap();
let f_reader = SerializedFileReader::new(file).unwrap();
f_reader.get_row_group(0).unwrap()
}

fn get_pages_bytes(col_idx: usize) -> Vec<u8> {
let mut data: Vec<u8> = Vec::new();
let rg_reader = get_rg_reader();
let mut pg_reader = rg_reader.get_column_page_reader(col_idx).unwrap();
loop {
if let Some(p) = pg_reader.get_next_page().unwrap() {
data.extend_from_slice(p.buffer().data());
} else {
break;
}
}
data
}

macro_rules! compress {
($fname:ident, $codec:expr, $col_idx:expr) => {
#[bench]
fn $fname(bench: &mut Bencher) {
lazy_static! {
static ref DATA: Vec<u8> = {
get_pages_bytes($col_idx)
};
}

let mut codec = create_codec($codec).unwrap().unwrap();
bench.bytes = DATA.len() as u64;
bench.iter(|| {
let _ = codec.compress(&DATA[..]).unwrap();
})
}
}
}

macro_rules! decompress {
($fname:ident, $codec:expr, $col_idx:expr) => {
#[bench]
fn $fname(bench: &mut Bencher) {
lazy_static! {
static ref COMPRESSED_PAGES: Vec<u8> = {
let mut codec = create_codec($codec).unwrap().unwrap();
let raw_data = get_pages_bytes($col_idx);
codec.compress(&raw_data[..]).unwrap()
};
}

let mut codec = create_codec($codec).unwrap().unwrap();
let rg_reader = get_rg_reader();
bench.bytes = rg_reader.metadata().total_byte_size() as u64;
bench.iter(|| {
let mut v = Vec::new();
let _ = codec.decompress(&COMPRESSED_PAGES[..], &mut v).unwrap();
})
}
}
}

compress!(compress_brotli_binary, Compression::BROTLI, 0);
compress!(compress_brotli_int32, Compression::BROTLI, 1);
compress!(compress_brotli_int64, Compression::BROTLI, 2);
compress!(compress_brotli_boolean, Compression::BROTLI, 3);
compress!(compress_brotli_float, Compression::BROTLI, 4);
compress!(compress_brotli_double, Compression::BROTLI, 5);
compress!(compress_brotli_fixed, Compression::BROTLI, 6);
compress!(compress_brotli_int96, Compression::BROTLI, 7);

compress!(compress_gzip_binary, Compression::GZIP, 0);
compress!(compress_gzip_int32, Compression::GZIP, 1);
compress!(compress_gzip_int64, Compression::GZIP, 2);
compress!(compress_gzip_boolean, Compression::GZIP, 3);
compress!(compress_gzip_float, Compression::GZIP, 4);
compress!(compress_gzip_double, Compression::GZIP, 5);
compress!(compress_gzip_fixed, Compression::GZIP, 6);
compress!(compress_gzip_int96, Compression::GZIP, 7);

compress!(compress_snappy_binary, Compression::SNAPPY, 0);
compress!(compress_snappy_int32, Compression::SNAPPY, 1);
compress!(compress_snappy_int64, Compression::SNAPPY, 2);
compress!(compress_snappy_boolean, Compression::SNAPPY, 3);
compress!(compress_snappy_float, Compression::SNAPPY, 4);
compress!(compress_snappy_double, Compression::SNAPPY, 5);
compress!(compress_snappy_fixed, Compression::SNAPPY, 6);
compress!(compress_snappy_int96, Compression::SNAPPY, 7);

decompress!(decompress_brotli_binary, Compression::BROTLI, 0);
decompress!(decompress_brotli_int32, Compression::BROTLI, 1);
decompress!(decompress_brotli_int64, Compression::BROTLI, 2);
decompress!(decompress_brotli_boolean, Compression::BROTLI, 3);
decompress!(decompress_brotli_float, Compression::BROTLI, 4);
decompress!(decompress_brotli_double, Compression::BROTLI, 5);
decompress!(decompress_brotli_fixed, Compression::BROTLI, 6);
decompress!(decompress_brotli_int96, Compression::BROTLI, 7);

decompress!(decompress_gzip_binary, Compression::GZIP, 0);
decompress!(decompress_gzip_int32, Compression::GZIP, 1);
decompress!(decompress_gzip_int64, Compression::GZIP, 2);
decompress!(decompress_gzip_boolean, Compression::GZIP, 3);
decompress!(decompress_gzip_float, Compression::GZIP, 4);
decompress!(decompress_gzip_double, Compression::GZIP, 5);
decompress!(decompress_gzip_fixed, Compression::GZIP, 6);
decompress!(decompress_gzip_int96, Compression::GZIP, 7);

decompress!(decompress_snappy_binary, Compression::SNAPPY, 0);
decompress!(decompress_snappy_int32, Compression::SNAPPY, 1);
decompress!(decompress_snappy_int64, Compression::SNAPPY, 2);
decompress!(decompress_snappy_boolean, Compression::SNAPPY, 3);
decompress!(decompress_snappy_float, Compression::SNAPPY, 4);
decompress!(decompress_snappy_double, Compression::SNAPPY, 5);
decompress!(decompress_snappy_fixed, Compression::SNAPPY, 6);
decompress!(decompress_snappy_int96, Compression::SNAPPY, 7);
Binary file added data/10k-v2.parquet
Binary file not shown.

0 comments on commit 8ef823d

Please sign in to comment.