Skip to content
New issue

Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.

By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.

Already on GitHub? Sign in to your account

[fix](hive) do not split compress data file and support lz4/snappy block codec #23245

Merged
merged 12 commits into from
Aug 26, 2023
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
185 changes: 178 additions & 7 deletions be/src/exec/decompressor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,12 @@ Status Decompressor::create_decompressor(CompressType type, Decompressor** decom
case CompressType::LZ4FRAME:
*decompressor = new Lz4FrameDecompressor();
break;
case CompressType::LZ4BLOCK:
*decompressor = new Lz4BlockDecompressor();
break;
case CompressType::SNAPPYBLOCK:
*decompressor = new SnappyBlockDecompressor();
break;
#ifdef DORIS_WITH_LZO
case CompressType::LZOP:
*decompressor = new LzopDecompressor();
Expand All @@ -59,6 +65,10 @@ Status Decompressor::create_decompressor(CompressType type, Decompressor** decom
return st;
}

uint32_t Decompressor::_read_int32(uint8_t* buf) {
return (buf[0] << 24) | (buf[1] << 16) | (buf[2] << 8) | buf[3];
}

std::string Decompressor::debug_info() {
return "Decompressor";
}
Expand Down Expand Up @@ -239,7 +249,7 @@ Status Lz4FrameDecompressor::decompress(uint8_t* input, size_t input_len, size_t
size_t* decompressed_len, bool* stream_end,
size_t* more_input_bytes, size_t* more_output_bytes) {
uint8_t* src = input;
size_t src_size = input_len;
size_t remaining_input_size = input_len;
size_t ret = 1;
*input_bytes_read = 0;

Expand All @@ -257,7 +267,7 @@ Status Lz4FrameDecompressor::decompress(uint8_t* input, size_t input_len, size_t
}

LZ4F_frameInfo_t info;
ret = LZ4F_getFrameInfo(_dctx, &info, (void*)src, &src_size);
ret = LZ4F_getFrameInfo(_dctx, &info, (void*)src, &remaining_input_size);
if (LZ4F_isError(ret)) {
return Status::InternalError("LZ4F_getFrameInfo error: {}",
std::string(LZ4F_getErrorName(ret)));
Expand All @@ -270,25 +280,25 @@ Status Lz4FrameDecompressor::decompress(uint8_t* input, size_t input_len, size_t
std::string(LZ4F_getErrorName(ret)));
}

*input_bytes_read = src_size;
*input_bytes_read = remaining_input_size;

src += src_size;
src_size = input_len - src_size;
src += remaining_input_size;
remaining_input_size = input_len - remaining_input_size;

LOG(INFO) << "lz4 block size: " << _expect_dec_buf_size;
}

// decompress
size_t output_len = output_max_len;
ret = LZ4F_decompress(_dctx, (void*)output, &output_len, (void*)src, &src_size,
ret = LZ4F_decompress(_dctx, (void*)output, &output_len, (void*)src, &remaining_input_size,
/* LZ4F_decompressOptions_t */ nullptr);
if (LZ4F_isError(ret)) {
return Status::InternalError("Decompression error: {}",
std::string(LZ4F_getErrorName(ret)));
}

// update
*input_bytes_read += src_size;
*input_bytes_read += remaining_input_size;
*decompressed_len = output_len;
if (ret == 0) {
*stream_end = true;
Expand Down Expand Up @@ -324,4 +334,165 @@ size_t Lz4FrameDecompressor::get_block_size(const LZ4F_frameInfo_t* info) {
}
}

/// Lz4BlockDecompressor
Status Lz4BlockDecompressor::init() {
return Status::OK();
}

Status Lz4BlockDecompressor::decompress(uint8_t* input, size_t input_len, size_t* input_bytes_read,
uint8_t* output, size_t output_max_len,
size_t* decompressed_len, bool* stream_end,
size_t* more_input_bytes, size_t* more_output_bytes) {
uint8_t* src = input;
size_t remaining_input_size = input_len;
int64_t uncompressed_total_len = 0;
*input_bytes_read = 0;

// The hadoop lz4 codec is as:
// <4 byte big endian uncompressed size>
// <4 byte big endian compressed size>
// <lz4 compressed block>
// ....
// <4 byte big endian uncompressed size>
// <4 byte big endian compressed size>
// <lz4 compressed block>
//
// See:
// https://github.com/apache/hadoop/blob/trunk/hadoop-mapreduce-project/hadoop-mapreduce-client/hadoop-mapreduce-client-nativetask/src/main/native/src/codec/Lz4Codec.cc
while (remaining_input_size > 0) {
// Read uncompressed size
uint32_t uncompressed_block_len = Decompressor::_read_int32(src);
int64_t remaining_output_size = output_max_len - uncompressed_total_len;
if (remaining_output_size < uncompressed_block_len) {
// Need more output buffer
*more_output_bytes = uncompressed_block_len - remaining_output_size;
break;
}

// Read compressed size
size_t tmp_src_size = remaining_input_size - sizeof(uint32_t);
size_t compressed_len = Decompressor::_read_int32(src + sizeof(uint32_t));
if (compressed_len == 0 || compressed_len > tmp_src_size) {
// Need more input data
*more_input_bytes = compressed_len - tmp_src_size;
break;
}

src += 2 * sizeof(uint32_t);
remaining_input_size -= 2 * sizeof(uint32_t);

// Decompress
int uncompressed_len = LZ4_decompress_safe(reinterpret_cast<const char*>(src),
reinterpret_cast<char*>(output), compressed_len,
remaining_output_size);
if (uncompressed_len < 0 || uncompressed_len != uncompressed_block_len) {
return Status::InternalError(
"lz4 block decompress failed. uncompressed_len: {}, expected: {}",
uncompressed_len, uncompressed_block_len);
}

output += uncompressed_len;
src += compressed_len;
remaining_input_size -= compressed_len;
uncompressed_total_len += uncompressed_len;
}

*input_bytes_read += (input_len - remaining_input_size);
*decompressed_len = uncompressed_total_len;
// If no more input and output need, means this is the end of a compressed block
*stream_end = (*more_input_bytes == 0 && *more_output_bytes == 0);

return Status::OK();
}

std::string Lz4BlockDecompressor::debug_info() {
std::stringstream ss;
ss << "Lz4BlockDecompressor.";
return ss.str();
}

/// SnappyBlockDecompressor
Status SnappyBlockDecompressor::init() {
return Status::OK();
}

Status SnappyBlockDecompressor::decompress(uint8_t* input, size_t input_len,
size_t* input_bytes_read, uint8_t* output,
size_t output_max_len, size_t* decompressed_len,
bool* stream_end, size_t* more_input_bytes,
size_t* more_output_bytes) {
uint8_t* src = input;
size_t remaining_input_size = input_len;
int64_t uncompressed_total_len = 0;
*input_bytes_read = 0;

// The hadoop snappy codec is as:
// <4 byte big endian uncompressed size>
// <4 byte big endian compressed size>
// <snappy compressed block>
// ....
// <4 byte big endian uncompressed size>
// <4 byte big endian compressed size>
// <snappy compressed block>
//
// See:
// https://github.com/apache/hadoop/blob/trunk/hadoop-mapreduce-project/hadoop-mapreduce-client/hadoop-mapreduce-client-nativetask/src/main/native/src/codec/SnappyCodec.cc
while (remaining_input_size > 0) {
// Read uncompressed size
uint32_t uncompressed_block_len = Decompressor::_read_int32(src);
int64_t remaining_output_size = output_max_len - uncompressed_total_len;
if (remaining_output_size < uncompressed_block_len) {
// Need more output buffer
*more_output_bytes = uncompressed_block_len - remaining_output_size;
break;
}

// Read compressed size
size_t tmp_src_size = remaining_input_size - sizeof(uint32_t);
size_t compressed_len = _read_int32(src + sizeof(uint32_t));
if (compressed_len == 0 || compressed_len > tmp_src_size) {
// Need more input data
*more_input_bytes = compressed_len - tmp_src_size;
break;
}

src += 2 * sizeof(uint32_t);
remaining_input_size -= 2 * sizeof(uint32_t);

// ATTN: the uncompressed len from GetUncompressedLength() is same as
// uncompressed_block_len, so I think it is unnecessary to get it again.
// Get uncompressed len from snappy
// size_t uncompressed_len;
// if (!snappy::GetUncompressedLength(reinterpret_cast<const char*>(src),
// compressed_len, &uncompressed_len)) {
// return Status::InternalError("snappy block decompress failed to get uncompressed len");
// }

// Decompress
if (!snappy::RawUncompress(reinterpret_cast<const char*>(src), compressed_len,
reinterpret_cast<char*>(output))) {
return Status::InternalError("snappy block decompress failed. uncompressed_len: {}",
uncompressed_block_len);
}

output += uncompressed_block_len;
src += compressed_len;
remaining_input_size -= compressed_len;
uncompressed_total_len += uncompressed_block_len;
}

*input_bytes_read += (input_len - remaining_input_size);
*decompressed_len = uncompressed_total_len;
// If no more input and output need, means this is the end of a compressed block
*stream_end = (*more_input_bytes == 0 && *more_output_bytes == 0);

return Status::OK();
}

std::string SnappyBlockDecompressor::debug_info() {
std::stringstream ss;
ss << "SnappyBlockDecompressor.";
return ss.str();
}

} // namespace doris
39 changes: 38 additions & 1 deletion be/src/exec/decompressor.h
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,10 @@
#pragma once

#include <bzlib.h>
#include <lz4/lz4.h>
#include <lz4/lz4frame.h>
#include <lz4/lz4hc.h>
#include <snappy.h>
#include <stddef.h>
#include <stdint.h>
#include <zlib.h>
Expand All @@ -34,7 +37,7 @@

namespace doris {

enum CompressType { UNCOMPRESSED, GZIP, DEFLATE, BZIP2, LZ4FRAME, LZOP };
enum CompressType { UNCOMPRESSED, GZIP, DEFLATE, BZIP2, LZ4FRAME, LZOP, LZ4BLOCK, SNAPPYBLOCK };

class Decompressor {
public:
Expand Down Expand Up @@ -68,6 +71,8 @@ class Decompressor {
protected:
virtual Status init() = 0;

static uint32_t _read_int32(uint8_t* buf);

Decompressor(CompressType ctype) : _ctype(ctype) {}

CompressType _ctype;
Expand Down Expand Up @@ -140,6 +145,38 @@ class Lz4FrameDecompressor : public Decompressor {
const static unsigned DORIS_LZ4F_VERSION;
};

class Lz4BlockDecompressor : public Decompressor {
public:
~Lz4BlockDecompressor() override {}

Status decompress(uint8_t* input, size_t input_len, size_t* input_bytes_read, uint8_t* output,
size_t output_max_len, size_t* decompressed_len, bool* stream_end,
size_t* more_input_bytes, size_t* more_output_bytes) override;

std::string debug_info() override;

private:
friend class Decompressor;
Lz4BlockDecompressor() : Decompressor(CompressType::LZ4FRAME) {}
Status init() override;
};

class SnappyBlockDecompressor : public Decompressor {
public:
~SnappyBlockDecompressor() override {}

Status decompress(uint8_t* input, size_t input_len, size_t* input_bytes_read, uint8_t* output,
size_t output_max_len, size_t* decompressed_len, bool* stream_end,
size_t* more_input_bytes, size_t* more_output_bytes) override;

std::string debug_info() override;

private:
friend class Decompressor;
SnappyBlockDecompressor() : Decompressor(CompressType::SNAPPYBLOCK) {}
Status init() override;
};

#ifdef DORIS_WITH_LZO
class LzopDecompressor : public Decompressor {
public:
Expand Down
2 changes: 2 additions & 0 deletions be/src/service/internal_service.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -632,6 +632,8 @@ void PInternalServiceImpl::fetch_table_schema(google::protobuf::RpcController* c
case TFileFormatType::FORMAT_CSV_GZ:
case TFileFormatType::FORMAT_CSV_BZ2:
case TFileFormatType::FORMAT_CSV_LZ4FRAME:
case TFileFormatType::FORMAT_CSV_LZ4BLOCK:
case TFileFormatType::FORMAT_CSV_SNAPPYBLOCK:
case TFileFormatType::FORMAT_CSV_LZOP:
case TFileFormatType::FORMAT_CSV_DEFLATE: {
// file_slots is no use
Expand Down
9 changes: 8 additions & 1 deletion be/src/util/load_util.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -46,9 +46,15 @@ void LoadUtil::parse_format(const std::string& format_str, const std::string& co
} else if (iequal(compress_type_str, "LZ4")) {
*format_type = TFileFormatType::FORMAT_CSV_LZ4FRAME;
*compress_type = TFileCompressType::LZ4FRAME;
} else if (iequal(compress_type_str, "LZ4_BLOCK")) {
*format_type = TFileFormatType::FORMAT_CSV_LZ4BLOCK;
*compress_type = TFileCompressType::LZ4BLOCK;
} else if (iequal(compress_type_str, "LZOP")) {
*format_type = TFileFormatType::FORMAT_CSV_LZOP;
*compress_type = TFileCompressType::LZO;
} else if (iequal(compress_type_str, "SNAPPY_BLOCK")) {
*format_type = TFileFormatType::FORMAT_CSV_SNAPPYBLOCK;
*compress_type = TFileCompressType::SNAPPYBLOCK;
} else if (iequal(compress_type_str, "DEFLATE")) {
*format_type = TFileFormatType::FORMAT_CSV_DEFLATE;
*compress_type = TFileCompressType::DEFLATE;
Expand All @@ -72,6 +78,7 @@ bool LoadUtil::is_format_support_streaming(TFileFormatType::type format) {
case TFileFormatType::FORMAT_CSV_DEFLATE:
case TFileFormatType::FORMAT_CSV_GZ:
case TFileFormatType::FORMAT_CSV_LZ4FRAME:
case TFileFormatType::FORMAT_CSV_LZ4BLOCK:
case TFileFormatType::FORMAT_CSV_LZO:
case TFileFormatType::FORMAT_CSV_LZOP:
case TFileFormatType::FORMAT_JSON:
Expand All @@ -81,4 +88,4 @@ bool LoadUtil::is_format_support_streaming(TFileFormatType::type format) {
}
return false;
}
} // namespace doris
} // namespace doris
Loading