2016-02-10 00:12:00 +01:00
|
|
|
// Copyright (c) 2011-present, Facebook, Inc. All rights reserved.
|
2017-07-16 01:03:42 +02:00
|
|
|
// This source code is licensed under both the GPLv2 (found in the
|
|
|
|
// COPYING file in the root directory) and Apache 2.0 License
|
|
|
|
// (found in the LICENSE.Apache file in the root directory).
|
2014-11-08 02:23:58 +01:00
|
|
|
//
|
2014-11-12 22:05:12 +01:00
|
|
|
#ifndef ROCKSDB_LITE
|
2014-11-08 02:23:58 +01:00
|
|
|
|
2020-06-25 04:30:15 +02:00
|
|
|
#include "rocksdb/sst_dump_tool.h"
|
2014-11-08 02:23:58 +01:00
|
|
|
|
2019-06-06 22:52:39 +02:00
|
|
|
#include <cinttypes>
|
2016-05-07 01:09:09 +02:00
|
|
|
#include <iostream>
|
2016-01-13 03:20:06 +01:00
|
|
|
|
2015-09-01 03:35:12 +02:00
|
|
|
#include "port/port.h"
|
2020-06-25 04:30:15 +02:00
|
|
|
#include "rocksdb/utilities/ldb_cmd.h"
|
|
|
|
#include "table/sst_file_dumper.h"
|
2014-11-08 02:23:58 +01:00
|
|
|
|
2020-02-20 21:07:53 +01:00
|
|
|
namespace ROCKSDB_NAMESPACE {
|
2014-11-08 02:23:58 +01:00
|
|
|
|
2017-08-12 00:49:17 +02:00
|
|
|
static const std::vector<std::pair<CompressionType, const char*>>
|
|
|
|
kCompressions = {
|
|
|
|
{CompressionType::kNoCompression, "kNoCompression"},
|
|
|
|
{CompressionType::kSnappyCompression, "kSnappyCompression"},
|
|
|
|
{CompressionType::kZlibCompression, "kZlibCompression"},
|
|
|
|
{CompressionType::kBZip2Compression, "kBZip2Compression"},
|
|
|
|
{CompressionType::kLZ4Compression, "kLZ4Compression"},
|
|
|
|
{CompressionType::kLZ4HCCompression, "kLZ4HCCompression"},
|
|
|
|
{CompressionType::kXpressCompression, "kXpressCompression"},
|
|
|
|
{CompressionType::kZSTD, "kZSTD"}};
|
|
|
|
|
2014-11-08 02:23:58 +01:00
|
|
|
namespace {
|
|
|
|
|
2020-06-09 19:01:12 +02:00
|
|
|
void print_help(bool to_stderr) {
|
2019-09-19 21:32:33 +02:00
|
|
|
fprintf(
|
2020-06-09 19:01:12 +02:00
|
|
|
to_stderr ? stderr : stdout,
|
2020-06-08 22:56:22 +02:00
|
|
|
R"(sst_dump --file=<data_dir_OR_sst_file> [--command=check|scan|raw|recompress|identify]
|
2016-04-08 21:05:02 +02:00
|
|
|
--file=<data_dir_OR_sst_file>
|
|
|
|
Path to SST file or directory containing SST files
|
|
|
|
|
2019-10-09 04:17:39 +02:00
|
|
|
--env_uri=<uri of underlying Env>
|
2021-03-10 05:47:26 +01:00
|
|
|
URI of underlying Env, mutually exclusive with fs_uri
|
|
|
|
|
|
|
|
--fs_uri=<uri of underlying FileSystem>
|
|
|
|
URI of underlying FileSystem, mutually exclusive with env_uri
|
2019-10-09 04:17:39 +02:00
|
|
|
|
2020-06-08 22:56:22 +02:00
|
|
|
--command=check|scan|raw|verify|identify
|
2019-09-14 01:29:16 +02:00
|
|
|
check: Iterate over entries in files but don't print anything except if an error is encountered (default command)
|
2016-04-08 21:05:02 +02:00
|
|
|
scan: Iterate over entries in files and print them to screen
|
|
|
|
raw: Dump all the table contents to <file_name>_dump.txt
|
2019-09-14 01:29:16 +02:00
|
|
|
verify: Iterate all the blocks in files verifying checksum to detect possible corruption but don't print anything except if a corruption is encountered
|
2017-08-12 00:49:17 +02:00
|
|
|
recompress: reports the SST file size if recompressed with different
|
|
|
|
compression types
|
2020-06-08 22:56:22 +02:00
|
|
|
identify: Reports a file is a valid SST file or lists all valid SST files under a directory
|
2016-04-08 21:05:02 +02:00
|
|
|
|
|
|
|
--output_hex
|
|
|
|
Can be combined with scan command to print the keys and values in Hex
|
|
|
|
|
2019-10-18 04:35:22 +02:00
|
|
|
--decode_blob_index
|
|
|
|
Decode blob indexes and print them in a human-readable format during scans.
|
|
|
|
|
2016-04-08 21:05:02 +02:00
|
|
|
--from=<user_key>
|
|
|
|
Key to start reading from when executing check|scan
|
|
|
|
|
|
|
|
--to=<user_key>
|
|
|
|
Key to stop reading at when executing check|scan
|
|
|
|
|
2017-03-13 18:24:52 +01:00
|
|
|
--prefix=<user_key>
|
|
|
|
Returns all keys with this prefix when executing check|scan
|
|
|
|
Cannot be used in conjunction with --from
|
|
|
|
|
2016-04-08 21:05:02 +02:00
|
|
|
--read_num=<num>
|
|
|
|
Maximum number of entries to read when executing check|scan
|
|
|
|
|
|
|
|
--verify_checksum
|
|
|
|
Verify file checksum when executing check|scan
|
|
|
|
|
|
|
|
--input_key_hex
|
|
|
|
Can be combined with --from and --to to indicate that these values are encoded in Hex
|
|
|
|
|
|
|
|
--show_properties
|
2017-08-12 00:49:17 +02:00
|
|
|
Print table properties after iterating over the file when executing
|
2020-06-08 22:56:22 +02:00
|
|
|
check|scan|raw|identify
|
2016-04-08 21:05:02 +02:00
|
|
|
|
|
|
|
--set_block_size=<block_size>
|
2017-08-12 00:49:17 +02:00
|
|
|
Can be combined with --command=recompress to set the block size that will
|
|
|
|
be used when trying different compression algorithms
|
|
|
|
|
|
|
|
--compression_types=<comma-separated list of CompressionType members, e.g.,
|
|
|
|
kSnappyCompression>
|
|
|
|
Can be combined with --command=recompress to run recompression for this
|
|
|
|
list of compression types
|
2016-11-10 19:06:06 +01:00
|
|
|
|
|
|
|
--parse_internal_key=<0xKEY>
|
|
|
|
Convenience option to parse an internal key on the command line. Dumps the
|
|
|
|
internal key in hex format {'key' @ SN: type}
|
2020-04-27 21:33:49 +02:00
|
|
|
|
|
|
|
--compression_level_from=<compression_level>
|
|
|
|
Compression level to start compressing when executing recompress. One compression type
|
|
|
|
and compression_level_to must also be specified
|
|
|
|
|
|
|
|
--compression_level_to=<compression_level>
|
|
|
|
Compression level to stop compressing when executing recompress. One compression type
|
|
|
|
and compression_level_from must also be specified
|
2020-09-04 00:48:29 +02:00
|
|
|
|
|
|
|
--compression_max_dict_bytes=<uint32_t>
|
|
|
|
Maximum size of dictionary used to prime the compression library
|
|
|
|
|
|
|
|
--compression_zstd_max_train_bytes=<uint32_t>
|
|
|
|
Maximum size of training data passed to zstd's dictionary trainer
|
Limit buffering for collecting samples for compression dictionary (#7970)
Summary:
For dictionary compression, we need to collect some representative samples of the data to be compressed, which we use to either generate or train (when `CompressionOptions::zstd_max_train_bytes > 0`) a dictionary. Previously, the strategy was to buffer all the data blocks during flush, and up to the target file size during compaction. That strategy allowed us to randomly pick samples from as wide a range as possible that'd be guaranteed to land in a single output file.
However, some users try to make huge files in memory-constrained environments, where this strategy can cause OOM. This PR introduces an option, `CompressionOptions::max_dict_buffer_bytes`, that limits how much data blocks are buffered before we switch to unbuffered mode (which means creating the per-SST dictionary, writing out the buffered data, and compressing/writing new blocks as soon as they are built). It is not strict as we currently buffer more than just data blocks -- also keys are buffered. But it does make a step towards giving users predictable memory usage.
Related changes include:
- Changed sampling for dictionary compression to select unique data blocks when there is limited availability of data blocks
- Made use of `BlockBuilder::SwapAndReset()` to save an allocation+memcpy when buffering data blocks for building a dictionary
- Changed `ParseBoolean()` to accept an input containing characters after the boolean. This is necessary since, with this PR, a value for `CompressionOptions::enabled` is no longer necessarily the final component in the `CompressionOptions` string.
Pull Request resolved: https://github.com/facebook/rocksdb/pull/7970
Test Plan:
- updated `CompressionOptions` unit tests to verify limit is respected (to the extent expected in the current implementation) in various scenarios of flush/compaction to bottommost/non-bottommost level
- looked at jemalloc heap profiles right before and after switching to unbuffered mode during flush/compaction. Verified memory usage in buffering is proportional to the limit set.
Reviewed By: pdillinger
Differential Revision: D26467994
Pulled By: ajkr
fbshipit-source-id: 3da4ef9fba59974e4ef40e40c01611002c861465
2021-02-19 23:06:59 +01:00
|
|
|
|
|
|
|
--compression_max_dict_buffer_bytes=<int64_t>
|
|
|
|
Limit on buffer size from which we collect samples for dictionary generation.
|
2016-04-08 21:05:02 +02:00
|
|
|
)");
|
2014-11-08 02:23:58 +01:00
|
|
|
}
|
|
|
|
|
2020-05-13 03:21:32 +02:00
|
|
|
// arg_name would include all prefix, e.g. "--my_arg="
|
|
|
|
// arg_val is the parses value.
|
|
|
|
// True if there is a match. False otherwise.
|
|
|
|
// Woud exit after printing errmsg if cannot be parsed.
|
|
|
|
bool ParseIntArg(const char* arg, const std::string arg_name,
|
|
|
|
const std::string err_msg, int64_t* arg_val) {
|
|
|
|
if (strncmp(arg, arg_name.c_str(), arg_name.size()) == 0) {
|
|
|
|
std::string input_str = arg + arg_name.size();
|
|
|
|
std::istringstream iss(input_str);
|
|
|
|
iss >> *arg_val;
|
|
|
|
if (iss.fail()) {
|
|
|
|
fprintf(stderr, "%s\n", err_msg.c_str());
|
|
|
|
exit(1);
|
|
|
|
}
|
|
|
|
return true;
|
|
|
|
}
|
|
|
|
return false;
|
|
|
|
}
|
2014-11-08 02:23:58 +01:00
|
|
|
} // namespace
|
|
|
|
|
2020-06-09 19:01:12 +02:00
|
|
|
int SSTDumpTool::Run(int argc, char const* const* argv, Options options) {
|
2021-06-15 12:42:52 +02:00
|
|
|
std::string env_uri, fs_uri;
|
2014-11-08 02:23:58 +01:00
|
|
|
const char* dir_or_file = nullptr;
|
2017-10-19 19:48:47 +02:00
|
|
|
uint64_t read_num = std::numeric_limits<uint64_t>::max();
|
2014-11-08 02:23:58 +01:00
|
|
|
std::string command;
|
|
|
|
|
|
|
|
char junk;
|
|
|
|
uint64_t n;
|
|
|
|
bool verify_checksum = false;
|
|
|
|
bool output_hex = false;
|
2019-10-18 04:35:22 +02:00
|
|
|
bool decode_blob_index = false;
|
2014-11-08 02:23:58 +01:00
|
|
|
bool input_key_hex = false;
|
|
|
|
bool has_from = false;
|
|
|
|
bool has_to = false;
|
2017-03-13 18:24:52 +01:00
|
|
|
bool use_from_as_prefix = false;
|
2014-11-08 02:23:58 +01:00
|
|
|
bool show_properties = false;
|
2017-01-04 03:24:15 +01:00
|
|
|
bool show_summary = false;
|
2015-07-24 02:05:33 +02:00
|
|
|
bool set_block_size = false;
|
2020-04-27 21:33:49 +02:00
|
|
|
bool has_compression_level_from = false;
|
|
|
|
bool has_compression_level_to = false;
|
|
|
|
bool has_specified_compression_types = false;
|
2014-11-08 02:23:58 +01:00
|
|
|
std::string from_key;
|
|
|
|
std::string to_key;
|
2015-07-24 02:05:33 +02:00
|
|
|
std::string block_size_str;
|
2020-04-27 21:33:49 +02:00
|
|
|
std::string compression_level_from_str;
|
|
|
|
std::string compression_level_to_str;
|
2017-10-19 19:48:47 +02:00
|
|
|
size_t block_size = 0;
|
2020-05-13 03:21:32 +02:00
|
|
|
size_t readahead_size = 2 * 1024 * 1024;
|
2017-08-12 00:49:17 +02:00
|
|
|
std::vector<std::pair<CompressionType, const char*>> compression_types;
|
2017-01-04 03:24:15 +01:00
|
|
|
uint64_t total_num_files = 0;
|
|
|
|
uint64_t total_num_data_blocks = 0;
|
|
|
|
uint64_t total_data_block_size = 0;
|
|
|
|
uint64_t total_index_block_size = 0;
|
|
|
|
uint64_t total_filter_block_size = 0;
|
2020-04-27 21:33:49 +02:00
|
|
|
int32_t compress_level_from = CompressionOptions::kDefaultCompressionLevel;
|
|
|
|
int32_t compress_level_to = CompressionOptions::kDefaultCompressionLevel;
|
2020-09-04 00:48:29 +02:00
|
|
|
uint32_t compression_max_dict_bytes =
|
|
|
|
ROCKSDB_NAMESPACE::CompressionOptions().max_dict_bytes;
|
|
|
|
uint32_t compression_zstd_max_train_bytes =
|
|
|
|
ROCKSDB_NAMESPACE::CompressionOptions().zstd_max_train_bytes;
|
Limit buffering for collecting samples for compression dictionary (#7970)
Summary:
For dictionary compression, we need to collect some representative samples of the data to be compressed, which we use to either generate or train (when `CompressionOptions::zstd_max_train_bytes > 0`) a dictionary. Previously, the strategy was to buffer all the data blocks during flush, and up to the target file size during compaction. That strategy allowed us to randomly pick samples from as wide a range as possible that'd be guaranteed to land in a single output file.
However, some users try to make huge files in memory-constrained environments, where this strategy can cause OOM. This PR introduces an option, `CompressionOptions::max_dict_buffer_bytes`, that limits how much data blocks are buffered before we switch to unbuffered mode (which means creating the per-SST dictionary, writing out the buffered data, and compressing/writing new blocks as soon as they are built). It is not strict as we currently buffer more than just data blocks -- also keys are buffered. But it does make a step towards giving users predictable memory usage.
Related changes include:
- Changed sampling for dictionary compression to select unique data blocks when there is limited availability of data blocks
- Made use of `BlockBuilder::SwapAndReset()` to save an allocation+memcpy when buffering data blocks for building a dictionary
- Changed `ParseBoolean()` to accept an input containing characters after the boolean. This is necessary since, with this PR, a value for `CompressionOptions::enabled` is no longer necessarily the final component in the `CompressionOptions` string.
Pull Request resolved: https://github.com/facebook/rocksdb/pull/7970
Test Plan:
- updated `CompressionOptions` unit tests to verify limit is respected (to the extent expected in the current implementation) in various scenarios of flush/compaction to bottommost/non-bottommost level
- looked at jemalloc heap profiles right before and after switching to unbuffered mode during flush/compaction. Verified memory usage in buffering is proportional to the limit set.
Reviewed By: pdillinger
Differential Revision: D26467994
Pulled By: ajkr
fbshipit-source-id: 3da4ef9fba59974e4ef40e40c01611002c861465
2021-02-19 23:06:59 +01:00
|
|
|
uint64_t compression_max_dict_buffer_bytes =
|
|
|
|
ROCKSDB_NAMESPACE::CompressionOptions().max_dict_buffer_bytes;
|
2020-05-13 03:21:32 +02:00
|
|
|
|
|
|
|
int64_t tmp_val;
|
|
|
|
|
2014-11-08 02:23:58 +01:00
|
|
|
for (int i = 1; i < argc; i++) {
|
2019-10-09 04:17:39 +02:00
|
|
|
if (strncmp(argv[i], "--env_uri=", 10) == 0) {
|
|
|
|
env_uri = argv[i] + 10;
|
2021-03-10 05:47:26 +01:00
|
|
|
} else if (strncmp(argv[i], "--fs_uri=", 9) == 0) {
|
|
|
|
fs_uri = argv[i] + 9;
|
2019-10-09 04:17:39 +02:00
|
|
|
} else if (strncmp(argv[i], "--file=", 7) == 0) {
|
2014-11-08 02:23:58 +01:00
|
|
|
dir_or_file = argv[i] + 7;
|
|
|
|
} else if (strcmp(argv[i], "--output_hex") == 0) {
|
|
|
|
output_hex = true;
|
2019-10-18 04:35:22 +02:00
|
|
|
} else if (strcmp(argv[i], "--decode_blob_index") == 0) {
|
|
|
|
decode_blob_index = true;
|
2014-11-08 02:23:58 +01:00
|
|
|
} else if (strcmp(argv[i], "--input_key_hex") == 0) {
|
|
|
|
input_key_hex = true;
|
2019-10-09 04:17:39 +02:00
|
|
|
} else if (sscanf(argv[i], "--read_num=%lu%c", (unsigned long*)&n, &junk) ==
|
|
|
|
1) {
|
2014-11-08 02:23:58 +01:00
|
|
|
read_num = n;
|
|
|
|
} else if (strcmp(argv[i], "--verify_checksum") == 0) {
|
|
|
|
verify_checksum = true;
|
|
|
|
} else if (strncmp(argv[i], "--command=", 10) == 0) {
|
|
|
|
command = argv[i] + 10;
|
|
|
|
} else if (strncmp(argv[i], "--from=", 7) == 0) {
|
|
|
|
from_key = argv[i] + 7;
|
|
|
|
has_from = true;
|
|
|
|
} else if (strncmp(argv[i], "--to=", 5) == 0) {
|
|
|
|
to_key = argv[i] + 5;
|
|
|
|
has_to = true;
|
2017-03-13 18:24:52 +01:00
|
|
|
} else if (strncmp(argv[i], "--prefix=", 9) == 0) {
|
|
|
|
from_key = argv[i] + 9;
|
|
|
|
use_from_as_prefix = true;
|
2014-11-08 02:23:58 +01:00
|
|
|
} else if (strcmp(argv[i], "--show_properties") == 0) {
|
|
|
|
show_properties = true;
|
2017-01-04 03:24:15 +01:00
|
|
|
} else if (strcmp(argv[i], "--show_summary") == 0) {
|
|
|
|
show_summary = true;
|
2020-05-13 03:21:32 +02:00
|
|
|
} else if (ParseIntArg(argv[i], "--set_block_size=",
|
|
|
|
"block size must be numeric", &tmp_val)) {
|
2015-07-24 02:05:33 +02:00
|
|
|
set_block_size = true;
|
2020-05-13 03:21:32 +02:00
|
|
|
block_size = static_cast<size_t>(tmp_val);
|
|
|
|
} else if (ParseIntArg(argv[i], "--readahead_size=",
|
|
|
|
"readahead_size must be numeric", &tmp_val)) {
|
|
|
|
readahead_size = static_cast<size_t>(tmp_val);
|
2017-08-12 00:49:17 +02:00
|
|
|
} else if (strncmp(argv[i], "--compression_types=", 20) == 0) {
|
|
|
|
std::string compression_types_csv = argv[i] + 20;
|
|
|
|
std::istringstream iss(compression_types_csv);
|
|
|
|
std::string compression_type;
|
2020-04-27 21:33:49 +02:00
|
|
|
has_specified_compression_types = true;
|
2017-08-12 00:49:17 +02:00
|
|
|
while (std::getline(iss, compression_type, ',')) {
|
|
|
|
auto iter = std::find_if(
|
|
|
|
kCompressions.begin(), kCompressions.end(),
|
|
|
|
[&compression_type](std::pair<CompressionType, const char*> curr) {
|
|
|
|
return curr.second == compression_type;
|
|
|
|
});
|
|
|
|
if (iter == kCompressions.end()) {
|
|
|
|
fprintf(stderr, "%s is not a valid CompressionType\n",
|
|
|
|
compression_type.c_str());
|
|
|
|
exit(1);
|
|
|
|
}
|
|
|
|
compression_types.emplace_back(*iter);
|
|
|
|
}
|
2016-11-10 19:06:06 +01:00
|
|
|
} else if (strncmp(argv[i], "--parse_internal_key=", 21) == 0) {
|
|
|
|
std::string in_key(argv[i] + 21);
|
|
|
|
try {
|
2020-02-20 21:07:53 +01:00
|
|
|
in_key = ROCKSDB_NAMESPACE::LDBCommand::HexToString(in_key);
|
2016-11-10 19:06:06 +01:00
|
|
|
} catch (...) {
|
|
|
|
std::cerr << "ERROR: Invalid key input '"
|
|
|
|
<< in_key
|
|
|
|
<< "' Use 0x{hex representation of internal rocksdb key}" << std::endl;
|
|
|
|
return -1;
|
|
|
|
}
|
2020-02-20 21:07:53 +01:00
|
|
|
Slice sl_key = ROCKSDB_NAMESPACE::Slice(in_key);
|
2016-11-10 19:06:06 +01:00
|
|
|
ParsedInternalKey ikey;
|
|
|
|
int retc = 0;
|
2020-10-28 18:11:13 +01:00
|
|
|
Status pik_status =
|
|
|
|
ParseInternalKey(sl_key, &ikey, true /* log_err_key */);
|
|
|
|
if (!pik_status.ok()) {
|
|
|
|
std::cerr << pik_status.getState() << "\n";
|
2016-11-10 19:06:06 +01:00
|
|
|
retc = -1;
|
|
|
|
}
|
2020-10-28 18:11:13 +01:00
|
|
|
fprintf(stdout, "key=%s\n", ikey.DebugString(true, true).c_str());
|
2016-11-10 19:06:06 +01:00
|
|
|
return retc;
|
2020-05-13 03:21:32 +02:00
|
|
|
} else if (ParseIntArg(argv[i], "--compression_level_from=",
|
|
|
|
"compression_level_from must be numeric",
|
|
|
|
&tmp_val)) {
|
2020-04-27 21:33:49 +02:00
|
|
|
has_compression_level_from = true;
|
2020-05-13 03:21:32 +02:00
|
|
|
compress_level_from = static_cast<int>(tmp_val);
|
|
|
|
} else if (ParseIntArg(argv[i], "--compression_level_to=",
|
|
|
|
"compression_level_to must be numeric", &tmp_val)) {
|
2020-04-27 21:33:49 +02:00
|
|
|
has_compression_level_to = true;
|
2020-05-13 03:21:32 +02:00
|
|
|
compress_level_to = static_cast<int>(tmp_val);
|
2020-09-04 00:48:29 +02:00
|
|
|
} else if (ParseIntArg(argv[i], "--compression_max_dict_bytes=",
|
|
|
|
"compression_max_dict_bytes must be numeric",
|
|
|
|
&tmp_val)) {
|
|
|
|
if (tmp_val < 0 || tmp_val > port::kMaxUint32) {
|
|
|
|
fprintf(stderr, "compression_max_dict_bytes must be a uint32_t: '%s'\n",
|
|
|
|
argv[i]);
|
|
|
|
print_help(/*to_stderr*/ true);
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
compression_max_dict_bytes = static_cast<uint32_t>(tmp_val);
|
|
|
|
} else if (ParseIntArg(argv[i], "--compression_zstd_max_train_bytes=",
|
|
|
|
"compression_zstd_max_train_bytes must be numeric",
|
|
|
|
&tmp_val)) {
|
|
|
|
if (tmp_val < 0 || tmp_val > port::kMaxUint32) {
|
|
|
|
fprintf(stderr,
|
|
|
|
"compression_zstd_max_train_bytes must be a uint32_t: '%s'\n",
|
|
|
|
argv[i]);
|
|
|
|
print_help(/*to_stderr*/ true);
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
compression_zstd_max_train_bytes = static_cast<uint32_t>(tmp_val);
|
Limit buffering for collecting samples for compression dictionary (#7970)
Summary:
For dictionary compression, we need to collect some representative samples of the data to be compressed, which we use to either generate or train (when `CompressionOptions::zstd_max_train_bytes > 0`) a dictionary. Previously, the strategy was to buffer all the data blocks during flush, and up to the target file size during compaction. That strategy allowed us to randomly pick samples from as wide a range as possible that'd be guaranteed to land in a single output file.
However, some users try to make huge files in memory-constrained environments, where this strategy can cause OOM. This PR introduces an option, `CompressionOptions::max_dict_buffer_bytes`, that limits how much data blocks are buffered before we switch to unbuffered mode (which means creating the per-SST dictionary, writing out the buffered data, and compressing/writing new blocks as soon as they are built). It is not strict as we currently buffer more than just data blocks -- also keys are buffered. But it does make a step towards giving users predictable memory usage.
Related changes include:
- Changed sampling for dictionary compression to select unique data blocks when there is limited availability of data blocks
- Made use of `BlockBuilder::SwapAndReset()` to save an allocation+memcpy when buffering data blocks for building a dictionary
- Changed `ParseBoolean()` to accept an input containing characters after the boolean. This is necessary since, with this PR, a value for `CompressionOptions::enabled` is no longer necessarily the final component in the `CompressionOptions` string.
Pull Request resolved: https://github.com/facebook/rocksdb/pull/7970
Test Plan:
- updated `CompressionOptions` unit tests to verify limit is respected (to the extent expected in the current implementation) in various scenarios of flush/compaction to bottommost/non-bottommost level
- looked at jemalloc heap profiles right before and after switching to unbuffered mode during flush/compaction. Verified memory usage in buffering is proportional to the limit set.
Reviewed By: pdillinger
Differential Revision: D26467994
Pulled By: ajkr
fbshipit-source-id: 3da4ef9fba59974e4ef40e40c01611002c861465
2021-02-19 23:06:59 +01:00
|
|
|
} else if (ParseIntArg(argv[i], "--compression_max_dict_buffer_bytes=",
|
|
|
|
"compression_max_dict_buffer_bytes must be numeric",
|
|
|
|
&tmp_val)) {
|
|
|
|
if (tmp_val < 0) {
|
|
|
|
fprintf(stderr,
|
|
|
|
"compression_max_dict_buffer_bytes must be positive: '%s'\n",
|
|
|
|
argv[i]);
|
|
|
|
print_help(/*to_stderr*/ true);
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
compression_max_dict_buffer_bytes = static_cast<uint64_t>(tmp_val);
|
2020-06-09 19:01:12 +02:00
|
|
|
} else if (strcmp(argv[i], "--help") == 0) {
|
|
|
|
print_help(/*to_stderr*/ false);
|
|
|
|
return 0;
|
|
|
|
} else if (strcmp(argv[i], "--version") == 0) {
|
2021-01-29 02:40:24 +01:00
|
|
|
printf("%s\n", GetRocksBuildInfoAsString("sst_dump").c_str());
|
2020-06-09 19:01:12 +02:00
|
|
|
return 0;
|
2020-05-13 03:21:32 +02:00
|
|
|
} else {
|
2017-03-13 18:24:52 +01:00
|
|
|
fprintf(stderr, "Unrecognized argument '%s'\n\n", argv[i]);
|
2020-06-09 19:01:12 +02:00
|
|
|
print_help(/*to_stderr*/ true);
|
|
|
|
return 1;
|
2014-11-08 02:23:58 +01:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2020-04-27 21:33:49 +02:00
|
|
|
if(has_compression_level_from && has_compression_level_to) {
|
|
|
|
if(!has_specified_compression_types || compression_types.size() != 1) {
|
|
|
|
fprintf(stderr, "Specify one compression type.\n\n");
|
|
|
|
exit(1);
|
|
|
|
}
|
|
|
|
} else if(has_compression_level_from || has_compression_level_to) {
|
|
|
|
fprintf(stderr, "Specify both --compression_level_from and "
|
|
|
|
"--compression_level_to.\n\n");
|
|
|
|
exit(1);
|
|
|
|
}
|
|
|
|
|
2017-03-13 18:24:52 +01:00
|
|
|
if (use_from_as_prefix && has_from) {
|
|
|
|
fprintf(stderr, "Cannot specify --prefix and --from\n\n");
|
|
|
|
exit(1);
|
|
|
|
}
|
|
|
|
|
2014-11-08 02:23:58 +01:00
|
|
|
if (input_key_hex) {
|
2017-03-13 18:24:52 +01:00
|
|
|
if (has_from || use_from_as_prefix) {
|
2020-02-20 21:07:53 +01:00
|
|
|
from_key = ROCKSDB_NAMESPACE::LDBCommand::HexToString(from_key);
|
2014-11-08 02:23:58 +01:00
|
|
|
}
|
|
|
|
if (has_to) {
|
2020-02-20 21:07:53 +01:00
|
|
|
to_key = ROCKSDB_NAMESPACE::LDBCommand::HexToString(to_key);
|
2014-11-08 02:23:58 +01:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
if (dir_or_file == nullptr) {
|
2017-03-13 18:24:52 +01:00
|
|
|
fprintf(stderr, "file or directory must be specified.\n\n");
|
2020-06-09 19:01:12 +02:00
|
|
|
print_help(/*to_stderr*/ true);
|
2014-11-08 02:23:58 +01:00
|
|
|
exit(1);
|
|
|
|
}
|
|
|
|
|
2020-02-20 21:07:53 +01:00
|
|
|
std::shared_ptr<ROCKSDB_NAMESPACE::Env> env_guard;
|
2019-10-09 04:17:39 +02:00
|
|
|
|
|
|
|
// If caller of SSTDumpTool::Run(...) does not specify a different env other
|
2021-03-10 05:47:26 +01:00
|
|
|
// than Env::Default(), then try to load custom env based on env_uri/fs_uri.
|
2019-10-09 04:17:39 +02:00
|
|
|
// Otherwise, the caller is responsible for creating custom env.
|
2021-06-15 12:42:52 +02:00
|
|
|
{
|
|
|
|
ConfigOptions config_options;
|
|
|
|
config_options.env = options.env;
|
|
|
|
Status s = Env::CreateFromUri(config_options, env_uri, fs_uri, &options.env,
|
|
|
|
&env_guard);
|
|
|
|
if (!s.ok()) {
|
|
|
|
fprintf(stderr, "CreateEnvFromUri: %s\n", s.ToString().c_str());
|
|
|
|
exit(1);
|
|
|
|
} else {
|
|
|
|
fprintf(stdout, "options.env is %p\n", options.env);
|
2019-10-09 04:17:39 +02:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2014-11-08 02:23:58 +01:00
|
|
|
std::vector<std::string> filenames;
|
2020-02-20 21:07:53 +01:00
|
|
|
ROCKSDB_NAMESPACE::Env* env = options.env;
|
|
|
|
ROCKSDB_NAMESPACE::Status st = env->GetChildren(dir_or_file, &filenames);
|
2014-11-08 02:23:58 +01:00
|
|
|
bool dir = true;
|
2020-06-08 22:56:22 +02:00
|
|
|
if (!st.ok() || filenames.empty()) {
|
|
|
|
// dir_or_file does not exist or does not contain children
|
|
|
|
// Check its existence first
|
|
|
|
Status s = env->FileExists(dir_or_file);
|
|
|
|
// dir_or_file does not exist
|
|
|
|
if (!s.ok()) {
|
|
|
|
fprintf(stderr, "%s%s: No such file or directory\n", s.ToString().c_str(),
|
|
|
|
dir_or_file);
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
// dir_or_file exists and is treated as a "file"
|
|
|
|
// since it has no children
|
|
|
|
// This is ok since later it will be checked
|
|
|
|
// that whether it is a valid sst or not
|
|
|
|
// (A directory "file" is not a valid sst)
|
2014-11-08 02:23:58 +01:00
|
|
|
filenames.clear();
|
|
|
|
filenames.push_back(dir_or_file);
|
|
|
|
dir = false;
|
|
|
|
}
|
|
|
|
|
|
|
|
uint64_t total_read = 0;
|
2020-06-08 22:56:22 +02:00
|
|
|
// List of RocksDB SST file without corruption
|
|
|
|
std::vector<std::string> valid_sst_files;
|
2014-11-08 02:23:58 +01:00
|
|
|
for (size_t i = 0; i < filenames.size(); i++) {
|
|
|
|
std::string filename = filenames.at(i);
|
|
|
|
if (filename.length() <= 4 ||
|
|
|
|
filename.rfind(".sst") != filename.length() - 4) {
|
|
|
|
// ignore
|
|
|
|
continue;
|
|
|
|
}
|
2020-06-08 22:56:22 +02:00
|
|
|
|
2014-11-08 02:23:58 +01:00
|
|
|
if (dir) {
|
|
|
|
filename = std::string(dir_or_file) + "/" + filename;
|
|
|
|
}
|
2014-12-23 22:24:07 +01:00
|
|
|
|
New backup meta schema, with file temperatures (#9660)
Summary:
The primary goal of this change is to add support for backing up and
restoring (applying on restore) file temperature metadata, without
committing to either the DB manifest or the FS reported "current"
temperatures being exclusive "source of truth".
To achieve this goal, we need to add temperature information to backup
metadata, which requires updated backup meta schema. Fortunately I
prepared for this in https://github.com/facebook/rocksdb/issues/8069, which began forward compatibility in version
6.19.0 for this kind of schema update. (Previously, backup meta schema
was not extensible! Making this schema update public will allow some
other "nice to have" features like taking backups with hard links, and
avoiding crc32c checksum computation when another checksum is already
available.) While schema version 2 is newly public, the default schema
version is still 1. Until we change the default, users will need to set
to 2 to enable features like temperature data backup+restore. New
metadata like temperature information will be ignored with a warning
in versions before this change and since 6.19.0. The metadata is
considered ignorable because a functioning DB can be restored without
it.
Some detail:
* Some renaming because "future schema" is now just public schema 2.
* Initialize some atomics in TestFs (linter reported)
* Add temperature hint support to SstFileDumper (used by BackupEngine)
Pull Request resolved: https://github.com/facebook/rocksdb/pull/9660
Test Plan:
related unit test majorly updated for the new functionality,
including some shared testing support for tracking temperatures in a FS.
Some other tests and testing hooks into production code also updated for
making the backup meta schema change public.
Reviewed By: ajkr
Differential Revision: D34686968
Pulled By: pdillinger
fbshipit-source-id: 3ac1fa3e67ee97ca8a5103d79cc87d872c1d862a
2022-03-18 19:06:17 +01:00
|
|
|
ROCKSDB_NAMESPACE::SstFileDumper dumper(
|
|
|
|
options, filename, Temperature::kUnknown, readahead_size,
|
|
|
|
verify_checksum, output_hex, decode_blob_index);
|
2020-06-08 22:56:22 +02:00
|
|
|
// Not a valid SST
|
2018-11-27 21:59:27 +01:00
|
|
|
if (!dumper.getStatus().ok()) {
|
2014-12-23 22:24:07 +01:00
|
|
|
fprintf(stderr, "%s: %s\n", filename.c_str(),
|
2018-11-27 21:59:27 +01:00
|
|
|
dumper.getStatus().ToString().c_str());
|
2017-01-04 03:24:15 +01:00
|
|
|
continue;
|
2020-06-08 22:56:22 +02:00
|
|
|
} else {
|
|
|
|
valid_sst_files.push_back(filename);
|
|
|
|
// Print out from and to key information once
|
|
|
|
// where there is at least one valid SST
|
|
|
|
if (valid_sst_files.size() == 1) {
|
|
|
|
// from_key and to_key are only used for "check", "scan", or ""
|
|
|
|
if (command == "check" || command == "scan" || command == "") {
|
|
|
|
fprintf(stdout, "from [%s] to [%s]\n",
|
|
|
|
ROCKSDB_NAMESPACE::Slice(from_key).ToString(true).c_str(),
|
|
|
|
ROCKSDB_NAMESPACE::Slice(to_key).ToString(true).c_str());
|
|
|
|
}
|
|
|
|
}
|
2014-12-23 22:24:07 +01:00
|
|
|
}
|
|
|
|
|
2017-08-12 00:49:17 +02:00
|
|
|
if (command == "recompress") {
|
2020-09-05 04:25:20 +02:00
|
|
|
st = dumper.ShowAllCompressionSizes(
|
2017-08-12 00:49:17 +02:00
|
|
|
set_block_size ? block_size : 16384,
|
2020-04-27 21:33:49 +02:00
|
|
|
compression_types.empty() ? kCompressions : compression_types,
|
2020-09-04 00:48:29 +02:00
|
|
|
compress_level_from, compress_level_to, compression_max_dict_bytes,
|
Limit buffering for collecting samples for compression dictionary (#7970)
Summary:
For dictionary compression, we need to collect some representative samples of the data to be compressed, which we use to either generate or train (when `CompressionOptions::zstd_max_train_bytes > 0`) a dictionary. Previously, the strategy was to buffer all the data blocks during flush, and up to the target file size during compaction. That strategy allowed us to randomly pick samples from as wide a range as possible that'd be guaranteed to land in a single output file.
However, some users try to make huge files in memory-constrained environments, where this strategy can cause OOM. This PR introduces an option, `CompressionOptions::max_dict_buffer_bytes`, that limits how much data blocks are buffered before we switch to unbuffered mode (which means creating the per-SST dictionary, writing out the buffered data, and compressing/writing new blocks as soon as they are built). It is not strict as we currently buffer more than just data blocks -- also keys are buffered. But it does make a step towards giving users predictable memory usage.
Related changes include:
- Changed sampling for dictionary compression to select unique data blocks when there is limited availability of data blocks
- Made use of `BlockBuilder::SwapAndReset()` to save an allocation+memcpy when buffering data blocks for building a dictionary
- Changed `ParseBoolean()` to accept an input containing characters after the boolean. This is necessary since, with this PR, a value for `CompressionOptions::enabled` is no longer necessarily the final component in the `CompressionOptions` string.
Pull Request resolved: https://github.com/facebook/rocksdb/pull/7970
Test Plan:
- updated `CompressionOptions` unit tests to verify limit is respected (to the extent expected in the current implementation) in various scenarios of flush/compaction to bottommost/non-bottommost level
- looked at jemalloc heap profiles right before and after switching to unbuffered mode during flush/compaction. Verified memory usage in buffering is proportional to the limit set.
Reviewed By: pdillinger
Differential Revision: D26467994
Pulled By: ajkr
fbshipit-source-id: 3da4ef9fba59974e4ef40e40c01611002c861465
2021-02-19 23:06:59 +01:00
|
|
|
compression_zstd_max_train_bytes, compression_max_dict_buffer_bytes);
|
2020-09-05 04:25:20 +02:00
|
|
|
if (!st.ok()) {
|
|
|
|
fprintf(stderr, "Failed to recompress: %s\n", st.ToString().c_str());
|
|
|
|
exit(1);
|
|
|
|
}
|
2015-07-24 02:05:33 +02:00
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
|
2014-12-23 22:24:07 +01:00
|
|
|
if (command == "raw") {
|
|
|
|
std::string out_filename = filename.substr(0, filename.length() - 4);
|
|
|
|
out_filename.append("_dump.txt");
|
|
|
|
|
2018-11-27 21:59:27 +01:00
|
|
|
st = dumper.DumpTable(out_filename);
|
2014-12-23 22:24:07 +01:00
|
|
|
if (!st.ok()) {
|
|
|
|
fprintf(stderr, "%s: %s\n", filename.c_str(), st.ToString().c_str());
|
|
|
|
exit(1);
|
|
|
|
} else {
|
|
|
|
fprintf(stdout, "raw dump written to file %s\n", &out_filename[0]);
|
|
|
|
}
|
|
|
|
continue;
|
|
|
|
}
|
|
|
|
|
2014-11-08 02:23:58 +01:00
|
|
|
// scan all files in give file path.
|
|
|
|
if (command == "" || command == "scan" || command == "check") {
|
2018-11-27 21:59:27 +01:00
|
|
|
st = dumper.ReadSequential(
|
2017-03-13 18:24:52 +01:00
|
|
|
command == "scan", read_num > 0 ? (read_num - total_read) : read_num,
|
|
|
|
has_from || use_from_as_prefix, from_key, has_to, to_key,
|
|
|
|
use_from_as_prefix);
|
2014-11-08 02:23:58 +01:00
|
|
|
if (!st.ok()) {
|
|
|
|
fprintf(stderr, "%s: %s\n", filename.c_str(),
|
|
|
|
st.ToString().c_str());
|
|
|
|
}
|
2018-11-27 21:59:27 +01:00
|
|
|
total_read += dumper.GetReadNumber();
|
2014-11-08 02:23:58 +01:00
|
|
|
if (read_num > 0 && total_read > read_num) {
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
}
|
2017-01-04 03:24:15 +01:00
|
|
|
|
2017-08-10 00:49:40 +02:00
|
|
|
if (command == "verify") {
|
2018-11-27 21:59:27 +01:00
|
|
|
st = dumper.VerifyChecksum();
|
2017-08-10 00:49:40 +02:00
|
|
|
if (!st.ok()) {
|
|
|
|
fprintf(stderr, "%s is corrupted: %s\n", filename.c_str(),
|
|
|
|
st.ToString().c_str());
|
|
|
|
} else {
|
|
|
|
fprintf(stdout, "The file is ok\n");
|
|
|
|
}
|
|
|
|
continue;
|
|
|
|
}
|
|
|
|
|
2017-01-04 03:24:15 +01:00
|
|
|
if (show_properties || show_summary) {
|
2020-02-20 21:07:53 +01:00
|
|
|
const ROCKSDB_NAMESPACE::TableProperties* table_properties;
|
2014-11-08 02:23:58 +01:00
|
|
|
|
2020-02-20 21:07:53 +01:00
|
|
|
std::shared_ptr<const ROCKSDB_NAMESPACE::TableProperties>
|
2014-11-08 02:23:58 +01:00
|
|
|
table_properties_from_reader;
|
2018-11-27 21:59:27 +01:00
|
|
|
st = dumper.ReadTableProperties(&table_properties_from_reader);
|
2014-11-08 02:23:58 +01:00
|
|
|
if (!st.ok()) {
|
|
|
|
fprintf(stderr, "%s: %s\n", filename.c_str(), st.ToString().c_str());
|
|
|
|
fprintf(stderr, "Try to use initial table properties\n");
|
2018-11-27 21:59:27 +01:00
|
|
|
table_properties = dumper.GetInitTableProperties();
|
2014-11-08 02:23:58 +01:00
|
|
|
} else {
|
|
|
|
table_properties = table_properties_from_reader.get();
|
|
|
|
}
|
|
|
|
if (table_properties != nullptr) {
|
2017-01-04 03:24:15 +01:00
|
|
|
if (show_properties) {
|
|
|
|
fprintf(stdout,
|
|
|
|
"Table Properties:\n"
|
|
|
|
"------------------------------\n"
|
|
|
|
" %s",
|
|
|
|
table_properties->ToString("\n ", ": ").c_str());
|
2016-05-19 23:24:48 +02:00
|
|
|
}
|
2017-01-04 03:24:15 +01:00
|
|
|
total_num_files += 1;
|
|
|
|
total_num_data_blocks += table_properties->num_data_blocks;
|
|
|
|
total_data_block_size += table_properties->data_size;
|
|
|
|
total_index_block_size += table_properties->index_size;
|
|
|
|
total_filter_block_size += table_properties->filter_size;
|
2019-10-18 23:43:17 +02:00
|
|
|
if (show_properties) {
|
|
|
|
fprintf(stdout,
|
|
|
|
"Raw user collected properties\n"
|
|
|
|
"------------------------------\n");
|
|
|
|
for (const auto& kv : table_properties->user_collected_properties) {
|
|
|
|
std::string prop_name = kv.first;
|
|
|
|
std::string prop_val = Slice(kv.second).ToString(true);
|
|
|
|
fprintf(stdout, " # %s: 0x%s\n", prop_name.c_str(),
|
|
|
|
prop_val.c_str());
|
|
|
|
}
|
2017-01-04 03:24:15 +01:00
|
|
|
}
|
2019-10-18 23:43:17 +02:00
|
|
|
} else {
|
|
|
|
fprintf(stderr, "Reader unexpectedly returned null properties\n");
|
2016-12-14 20:09:50 +01:00
|
|
|
}
|
2014-11-08 02:23:58 +01:00
|
|
|
}
|
|
|
|
}
|
2017-01-04 03:24:15 +01:00
|
|
|
if (show_summary) {
|
|
|
|
fprintf(stdout, "total number of files: %" PRIu64 "\n", total_num_files);
|
|
|
|
fprintf(stdout, "total number of data blocks: %" PRIu64 "\n",
|
|
|
|
total_num_data_blocks);
|
|
|
|
fprintf(stdout, "total data block size: %" PRIu64 "\n",
|
|
|
|
total_data_block_size);
|
|
|
|
fprintf(stdout, "total index block size: %" PRIu64 "\n",
|
|
|
|
total_index_block_size);
|
|
|
|
fprintf(stdout, "total filter block size: %" PRIu64 "\n",
|
|
|
|
total_filter_block_size);
|
|
|
|
}
|
2020-06-08 22:56:22 +02:00
|
|
|
|
|
|
|
if (valid_sst_files.empty()) {
|
|
|
|
// No valid SST files are found
|
|
|
|
// Exit with an error state
|
|
|
|
if (dir) {
|
|
|
|
fprintf(stdout, "------------------------------\n");
|
|
|
|
fprintf(stderr, "No valid SST files found in %s\n", dir_or_file);
|
|
|
|
} else {
|
|
|
|
fprintf(stderr, "%s is not a valid SST file\n", dir_or_file);
|
|
|
|
}
|
|
|
|
return 1;
|
|
|
|
} else {
|
|
|
|
if (command == "identify") {
|
|
|
|
if (dir) {
|
|
|
|
fprintf(stdout, "------------------------------\n");
|
|
|
|
fprintf(stdout, "List of valid SST files found in %s:\n", dir_or_file);
|
|
|
|
for (const auto& f : valid_sst_files) {
|
|
|
|
fprintf(stdout, "%s\n", f.c_str());
|
|
|
|
}
|
|
|
|
fprintf(stdout, "Number of valid SST files: %zu\n",
|
|
|
|
valid_sst_files.size());
|
|
|
|
} else {
|
|
|
|
fprintf(stdout, "%s is a valid SST file\n", dir_or_file);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
// At least one valid SST
|
|
|
|
// exit with a success state
|
|
|
|
return 0;
|
|
|
|
}
|
2014-11-08 02:23:58 +01:00
|
|
|
}
|
2020-02-20 21:07:53 +01:00
|
|
|
} // namespace ROCKSDB_NAMESPACE
|
2014-11-13 20:39:30 +01:00
|
|
|
|
|
|
|
#endif // ROCKSDB_LITE
|