Merge pull request #250 from WillemKauf:compression-context-static PiperOrigin-RevId: 959481048
diff --git a/snappy-c.cc b/snappy-c.cc index 6da18d8..473a0b0 100644 --- a/snappy-c.cc +++ b/snappy-c.cc
@@ -39,9 +39,6 @@ return SNAPPY_BUFFER_TOO_SMALL; } snappy::RawCompress(input, input_length, compressed, compressed_length); - if (*compressed_length == 0) { - return SNAPPY_INVALID_INPUT; - } return SNAPPY_OK; }
diff --git a/snappy-c.h b/snappy-c.h index e228785..32aa0c6 100644 --- a/snappy-c.h +++ b/snappy-c.h
@@ -59,9 +59,6 @@ * <compressed_length> contains the true length of the compressed output, * and SNAPPY_OK is returned. * - * An <input_length> of 2^32 or more returns SNAPPY_INVALID_INPUT with - * <compressed_length> set to 0 and nothing written. - * * Example: * size_t output_length = snappy_max_compressed_length(input_length); * char* output = (char*)malloc(output_length);
diff --git a/snappy-internal.h b/snappy-internal.h index 582e886..403f523 100644 --- a/snappy-internal.h +++ b/snappy-internal.h
@@ -157,7 +157,7 @@ private: char* mem_; // the allocated memory, never nullptr size_t size_; // the size of the allocated memory, never 0 - bool owns_mem_; // whether the destructor should free mem_ + bool owns_mem_; uint16_t* table_; // the pointer to the hashtable char* input_; // the pointer to the input scratch buffer char* output_; // the pointer to the output scratch buffer
diff --git a/snappy-test.cc b/snappy-test.cc index 5c9d041..aae6072 100644 --- a/snappy-test.cc +++ b/snappy-test.cc
@@ -464,8 +464,7 @@ // Make sure we're at the end-of-compressed-data point. This means // if we call inflate with Z_FINISH we won't consume any input or // write any output - Bytef dummyout; - Bytef dummyin = 0; + Bytef dummyin, dummyout; uLongf dummylen = 0; if ( UncompressChunkOrAll(&dummyout, &dummylen, &dummyin, 0, Z_FINISH) != Z_OK ) {
diff --git a/snappy.cc b/snappy.cc index 555588a..5bcd580 100644 --- a/snappy.cc +++ b/snappy.cc
@@ -76,7 +76,6 @@ #include <cstring> #include <limits> #include <memory> -#include <new> #include <string> #include <utility> #include <vector> @@ -1889,10 +1888,7 @@ assert(options.level == 1 || options.level == 2); size_t written = 0; size_t N = reader->Available(); - // The uncompressed length is a 32-bit varint in the stream format. - if (static_cast<uint64_t>(N) > std::numeric_limits<uint32_t>::max()) { - return 0; - } + assert(N <= 0xFFFFFFFFu); char ulength[Varint::kMax32]; char* p = Varint::Encode32(ulength, N); writer->Append(ulength, p - ulength); @@ -2469,17 +2465,6 @@ *compressed_length = (writer.CurrentDestination() - compressed); } -void RawCompress(const char* input, size_t input_length, char* compressed, - size_t* compressed_length, CompressionOptions options, - CompressionContext* ctx) { - ByteArraySource reader(input, input_length); - UncheckedByteArraySink writer(compressed); - Compress(&reader, &writer, options, ctx); - - // Compute how many bytes were added - *compressed_length = (writer.CurrentDestination() - compressed); -} - void RawCompressFromIOVec(const struct iovec* iov, size_t uncompressed_length, char* compressed, size_t* compressed_length) { RawCompressFromIOVec(iov, uncompressed_length, compressed, compressed_length, @@ -2497,6 +2482,17 @@ *compressed_length = writer.CurrentDestination() - compressed; } +void RawCompress(const char* input, size_t input_length, char* compressed, + size_t* compressed_length, CompressionOptions options, + CompressionContext* ctx) { + ByteArraySource reader(input, input_length); + UncheckedByteArraySink writer(compressed); + Compress(&reader, &writer, options, ctx); + + // Compute how many bytes were added + *compressed_length = (writer.CurrentDestination() - compressed); +} + size_t Compress(const char* input, size_t input_length, std::string* compressed) { return Compress(input, input_length, compressed, CompressionOptions{});
diff --git a/snappy.h b/snappy.h index 9f05ca4..3653739 100644 --- a/snappy.h +++ b/snappy.h
@@ -50,9 +50,9 @@ class Source; class Sink; - namespace internal { - class WorkingMemory; - } // end namespace internal +namespace internal { +class WorkingMemory; +} // end namespace internal struct CompressionOptions { // Compression level. @@ -129,7 +129,7 @@ // ------------------------------------------------------------------------ // Compress the bytes read from "*reader" and append to "*writer". Return the - // number of bytes written, or zero if "*reader" has 2^32 or more bytes. + // number of bytes written. // First version is to preserve ABI. size_t Compress(Source* reader, Sink* writer); size_t Compress(Source* reader, Sink* writer, @@ -154,8 +154,7 @@ // ------------------------------------------------------------------------ // Sets "*compressed" to the compressed version of "input[0..input_length-1]". - // Original contents of *compressed are lost. Returns zero and writes nothing - // if "input_length" is 2^32 or more. + // Original contents of *compressed are lost. // // REQUIRES: "input[]" is not an alias of "*compressed". // First version is to preserve ABI. @@ -167,8 +166,7 @@ // Same as `Compress` above but taking an `iovec` array as input. Note that // this function preprocesses the inputs to compute the sum of // `iov[0..iov_cnt-1].iov_len` before reading. To avoid this, use - // `RawCompressFromIOVec` below. Returns zero and writes nothing if that sum - // is 2^32 or more. + // `RawCompressFromIOVec` below. // First version is to preserve ABI. size_t CompressFromIOVec(const struct iovec* iov, size_t iov_cnt, std::string* compressed); @@ -209,8 +207,7 @@ // Takes the data stored in "input[0..input_length]" and stores // it in the array pointed to by "compressed". // - // "*compressed_length" is set to the length of the compressed output, or to - // zero, with nothing written, if "input_length" is 2^32 or more. + // "*compressed_length" is set to the length of the compressed output. // // Example: // char* output = new char[snappy::MaxCompressedLength(input_length)]; @@ -223,7 +220,6 @@ size_t* compressed_length); void RawCompress(const char* input, size_t input_length, char* compressed, size_t* compressed_length, CompressionOptions options); - // Same as the above, but uses the working memory of "*ctx" instead of // allocating it internally. See CompressionContext. void RawCompress(const char* input, size_t input_length, char* compressed, @@ -232,9 +228,7 @@ // Same as `RawCompress` above but taking an `iovec` array as input. Note that // `uncompressed_length` is the total number of bytes to be read from the - // elements of `iov` (_not_ the number of elements in `iov`). Sets - // "*compressed_length" to zero and writes nothing if `uncompressed_length` - // is 2^32 or more. + // elements of `iov` (_not_ the number of elements in `iov`). // First version is to preserve ABI. void RawCompressFromIOVec(const struct iovec* iov, size_t uncompressed_length, char* compressed, size_t* compressed_length);
diff --git a/snappy_unittest.cc b/snappy_unittest.cc index bbce75b..8d83858 100644 --- a/snappy_unittest.cc +++ b/snappy_unittest.cc
@@ -544,7 +544,7 @@ } TEST(Snappy, CompressionContext) { - std::minstd_rand0 rng(snappy::GetFlag(FLAGS_test_random_seed)); + std::minstd_rand0 rng(absl::GetFlag(FLAGS_test_random_seed)); std::uniform_int_distribution<int> uniform_byte(0, 255); // A single context, reused across every compression below. @@ -647,70 +647,6 @@ EXPECT_EQ(a, b); } -// An input of 2^32 bytes or more cannot be expressed by the stream format and -// must be refused rather than compressed under a truncated length. -#if SIZE_MAX > 0xFFFFFFFFu - -// Reports an arbitrary number of bytes available without materializing them. -class OversizedSource : public Source { - public: - explicit OversizedSource(uint64_t total) - : left_(total), buf_(1 << 16, 'a') {} - size_t Available() const override { return static_cast<size_t>(left_); } - const char* Peek(size_t* len) override { - *len = static_cast<size_t>(std::min<uint64_t>(left_, buf_.size())); - return buf_.data(); - } - void Skip(size_t n) override { left_ -= n; } - - private: - uint64_t left_; - std::string buf_; -}; - -// Counts every appended byte and keeps the first few, which is where the -// uncompressed-length varint lives. -class CountingSink : public Sink { - public: - void Append(const char* data, size_t n) override { - for (size_t i = 0; i < n && head_.size() < 8; ++i) head_.push_back(data[i]); - total_ += n; - } - - std::string head_; - uint64_t total_ = 0; -}; - -// Decodes the uncompressed-length varint a compressed stream starts with. -uint32_t DeclaredLength(const std::string& stream) { - uint32_t result = 0; - int shift = 0; - for (size_t i = 0; i < stream.size(); ++i) { - const unsigned char c = static_cast<unsigned char>(stream[i]); - result |= static_cast<uint32_t>(c & 0x7f) << shift; - if (c < 128) break; - shift += 7; - } - return result; -} - -TEST(Snappy, RefusesInputLongerThanTheFormatCanExpress) { - OversizedSource too_big(uint64_t{1} << 32); - CountingSink refused; - EXPECT_EQ(0u, Compress(&too_big, &refused)); - EXPECT_EQ(0u, refused.total_); - EXPECT_TRUE(refused.head_.empty()); - - // One byte below that is the largest input the format can express, and it - // still compresses to a stream whose header names its real length. - OversizedSource largest((uint64_t{1} << 32) - 1); - CountingSink accepted; - EXPECT_GT(Compress(&largest, &accepted), 0u); - EXPECT_EQ(0xFFFFFFFFu, DeclaredLength(accepted.head_)); -} - -#endif // SIZE_MAX > 0xFFFFFFFFu - TEST(Snappy, FourByteOffset) { // The new compressor cannot generate four-byte offsets since // it chops up the input into 32KB pieces. So we hand-emit the