Internal changes
PiperOrigin-RevId: 822002697
This commit is contained in:
43
snappy.cc
43
snappy.cc
@@ -280,6 +280,8 @@ inline char* IncrementalCopySlow(const char* src, char* op,
|
||||
// 4, 5, 0, 1, 2, 3, 4, 5, 0, 1}. These byte index sequences are generated by
|
||||
// calling MakePatternMaskBytes(0, 6, index_sequence<16>()) and
|
||||
// MakePatternMaskBytes(16, 6, index_sequence<16>()) respectively.
|
||||
|
||||
|
||||
template <size_t... indexes>
|
||||
inline constexpr std::array<char, sizeof...(indexes)> MakePatternMaskBytes(
|
||||
int index_offset, int pattern_size, index_sequence<indexes...>) {
|
||||
@@ -297,7 +299,6 @@ MakePatternMaskBytesTable(int index_offset,
|
||||
MakePatternMaskBytes(index_offset, pattern_sizes_minus_one + 1,
|
||||
make_index_sequence</*indexes=*/sizeof(V128)>())...};
|
||||
}
|
||||
|
||||
// This is an array of shuffle control masks that can be used as the source
|
||||
// operand for PSHUFB to permute the contents of the destination XMM register
|
||||
// into a repeating byte pattern.
|
||||
@@ -493,7 +494,6 @@ inline char* IncrementalCopy(const char* src, char* op, char* const op_limit,
|
||||
LoadPatternAndReshuffleMask(src, pattern_size);
|
||||
V128 pattern = pattern_and_reshuffle_mask.first;
|
||||
V128 reshuffle_mask = pattern_and_reshuffle_mask.second;
|
||||
|
||||
// There is at least one, and at most four 16-byte blocks. Writing four
|
||||
// conditionals instead of a loop allows FDO to layout the code with
|
||||
// respect to the actual probabilities of each length.
|
||||
@@ -520,7 +520,6 @@ inline char* IncrementalCopy(const char* src, char* op, char* const op_limit,
|
||||
LoadPatternAndReshuffleMask(src, pattern_size);
|
||||
V128 pattern = pattern_and_reshuffle_mask.first;
|
||||
V128 reshuffle_mask = pattern_and_reshuffle_mask.second;
|
||||
|
||||
// This code path is relatively cold however so we save code size
|
||||
// by avoiding unrolling and vectorizing.
|
||||
//
|
||||
@@ -959,6 +958,8 @@ emit_remainder:
|
||||
char* CompressFragmentDoubleHash(const char* input, size_t input_size, char* op,
|
||||
uint16_t* table, const int table_size,
|
||||
uint16_t* table2, const int table_size2) {
|
||||
(void)table_size2;
|
||||
assert(table_size == table_size2);
|
||||
// "ip" is the input pointer, and "op" is the output pointer.
|
||||
const char* ip = input;
|
||||
assert(input_size <= kBlockSize);
|
||||
@@ -1224,7 +1225,7 @@ inline bool Copy64BytesWithPatternExtension(ptrdiff_t dst, size_t offset) {
|
||||
void MemCopy64(char* dst, const void* src, size_t size) {
|
||||
// Always copy this many bytes. If that's below size then copy the full 64.
|
||||
constexpr int kShortMemCopy = 32;
|
||||
|
||||
(void)kShortMemCopy;
|
||||
assert(size <= 64);
|
||||
assert(std::less_equal<const void*>()(static_cast<const char*>(src) + size,
|
||||
dst) ||
|
||||
@@ -1243,6 +1244,27 @@ void MemCopy64(char* dst, const void* src, size_t size) {
|
||||
data = _mm256_lddqu_si256(static_cast<const __m256i *>(src) + 1);
|
||||
_mm256_storeu_si256(reinterpret_cast<__m256i *>(dst) + 1, data);
|
||||
}
|
||||
// RVV acceleration available on RISC-V when compiled with -march=rv64gcv
|
||||
#elif defined(__riscv) && SNAPPY_HAVE_RVV
|
||||
// Cast pointers to the type we will operate on.
|
||||
unsigned char* dst_ptr = reinterpret_cast<unsigned char*>(dst);
|
||||
const unsigned char* src_ptr = reinterpret_cast<const unsigned char*>(src);
|
||||
size_t remaining_bytes = size;
|
||||
// Loop as long as there are bytes remaining to be copied.
|
||||
while (remaining_bytes > 0) {
|
||||
// Set vector configuration: e8 (8-bit elements), m2 (LMUL=2).
|
||||
// Use e8m2 configuration to maximize throughput.
|
||||
size_t vl = VSETVL_E8M2(remaining_bytes);
|
||||
// Load data from the current source pointer.
|
||||
vuint8m2_t vec = VLE8_V_U8M2(src_ptr, vl);
|
||||
// Store data to the current destination pointer.
|
||||
VSE8_V_U8M2(dst_ptr, vec, vl);
|
||||
// Update pointers and the remaining count.
|
||||
src_ptr += vl;
|
||||
dst_ptr += vl;
|
||||
remaining_bytes -= vl;
|
||||
}
|
||||
|
||||
#else
|
||||
std::memmove(dst, src, kShortMemCopy);
|
||||
// Profiling shows that nearly all copies are short.
|
||||
@@ -1687,7 +1709,8 @@ constexpr uint32_t CalculateNeeded(uint8_t tag) {
|
||||
#if __cplusplus >= 201402L
|
||||
constexpr bool VerifyCalculateNeeded() {
|
||||
for (int i = 0; i < 1; i++) {
|
||||
if (CalculateNeeded(i) != (char_table[i] >> 11) + 1) return false;
|
||||
if (CalculateNeeded(i) != static_cast<uint32_t>((char_table[i] >> 11)) + 1)
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1775,6 +1798,7 @@ static bool InternalUncompressAllTags(SnappyDecompressor* decompressor,
|
||||
Writer* writer, uint32_t compressed_len,
|
||||
uint32_t uncompressed_len) {
|
||||
int token = 0;
|
||||
|
||||
writer->SetExpectedLength(uncompressed_len);
|
||||
|
||||
// Process the entire input
|
||||
@@ -1885,16 +1909,16 @@ class SnappyIOVecReader : public Source {
|
||||
if (total_size > 0 && curr_size_remaining_ == 0) Advance();
|
||||
}
|
||||
|
||||
~SnappyIOVecReader() = default;
|
||||
~SnappyIOVecReader() override = default;
|
||||
|
||||
size_t Available() const { return total_size_remaining_; }
|
||||
size_t Available() const override { return total_size_remaining_; }
|
||||
|
||||
const char* Peek(size_t* len) {
|
||||
const char* Peek(size_t* len) override {
|
||||
*len = curr_size_remaining_;
|
||||
return curr_pos_;
|
||||
}
|
||||
|
||||
void Skip(size_t n) {
|
||||
void Skip(size_t n) override {
|
||||
while (n >= curr_size_remaining_ && n > 0) {
|
||||
n -= curr_size_remaining_;
|
||||
Advance();
|
||||
@@ -2532,7 +2556,6 @@ bool SnappyScatteredWriter<Allocator>::SlowAppendFromSelf(size_t offset,
|
||||
class SnappySinkAllocator {
|
||||
public:
|
||||
explicit SnappySinkAllocator(Sink* dest) : dest_(dest) {}
|
||||
~SnappySinkAllocator() {}
|
||||
|
||||
char* Allocate(int size) {
|
||||
Datablock block(new char[size], size);
|
||||
|
||||
Reference in New Issue
Block a user