RISC-V: Optimize Snappy decompression tag advance (AdvanceToNextTagRVOptimized) for +4% throughput
This commit is contained in:
31
snappy.cc
31
snappy.cc
@@ -1363,6 +1363,32 @@ inline size_t AdvanceToNextTagX86Optimized(const uint8_t** ip_p, size_t* tag) {
|
||||
return tag_type;
|
||||
}
|
||||
|
||||
SNAPPY_ATTRIBUTE_ALWAYS_INLINE
|
||||
inline size_t AdvanceToNextTagRVOptimized(const uint8_t** ip_p, size_t* tag) {
|
||||
const uint8_t*& ip = *ip_p;
|
||||
// This section is crucial for the throughput of the decompression loop.
|
||||
// The latency of an iteration is fundamentally constrained by the data chain on ip:
|
||||
// ip -> c = *tag -> literal_len = c >> 2, tag_type = c & 3
|
||||
// -> literal_advance = literal_len + 2, copy_advance = tag_type + 1
|
||||
// -> next_ip = ip + literal_advance OR ip + copy_advance (literal vs copy)
|
||||
// -> *tag = byte at (next_ip - 1); ip = next_ip
|
||||
//
|
||||
// Base RISC-V has no x86-style cmov and no AArch64 csinc on the same shape; this
|
||||
// computes both candidate advances and both load offsets, then selects with
|
||||
// (is_literal ? ... : ...). With the Zicond extension (czero.eqz / czero.nez), those
|
||||
// selections typically lower to branchless conditional-zero ops instead of a
|
||||
// hard-to-predict literal/copy branch, which is why this form tends to win there.
|
||||
const size_t literal_len = *tag >> 2;
|
||||
const size_t tag_type = *tag & 3;
|
||||
const bool is_literal = (tag_type == 0);
|
||||
const size_t copy_advance = tag_type + 1;
|
||||
const size_t literal_advance = literal_len + 2;
|
||||
const uint8_t* next_ip = is_literal ? (ip + literal_advance) : (ip + copy_advance);
|
||||
*tag = is_literal ? ip[literal_advance - 1] : ip[copy_advance - 1];
|
||||
ip = next_ip;
|
||||
return tag_type;
|
||||
}
|
||||
|
||||
// Extract the offset for copy-1 and copy-2 returns 0 for literals or copy-4.
|
||||
inline uint32_t ExtractOffset(uint32_t val, size_t tag_type) {
|
||||
// For x86 non-static storage works better. For ARM static storage is better.
|
||||
@@ -1439,6 +1465,11 @@ std::pair<const uint8_t*, ptrdiff_t> DecompressBranchless(
|
||||
// We never need more than 16 bits. Doing a Load16 allows the compiler
|
||||
// to elide the masking operation in ExtractOffset.
|
||||
next = LittleEndian::Load16(old_ip);
|
||||
#elif defined(__riscv)
|
||||
size_t tag_type = AdvanceToNextTagRVOptimized(&ip, &tag);
|
||||
// We never need more than 16 bits. Doing a Load16 allows the compiler
|
||||
// to elide the masking operation in ExtractOffset.
|
||||
next = LittleEndian::Load16(old_ip);
|
||||
#else
|
||||
size_t tag_type = AdvanceToNextTagX86Optimized(&ip, &tag);
|
||||
next = LittleEndian::Load32(old_ip);
|
||||
|
||||
Reference in New Issue
Block a user