RISC-V: Optimize Snappy decompression tag advance (AdvanceToNextTagRVOptimized) for +4% throughput
diff --git a/snappy.cc b/snappy.cc
index d6d709a..8be5fa4 100644
--- a/snappy.cc
+++ b/snappy.cc
@@ -1363,6 +1363,32 @@
   return tag_type;
 }
 
+SNAPPY_ATTRIBUTE_ALWAYS_INLINE
+inline size_t AdvanceToNextTagRVOptimized(const uint8_t** ip_p, size_t* tag) {
+  const uint8_t*& ip = *ip_p;
+  // This section is crucial for the throughput of the decompression loop.
+  // The latency of an iteration is fundamentally constrained by the data chain on ip:
+  // ip -> c = *tag -> literal_len = c >> 2, tag_type = c & 3
+  //      -> literal_advance = literal_len + 2, copy_advance = tag_type + 1
+  //      -> next_ip = ip + literal_advance  OR  ip + copy_advance (literal vs copy)
+  //      -> *tag = byte at (next_ip - 1); ip = next_ip
+  //
+  // Base RISC-V has no x86-style cmov and no AArch64 csinc on the same shape; this
+  // computes both candidate advances and both load offsets, then selects with
+  // (is_literal ? ... : ...). With the Zicond extension (czero.eqz / czero.nez), those
+  // selections typically lower to branchless conditional-zero ops instead of a
+  // hard-to-predict literal/copy branch, which is why this form tends to win there.
+  const size_t literal_len = *tag >> 2;
+  const size_t tag_type = *tag & 3;
+  const bool is_literal = (tag_type == 0);
+  const size_t copy_advance = tag_type + 1;
+  const size_t literal_advance = literal_len + 2;
+  const uint8_t* next_ip = is_literal ? (ip + literal_advance) : (ip + copy_advance);
+  *tag = is_literal ? ip[literal_advance - 1] : ip[copy_advance - 1];
+  ip = next_ip;
+  return tag_type;
+}
+
 // Extract the offset for copy-1 and copy-2 returns 0 for literals or copy-4.
 inline uint32_t ExtractOffset(uint32_t val, size_t tag_type) {
   // For x86 non-static storage works better. For ARM static storage is better.
@@ -1439,6 +1465,11 @@
         // We never need more than 16 bits. Doing a Load16 allows the compiler
         // to elide the masking operation in ExtractOffset.
         next = LittleEndian::Load16(old_ip);
+#elif defined(__riscv)
+        size_t tag_type = AdvanceToNextTagRVOptimized(&ip, &tag);
+        // We never need more than 16 bits. Doing a Load16 allows the compiler
+        // to elide the masking operation in ExtractOffset.
+        next = LittleEndian::Load16(old_ip);
 #else
         size_t tag_type = AdvanceToNextTagX86Optimized(&ip, &tag);
         next = LittleEndian::Load32(old_ip);