mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
Merge branch 'master' into jkeiser/no-padding
This commit is contained in:
@@ -31,8 +31,62 @@ public:
|
||||
// it helps tremendously.
|
||||
if (bits == 0)
|
||||
return;
|
||||
int cnt = static_cast<int>(count_ones(bits));
|
||||
#if defined(SIMDJSON_PREFER_REVERSE_BITS)
|
||||
/**
|
||||
* ARM lacks a fast trailing zero instruction, but it has a fast
|
||||
* bit reversal instruction and a fast leading zero instruction.
|
||||
* Thus it may be profitable to reverse the bits (once) and then
|
||||
* to rely on a sequence of instructions that call the leading
|
||||
* zero instruction.
|
||||
*
|
||||
* Performance notes:
|
||||
* The chosen routine is not optimal in terms of data dependency
|
||||
* since zero_leading_bit might require two instructions. However,
|
||||
* it tends to minimize the total number of instructions which is
|
||||
* beneficial.
|
||||
*/
|
||||
|
||||
uint64_t rev_bits = reverse_bits(bits);
|
||||
int cnt = static_cast<int>(count_ones(bits));
|
||||
int i = 0;
|
||||
// Do the first 8 all together
|
||||
for (; i<8; i++) {
|
||||
int lz = leading_zeroes(rev_bits);
|
||||
this->tail[i] = static_cast<uint32_t>(idx) + lz;
|
||||
rev_bits = zero_leading_bit(rev_bits, lz);
|
||||
}
|
||||
// Do the next 8 all together (we hope in most cases it won't happen at all
|
||||
// and the branch is easily predicted).
|
||||
if (simdjson_unlikely(cnt > 8)) {
|
||||
i = 8;
|
||||
for (; i<16; i++) {
|
||||
int lz = leading_zeroes(rev_bits);
|
||||
this->tail[i] = static_cast<uint32_t>(idx) + lz;
|
||||
rev_bits = zero_leading_bit(rev_bits, lz);
|
||||
}
|
||||
|
||||
|
||||
// Most files don't have 16+ structurals per block, so we take several basically guaranteed
|
||||
// branch mispredictions here. 16+ structurals per block means either punctuation ({} [] , :)
|
||||
// or the start of a value ("abc" true 123) every four characters.
|
||||
if (simdjson_unlikely(cnt > 16)) {
|
||||
i = 16;
|
||||
while (rev_bits != 0) {
|
||||
int lz = leading_zeroes(rev_bits);
|
||||
this->tail[i++] = static_cast<uint32_t>(idx) + lz;
|
||||
rev_bits = zero_leading_bit(rev_bits, lz);
|
||||
}
|
||||
}
|
||||
}
|
||||
this->tail += cnt;
|
||||
#else // SIMDJSON_PREFER_REVERSE_BITS
|
||||
/**
|
||||
* Under recent x64 systems, we often have both a fast trailing zero
|
||||
* instruction and a fast 'clear-lower-bit' instruction so the following
|
||||
* algorithm can be competitive.
|
||||
*/
|
||||
|
||||
int cnt = static_cast<int>(count_ones(bits));
|
||||
// Do the first 8 all together
|
||||
for (int i=0; i<8; i++) {
|
||||
this->tail[i] = idx + trailing_zeroes(bits);
|
||||
@@ -61,6 +115,7 @@ public:
|
||||
}
|
||||
|
||||
this->tail += cnt;
|
||||
#endif
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
Reference in New Issue
Block a user