mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
Add LoongArch LSX and LASX support (#2159)
* Add LoongArch SX support * Add LoongArch ASX support
This commit is contained in:
@@ -106,6 +106,24 @@ if(
|
||||
)
|
||||
endif()
|
||||
|
||||
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(loongarch64)$")
|
||||
option(SIMDJSON_PREFER_LSX "Prefer LoongArch SX" ON)
|
||||
include(CheckCXXCompilerFlag)
|
||||
check_cxx_compiler_flag(-mlasx COMPILER_SUPPORTS_LASX)
|
||||
check_cxx_compiler_flag(-mlsx COMPILER_SUPPORTS_LSX)
|
||||
if(COMPILER_SUPPORTS_LASX AND NOT SIMDJSON_PREFER_LSX)
|
||||
simdjson_add_props(
|
||||
target_compile_options PRIVATE
|
||||
-mlasx
|
||||
)
|
||||
elseif(COMPILER_SUPPORTS_LSX)
|
||||
simdjson_add_props(
|
||||
target_compile_options PRIVATE
|
||||
-mlsx
|
||||
)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# GCC and Clang have horrendous Debug builds when using SIMD.
|
||||
# A common fix is to use '-Og' instead.
|
||||
# bug https://gcc.gnu.org/bugzilla/show_bug.cgi?id=54412
|
||||
|
||||
@@ -20,6 +20,10 @@
|
||||
#include "simdjson/ppc64.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(westmere)
|
||||
#include "simdjson/westmere.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lsx)
|
||||
#include "simdjson/lsx.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
#include "simdjson/lasx.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
|
||||
@@ -17,6 +17,10 @@ namespace simdjson {
|
||||
namespace ppc64 {}
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(westmere)
|
||||
namespace westmere {}
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lsx)
|
||||
namespace lsx {}
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
namespace lasx {}
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
|
||||
@@ -19,6 +19,10 @@
|
||||
#include "simdjson/ppc64/implementation.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(westmere)
|
||||
#include "simdjson/westmere/implementation.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lsx)
|
||||
#include "simdjson/lsx/implementation.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
#include "simdjson/lasx/implementation.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
|
||||
@@ -20,6 +20,10 @@
|
||||
#include "simdjson/ppc64/ondemand.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(westmere)
|
||||
#include "simdjson/westmere/ondemand.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lsx)
|
||||
#include "simdjson/lsx/ondemand.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
#include "simdjson/lasx/ondemand.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
|
||||
@@ -17,6 +17,10 @@
|
||||
#include "simdjson/arm64/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_PPC64
|
||||
#include "simdjson/ppc64/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_LSX
|
||||
#include "simdjson/lsx/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_LASX
|
||||
#include "simdjson/lasx/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
#include "simdjson/fallback/begin.h"
|
||||
#else
|
||||
|
||||
@@ -10,6 +10,8 @@
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_icelake 4
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_ppc64 5
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_westmere 6
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_lsx 7
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_lasx 8
|
||||
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_FOR(IMPL) SIMDJSON_CAT(SIMDJSON_IMPLEMENTATION_ID_, IMPL)
|
||||
#define SIMDJSON_IMPLEMENTATION_ID SIMDJSON_IMPLEMENTATION_ID_FOR(SIMDJSON_IMPLEMENTATION)
|
||||
@@ -74,9 +76,23 @@
|
||||
#endif
|
||||
#define SIMDJSON_CAN_ALWAYS_RUN_PPC64 SIMDJSON_IMPLEMENTATION_PPC64 && SIMDJSON_IS_PPC64 && SIMDJSON_IS_PPC64_VMX
|
||||
|
||||
#ifndef SIMDJSON_IMPLEMENTATION_LASX
|
||||
#define SIMDJSON_IMPLEMENTATION_LASX (SIMDJSON_IS_LOONGARCH64 && __loongarch_asx)
|
||||
#endif
|
||||
#define SIMDJSON_CAN_ALWAYS_RUN_LASX (SIMDJSON_IMPLEMENTATION_LASX)
|
||||
|
||||
#ifndef SIMDJSON_IMPLEMENTATION_LSX
|
||||
#if SIMDJSON_CAN_ALWAYS_RUN_LASX
|
||||
#define SIMDJSON_IMPLEMENTATION_LSX 0
|
||||
#else
|
||||
#define SIMDJSON_IMPLEMENTATION_LSX (SIMDJSON_IS_LOONGARCH64 && __loongarch_sx)
|
||||
#endif
|
||||
#endif
|
||||
#define SIMDJSON_CAN_ALWAYS_RUN_LSX (SIMDJSON_IMPLEMENTATION_LSX)
|
||||
|
||||
// Default Fallback to on unless a builtin implementation has already been selected.
|
||||
#ifndef SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
#if SIMDJSON_CAN_ALWAYS_RUN_ARM64 || SIMDJSON_CAN_ALWAYS_RUN_ICELAKE || SIMDJSON_CAN_ALWAYS_RUN_HASWELL || SIMDJSON_CAN_ALWAYS_RUN_WESTMERE || SIMDJSON_CAN_ALWAYS_RUN_PPC64
|
||||
#if SIMDJSON_CAN_ALWAYS_RUN_ARM64 || SIMDJSON_CAN_ALWAYS_RUN_ICELAKE || SIMDJSON_CAN_ALWAYS_RUN_HASWELL || SIMDJSON_CAN_ALWAYS_RUN_WESTMERE || SIMDJSON_CAN_ALWAYS_RUN_PPC64 || SIMDJSON_CAN_ALWAYS_RUN_LSX || SIMDJSON_CAN_ALWAYS_RUN_LASX
|
||||
// if anything at all except fallback can always run, then disable fallback.
|
||||
#define SIMDJSON_IMPLEMENTATION_FALLBACK 0
|
||||
#else
|
||||
@@ -98,6 +114,10 @@
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION arm64
|
||||
#elif SIMDJSON_CAN_ALWAYS_RUN_PPC64
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION ppc64
|
||||
#elif SIMDJSON_CAN_ALWAYS_RUN_LSX
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION lsx
|
||||
#elif SIMDJSON_CAN_ALWAYS_RUN_LASX
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION lasx
|
||||
#elif SIMDJSON_CAN_ALWAYS_RUN_FALLBACK
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION fallback
|
||||
#else
|
||||
|
||||
@@ -66,7 +66,9 @@ enum instruction_set {
|
||||
AVX512CD = 0x2000,
|
||||
AVX512BW = 0x4000,
|
||||
AVX512VL = 0x8000,
|
||||
AVX512VBMI2 = 0x10000
|
||||
AVX512VBMI2 = 0x10000,
|
||||
LSX = 0x20000,
|
||||
LASX = 0x40000,
|
||||
};
|
||||
|
||||
} // namespace internal
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SIMDJSON_LASX_H
|
||||
#define SIMDJSON_LASX_H
|
||||
|
||||
#include "simdjson/lasx/begin.h"
|
||||
#include "simdjson/generic/amalgamated.h"
|
||||
#include "simdjson/lasx/end.h"
|
||||
|
||||
#endif // SIMDJSON_LASX_H
|
||||
@@ -0,0 +1,26 @@
|
||||
#ifndef SIMDJSON_LASX_BASE_H
|
||||
#define SIMDJSON_LASX_BASE_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
/**
|
||||
* Implementation for LASX.
|
||||
*/
|
||||
namespace lasx {
|
||||
|
||||
class implementation;
|
||||
|
||||
namespace {
|
||||
namespace simd {
|
||||
template <typename T> struct simd8;
|
||||
template <typename T> struct simd8x64;
|
||||
} // namespace simd
|
||||
} // unnamed namespace
|
||||
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LASX_BASE_H
|
||||
@@ -0,0 +1,10 @@
|
||||
#define SIMDJSON_IMPLEMENTATION lasx
|
||||
#include "simdjson/lasx/base.h"
|
||||
#include "simdjson/lasx/intrinsics.h"
|
||||
#include "simdjson/lasx/bitmanipulation.h"
|
||||
#include "simdjson/lasx/bitmask.h"
|
||||
#include "simdjson/lasx/numberparsing_defs.h"
|
||||
#include "simdjson/lasx/simd.h"
|
||||
#include "simdjson/lasx/stringparsing_defs.h"
|
||||
|
||||
#define SIMDJSON_SKIP_BACKSLASH_SHORT_CIRCUIT 1
|
||||
@@ -0,0 +1,50 @@
|
||||
#ifndef SIMDJSON_LASX_BITMANIPULATION_H
|
||||
#define SIMDJSON_LASX_BITMANIPULATION_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lasx/base.h"
|
||||
#include "simdjson/lasx/intrinsics.h"
|
||||
#include "simdjson/lasx/bitmask.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lasx {
|
||||
namespace {
|
||||
|
||||
// We sometimes call trailing_zero on inputs that are zero,
|
||||
// but the algorithms do not end up using the returned value.
|
||||
// Sadly, sanitizers are not smart enough to figure it out.
|
||||
SIMDJSON_NO_SANITIZE_UNDEFINED
|
||||
// This function can be used safely even if not all bytes have been
|
||||
// initialized.
|
||||
// See issue https://github.com/simdjson/simdjson/issues/1965
|
||||
SIMDJSON_NO_SANITIZE_MEMORY
|
||||
simdjson_inline int trailing_zeroes(uint64_t input_num) {
|
||||
return __builtin_ctzll(input_num);
|
||||
}
|
||||
|
||||
/* result might be undefined when input_num is zero */
|
||||
simdjson_inline uint64_t clear_lowest_bit(uint64_t input_num) {
|
||||
return input_num & (input_num-1);
|
||||
}
|
||||
|
||||
/* result might be undefined when input_num is zero */
|
||||
simdjson_inline int leading_zeroes(uint64_t input_num) {
|
||||
return __builtin_clzll(input_num);
|
||||
}
|
||||
|
||||
/* result might be undefined when input_num is zero */
|
||||
simdjson_inline int count_ones(uint64_t input_num) {
|
||||
return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
|
||||
}
|
||||
|
||||
simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
|
||||
return __builtin_uaddll_overflow(value1, value2,
|
||||
reinterpret_cast<unsigned long long *>(result));
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LASX_BITMANIPULATION_H
|
||||
@@ -0,0 +1,31 @@
|
||||
#ifndef SIMDJSON_LASX_BITMASK_H
|
||||
#define SIMDJSON_LASX_BITMASK_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lasx/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lasx {
|
||||
namespace {
|
||||
|
||||
//
|
||||
// Perform a "cumulative bitwise xor," flipping bits each time a 1 is encountered.
|
||||
//
|
||||
// For example, prefix_xor(00100100) == 00011100
|
||||
//
|
||||
simdjson_inline uint64_t prefix_xor(uint64_t bitmask) {
|
||||
bitmask ^= bitmask << 1;
|
||||
bitmask ^= bitmask << 2;
|
||||
bitmask ^= bitmask << 4;
|
||||
bitmask ^= bitmask << 8;
|
||||
bitmask ^= bitmask << 16;
|
||||
bitmask ^= bitmask << 32;
|
||||
return bitmask;
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,6 @@
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lasx/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#undef SIMDJSON_SKIP_BACKSLASH_SHORT_CIRCUIT
|
||||
#undef SIMDJSON_IMPLEMENTATION
|
||||
@@ -0,0 +1,31 @@
|
||||
#ifndef SIMDJSON_LASX_IMPLEMENTATION_H
|
||||
#define SIMDJSON_LASX_IMPLEMENTATION_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/base.h"
|
||||
#include "simdjson/implementation.h"
|
||||
#include "simdjson/internal/instruction_set.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lasx {
|
||||
|
||||
/**
|
||||
* @private
|
||||
*/
|
||||
class implementation final : public simdjson::implementation {
|
||||
public:
|
||||
simdjson_inline implementation() : simdjson::implementation("lasx", "LoongArch ASX", internal::instruction_set::LASX) {}
|
||||
simdjson_warn_unused error_code create_dom_parser_implementation(
|
||||
size_t capacity,
|
||||
size_t max_length,
|
||||
std::unique_ptr<internal::dom_parser_implementation>& dst
|
||||
) const noexcept final;
|
||||
simdjson_warn_unused error_code minify(const uint8_t *buf, size_t len, uint8_t *dst, size_t &dst_len) const noexcept final;
|
||||
simdjson_warn_unused bool validate_utf8(const char *buf, size_t len) const noexcept final;
|
||||
};
|
||||
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LASX_IMPLEMENTATION_H
|
||||
@@ -0,0 +1,14 @@
|
||||
#ifndef SIMDJSON_LASX_INTRINSICS_H
|
||||
#define SIMDJSON_LASX_INTRINSICS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lasx/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
// This should be the correct header whether
|
||||
// you use visual studio or other compilers.
|
||||
#include <lasxintrin.h>
|
||||
|
||||
static_assert(sizeof(__m256i) <= simdjson::SIMDJSON_PADDING, "insufficient padding for LoongArch ASX");
|
||||
|
||||
#endif // SIMDJSON_LASX_INTRINSICS_H
|
||||
@@ -0,0 +1,41 @@
|
||||
#ifndef SIMDJSON_LASX_NUMBERPARSING_DEFS_H
|
||||
#define SIMDJSON_LASX_NUMBERPARSING_DEFS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lasx/base.h"
|
||||
#include "simdjson/lasx/intrinsics.h"
|
||||
#include "simdjson/internal/numberparsing_tables.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace simdjson {
|
||||
namespace lasx {
|
||||
namespace numberparsing {
|
||||
|
||||
// we don't have appropriate instructions, so let us use a scalar function
|
||||
// credit: https://johnnylee-sde.github.io/Fast-numeric-string-to-int/
|
||||
/** @private */
|
||||
static simdjson_inline uint32_t parse_eight_digits_unrolled(const uint8_t *chars) {
|
||||
uint64_t val;
|
||||
std::memcpy(&val, chars, sizeof(uint64_t));
|
||||
val = (val & 0x0F0F0F0F0F0F0F0F) * 2561 >> 8;
|
||||
val = (val & 0x00FF00FF00FF00FF) * 6553601 >> 16;
|
||||
return uint32_t((val & 0x0000FFFF0000FFFF) * 42949672960001 >> 32);
|
||||
}
|
||||
|
||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||
internal::value128 answer;
|
||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||
answer.low = uint64_t(r);
|
||||
answer.high = uint64_t(r >> 64);
|
||||
return answer;
|
||||
}
|
||||
|
||||
} // namespace numberparsing
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
||||
|
||||
#endif // SIMDJSON_LASX_NUMBERPARSING_DEFS_H
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SIMDJSON_LASX_ONDEMAND_H
|
||||
#define SIMDJSON_LASX_ONDEMAND_H
|
||||
|
||||
#include "simdjson/lasx/begin.h"
|
||||
#include "simdjson/generic/ondemand/amalgamated.h"
|
||||
#include "simdjson/lasx/end.h"
|
||||
|
||||
#endif // SIMDJSON_LASX_ONDEMAND_H
|
||||
@@ -0,0 +1,376 @@
|
||||
#ifndef SIMDJSON_LASX_SIMD_H
|
||||
#define SIMDJSON_LASX_SIMD_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lasx/base.h"
|
||||
#include "simdjson/lasx/bitmanipulation.h"
|
||||
#include "simdjson/internal/simdprune_tables.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lasx {
|
||||
namespace {
|
||||
namespace simd {
|
||||
|
||||
// Forward-declared so they can be used by splat and friends.
|
||||
template<typename Child>
|
||||
struct base {
|
||||
__m256i value;
|
||||
|
||||
// Zero constructor
|
||||
simdjson_inline base() : value{__m256i()} {}
|
||||
|
||||
// Conversion from SIMD register
|
||||
simdjson_inline base(const __m256i _value) : value(_value) {}
|
||||
|
||||
// Conversion to SIMD register
|
||||
simdjson_inline operator const __m256i&() const { return this->value; }
|
||||
simdjson_inline operator __m256i&() { return this->value; }
|
||||
simdjson_inline operator const v32i8&() const { return (v32i8&)this->value; }
|
||||
simdjson_inline operator v32i8&() { return (v32i8&)this->value; }
|
||||
|
||||
// Bit operations
|
||||
simdjson_inline Child operator|(const Child other) const { return __lasx_xvor_v(*this, other); }
|
||||
simdjson_inline Child operator&(const Child other) const { return __lasx_xvand_v(*this, other); }
|
||||
simdjson_inline Child operator^(const Child other) const { return __lasx_xvxor_v(*this, other); }
|
||||
simdjson_inline Child bit_andnot(const Child other) const { return __lasx_xvandn_v(other, *this); }
|
||||
simdjson_inline Child& operator|=(const Child other) { auto this_cast = static_cast<Child*>(this); *this_cast = *this_cast | other; return *this_cast; }
|
||||
simdjson_inline Child& operator&=(const Child other) { auto this_cast = static_cast<Child*>(this); *this_cast = *this_cast & other; return *this_cast; }
|
||||
simdjson_inline Child& operator^=(const Child other) { auto this_cast = static_cast<Child*>(this); *this_cast = *this_cast ^ other; return *this_cast; }
|
||||
};
|
||||
|
||||
// Forward-declared so they can be used by splat and friends.
|
||||
template<typename T>
|
||||
struct simd8;
|
||||
|
||||
template<typename T, typename Mask=simd8<bool>>
|
||||
struct base8: base<simd8<T>> {
|
||||
simdjson_inline base8() : base<simd8<T>>() {}
|
||||
simdjson_inline base8(const __m256i _value) : base<simd8<T>>(_value) {}
|
||||
|
||||
friend simdjson_really_inline Mask operator==(const simd8<T> lhs, const simd8<T> rhs) { return __lasx_xvseq_b(lhs, rhs); }
|
||||
|
||||
static const int SIZE = sizeof(base<simd8<T>>::value);
|
||||
|
||||
template<int N=1>
|
||||
simdjson_inline simd8<T> prev(const simd8<T> prev_chunk) const {
|
||||
__m256i hi = __lasx_xvbsll_v(*this, N);
|
||||
__m256i lo = __lasx_xvbsrl_v(*this, 16 - N);
|
||||
__m256i tmp = __lasx_xvbsrl_v(prev_chunk, 16 - N);
|
||||
lo = __lasx_xvpermi_q(lo, tmp, 0x21);
|
||||
return __lasx_xvor_v(hi, lo);
|
||||
}
|
||||
};
|
||||
|
||||
// SIMD byte mask type (returned by things like eq and gt)
|
||||
template<>
|
||||
struct simd8<bool>: base8<bool> {
|
||||
static simdjson_inline simd8<bool> splat(bool _value) { return __lasx_xvreplgr2vr_b(uint8_t(-(!!_value))); }
|
||||
|
||||
simdjson_inline simd8<bool>() : base8() {}
|
||||
simdjson_inline simd8<bool>(const __m256i _value) : base8<bool>(_value) {}
|
||||
// Splat constructor
|
||||
simdjson_inline simd8<bool>(bool _value) : base8<bool>(splat(_value)) {}
|
||||
|
||||
simdjson_inline int to_bitmask() const {
|
||||
__m256i mask = __lasx_xvmskltz_b(*this);
|
||||
return (__lasx_xvpickve2gr_w(mask, 4) << 16) | (__lasx_xvpickve2gr_w(mask, 0));
|
||||
}
|
||||
simdjson_inline bool any() const {
|
||||
__m256i v = __lasx_xvmsknz_b(*this);
|
||||
return (0 == __lasx_xvpickve2gr_w(v, 0)) && (0 == __lasx_xvpickve2gr_w(v, 4));
|
||||
}
|
||||
simdjson_inline simd8<bool> operator~() const { return *this ^ true; }
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
struct base8_numeric: base8<T> {
|
||||
static simdjson_inline simd8<T> splat(T _value) {
|
||||
return __lasx_xvreplgr2vr_b(_value);
|
||||
}
|
||||
static simdjson_inline simd8<T> zero() { return __lasx_xvldi(0); }
|
||||
static simdjson_inline simd8<T> load(const T values[32]) {
|
||||
return __lasx_xvld(reinterpret_cast<const __m256i *>(values), 0);
|
||||
}
|
||||
// Repeat 16 values as many times as necessary (usually for lookup tables)
|
||||
static simdjson_inline simd8<T> repeat_16(
|
||||
T v0, T v1, T v2, T v3, T v4, T v5, T v6, T v7,
|
||||
T v8, T v9, T v10, T v11, T v12, T v13, T v14, T v15
|
||||
) {
|
||||
return simd8<T>(
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15,
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
);
|
||||
}
|
||||
|
||||
simdjson_inline base8_numeric() : base8<T>() {}
|
||||
simdjson_inline base8_numeric(const __m256i _value) : base8<T>(_value) {}
|
||||
|
||||
// Store to array
|
||||
simdjson_inline void store(T dst[32]) const {
|
||||
return __lasx_xvst(*this, reinterpret_cast<__m256i *>(dst), 0);
|
||||
}
|
||||
|
||||
// Addition/subtraction are the same for signed and unsigned
|
||||
simdjson_inline simd8<T> operator+(const simd8<T> other) const { return __lasx_xvadd_b(*this, other); }
|
||||
simdjson_inline simd8<T> operator-(const simd8<T> other) const { return __lasx_xvsub_b(*this, other); }
|
||||
simdjson_inline simd8<T>& operator+=(const simd8<T> other) { *this = *this + other; return *static_cast<simd8<T>*>(this); }
|
||||
simdjson_inline simd8<T>& operator-=(const simd8<T> other) { *this = *this - other; return *static_cast<simd8<T>*>(this); }
|
||||
|
||||
// Override to distinguish from bool version
|
||||
simdjson_inline simd8<T> operator~() const { return *this ^ 0xFFu; }
|
||||
|
||||
// Perform a lookup assuming the value is between 0 and 16 (undefined behavior for out of range values)
|
||||
template<typename L>
|
||||
simdjson_inline simd8<L> lookup_16(simd8<L> lookup_table) const {
|
||||
return __lasx_xvshuf_b(lookup_table, lookup_table, *this);
|
||||
}
|
||||
|
||||
// Copies to 'output" all bytes corresponding to a 0 in the mask (interpreted as a bitset).
|
||||
// Passing a 0 value for mask would be equivalent to writing out every byte to output.
|
||||
// Only the first 16 - count_ones(mask) bytes of the result are significant but 16 bytes
|
||||
// get written.
|
||||
template<typename L>
|
||||
simdjson_inline void compress(uint32_t mask, L * output) const {
|
||||
using internal::thintable_epi8;
|
||||
using internal::BitsSetTable256mul2;
|
||||
using internal::pshufb_combine_table;
|
||||
// this particular implementation was inspired by haswell
|
||||
// lasx do it in 4 steps, first 8 bytes and then second 8 bytes...
|
||||
uint8_t mask1 = uint8_t(mask); // least significant 8 bits
|
||||
uint8_t mask2 = uint8_t(mask >> 8); // second significant 8 bits
|
||||
uint8_t mask3 = uint8_t(mask >> 16); // ...
|
||||
uint8_t mask4 = uint8_t(mask >> 24); // ...
|
||||
// next line just loads the 64-bit values thintable_epi8[mask{1,2,3,4}]
|
||||
// into a 256-bit register.
|
||||
__m256i shufmask = {int64_t(thintable_epi8[mask1]), int64_t(thintable_epi8[mask2]) + 0x0808080808080808, int64_t(thintable_epi8[mask3]), int64_t(thintable_epi8[mask4]) + 0x0808080808080808};
|
||||
// this is the version "nearly pruned"
|
||||
__m256i pruned = __lasx_xvshuf_b(*this, *this, shufmask);
|
||||
// we still need to put the pieces back together.
|
||||
// we compute the popcount of the first words:
|
||||
int pop1 = BitsSetTable256mul2[mask1];
|
||||
int pop2 = BitsSetTable256mul2[mask2];
|
||||
int pop3 = BitsSetTable256mul2[mask3];
|
||||
|
||||
// then load the corresponding mask
|
||||
__m256i masklo = __lasx_xvldx(reinterpret_cast<void*>(reinterpret_cast<unsigned long>(pshufb_combine_table)), pop1 * 8);
|
||||
__m256i maskhi = __lasx_xvldx(reinterpret_cast<void*>(reinterpret_cast<unsigned long>(pshufb_combine_table)), pop3 * 8);
|
||||
__m256i compactmask = __lasx_xvpermi_q(maskhi, masklo, 0x20);
|
||||
__m256i answer = __lasx_xvshuf_b(pruned, pruned, compactmask);
|
||||
__lasx_xvst(answer, reinterpret_cast<uint8_t*>(output), 0);
|
||||
uint64_t value3 = __lasx_xvpickve2gr_du(answer, 2);
|
||||
uint64_t value4 = __lasx_xvpickve2gr_du(answer, 3);
|
||||
uint64_t *pos = reinterpret_cast<uint64_t*>(reinterpret_cast<uint8_t*>(output) + 16 - (pop1 + pop2) / 2);
|
||||
pos[0] = value3;
|
||||
pos[1] = value4;
|
||||
}
|
||||
|
||||
template<typename L>
|
||||
simdjson_inline simd8<L> lookup_16(
|
||||
L replace0, L replace1, L replace2, L replace3,
|
||||
L replace4, L replace5, L replace6, L replace7,
|
||||
L replace8, L replace9, L replace10, L replace11,
|
||||
L replace12, L replace13, L replace14, L replace15) const {
|
||||
return lookup_16(simd8<L>::repeat_16(
|
||||
replace0, replace1, replace2, replace3,
|
||||
replace4, replace5, replace6, replace7,
|
||||
replace8, replace9, replace10, replace11,
|
||||
replace12, replace13, replace14, replace15
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
// Signed bytes
|
||||
template<>
|
||||
struct simd8<int8_t> : base8_numeric<int8_t> {
|
||||
simdjson_inline simd8() : base8_numeric<int8_t>() {}
|
||||
simdjson_inline simd8(const __m256i _value) : base8_numeric<int8_t>(_value) {}
|
||||
// Splat constructor
|
||||
simdjson_inline simd8(int8_t _value) : simd8(splat(_value)) {}
|
||||
// Array constructor
|
||||
simdjson_inline simd8(const int8_t values[32]) : simd8(load(values)) {}
|
||||
// Member-by-member initialization
|
||||
simdjson_inline simd8(
|
||||
int8_t v0, int8_t v1, int8_t v2, int8_t v3, int8_t v4, int8_t v5, int8_t v6, int8_t v7,
|
||||
int8_t v8, int8_t v9, int8_t v10, int8_t v11, int8_t v12, int8_t v13, int8_t v14, int8_t v15,
|
||||
int8_t v16, int8_t v17, int8_t v18, int8_t v19, int8_t v20, int8_t v21, int8_t v22, int8_t v23,
|
||||
int8_t v24, int8_t v25, int8_t v26, int8_t v27, int8_t v28, int8_t v29, int8_t v30, int8_t v31
|
||||
) : simd8({
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15,
|
||||
v16,v17,v18,v19,v20,v21,v22,v23,
|
||||
v24,v25,v26,v27,v28,v29,v30,v31
|
||||
}) {}
|
||||
// Repeat 16 values as many times as necessary (usually for lookup tables)
|
||||
simdjson_inline static simd8<int8_t> repeat_16(
|
||||
int8_t v0, int8_t v1, int8_t v2, int8_t v3, int8_t v4, int8_t v5, int8_t v6, int8_t v7,
|
||||
int8_t v8, int8_t v9, int8_t v10, int8_t v11, int8_t v12, int8_t v13, int8_t v14, int8_t v15
|
||||
) {
|
||||
return simd8<int8_t>(
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15,
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
);
|
||||
}
|
||||
|
||||
// Order-sensitive comparisons
|
||||
simdjson_inline simd8<int8_t> max_val(const simd8<int8_t> other) const { return __lasx_xvmax_b(*this, other); }
|
||||
simdjson_inline simd8<int8_t> min_val(const simd8<int8_t> other) const { return __lasx_xvmin_b(*this, other); }
|
||||
simdjson_inline simd8<bool> operator>(const simd8<int8_t> other) const { return __lasx_xvslt_b(other, *this); }
|
||||
simdjson_inline simd8<bool> operator<(const simd8<int8_t> other) const { return __lasx_xvslt_b(*this, other); }
|
||||
};
|
||||
|
||||
// Unsigned bytes
|
||||
template<>
|
||||
struct simd8<uint8_t>: base8_numeric<uint8_t> {
|
||||
simdjson_inline simd8() : base8_numeric<uint8_t>() {}
|
||||
simdjson_inline simd8(const __m256i _value) : base8_numeric<uint8_t>(_value) {}
|
||||
// Splat constructor
|
||||
simdjson_inline simd8(uint8_t _value) : simd8(splat(_value)) {}
|
||||
// Array constructor
|
||||
simdjson_inline simd8(const uint8_t values[32]) : simd8(load(values)) {}
|
||||
// Member-by-member initialization
|
||||
simdjson_inline simd8(
|
||||
uint8_t v0, uint8_t v1, uint8_t v2, uint8_t v3, uint8_t v4, uint8_t v5, uint8_t v6, uint8_t v7,
|
||||
uint8_t v8, uint8_t v9, uint8_t v10, uint8_t v11, uint8_t v12, uint8_t v13, uint8_t v14, uint8_t v15,
|
||||
uint8_t v16, uint8_t v17, uint8_t v18, uint8_t v19, uint8_t v20, uint8_t v21, uint8_t v22, uint8_t v23,
|
||||
uint8_t v24, uint8_t v25, uint8_t v26, uint8_t v27, uint8_t v28, uint8_t v29, uint8_t v30, uint8_t v31
|
||||
) : simd8(__m256i(v32u8{
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15,
|
||||
v16,v17,v18,v19,v20,v21,v22,v23,
|
||||
v24,v25,v26,v27,v28,v29,v30,v31
|
||||
})) {}
|
||||
// Repeat 16 values as many times as necessary (usually for lookup tables)
|
||||
simdjson_inline static simd8<uint8_t> repeat_16(
|
||||
uint8_t v0, uint8_t v1, uint8_t v2, uint8_t v3, uint8_t v4, uint8_t v5, uint8_t v6, uint8_t v7,
|
||||
uint8_t v8, uint8_t v9, uint8_t v10, uint8_t v11, uint8_t v12, uint8_t v13, uint8_t v14, uint8_t v15
|
||||
) {
|
||||
return simd8<uint8_t>(
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15,
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
);
|
||||
}
|
||||
|
||||
// Saturated math
|
||||
simdjson_inline simd8<uint8_t> saturating_add(const simd8<uint8_t> other) const { return __lasx_xvsadd_bu(*this, other); }
|
||||
simdjson_inline simd8<uint8_t> saturating_sub(const simd8<uint8_t> other) const { return __lasx_xvssub_bu(*this, other); }
|
||||
|
||||
// Order-specific operations
|
||||
simdjson_inline simd8<uint8_t> max_val(const simd8<uint8_t> other) const { return __lasx_xvmax_bu(*this, other); }
|
||||
simdjson_inline simd8<uint8_t> min_val(const simd8<uint8_t> other) const { return __lasx_xvmin_bu(other, *this); }
|
||||
// Same as >, but only guarantees true is nonzero (< guarantees true = -1)
|
||||
simdjson_inline simd8<uint8_t> gt_bits(const simd8<uint8_t> other) const { return this->saturating_sub(other); }
|
||||
// Same as <, but only guarantees true is nonzero (< guarantees true = -1)
|
||||
simdjson_inline simd8<uint8_t> lt_bits(const simd8<uint8_t> other) const { return other.saturating_sub(*this); }
|
||||
simdjson_inline simd8<bool> operator<=(const simd8<uint8_t> other) const { return other.max_val(*this) == other; }
|
||||
simdjson_inline simd8<bool> operator>=(const simd8<uint8_t> other) const { return other.min_val(*this) == other; }
|
||||
simdjson_inline simd8<bool> operator>(const simd8<uint8_t> other) const { return this->gt_bits(other).any_bits_set(); }
|
||||
simdjson_inline simd8<bool> operator<(const simd8<uint8_t> other) const { return this->lt_bits(other).any_bits_set(); }
|
||||
|
||||
// Bit-specific operations
|
||||
simdjson_inline simd8<bool> bits_not_set() const { return *this == uint8_t(0); }
|
||||
simdjson_inline simd8<bool> bits_not_set(simd8<uint8_t> bits) const { return (*this & bits).bits_not_set(); }
|
||||
simdjson_inline simd8<bool> any_bits_set() const { return ~this->bits_not_set(); }
|
||||
simdjson_inline simd8<bool> any_bits_set(simd8<uint8_t> bits) const { return ~this->bits_not_set(bits); }
|
||||
simdjson_inline bool is_ascii() const {
|
||||
__m256i mask = __lasx_xvmskltz_b(*this);
|
||||
return (0 == __lasx_xvpickve2gr_w(mask, 0)) && (0 == __lasx_xvpickve2gr_w(mask, 4));
|
||||
}
|
||||
simdjson_inline bool bits_not_set_anywhere() const {
|
||||
__m256i v = __lasx_xvmsknz_b(*this);
|
||||
return (0 == __lasx_xvpickve2gr_w(v, 0)) && (0 == __lasx_xvpickve2gr_w(v, 4));
|
||||
}
|
||||
simdjson_inline bool any_bits_set_anywhere() const { return !bits_not_set_anywhere(); }
|
||||
simdjson_inline bool bits_not_set_anywhere(simd8<uint8_t> bits) const {
|
||||
__m256i v = __lasx_xvmsknz_b(__lasx_xvand_v(*this, bits));
|
||||
return (0 == __lasx_xvpickve2gr_w(v, 0)) && (0 == __lasx_xvpickve2gr_w(v, 4));
|
||||
}
|
||||
simdjson_inline bool any_bits_set_anywhere(simd8<uint8_t> bits) const { return !bits_not_set_anywhere(bits); }
|
||||
template<int N>
|
||||
simdjson_inline simd8<uint8_t> shr() const { return simd8<uint8_t>(__lasx_xvsrli_b(*this, N)); }
|
||||
template<int N>
|
||||
simdjson_inline simd8<uint8_t> shl() const { return simd8<uint8_t>(__lasx_xvslli_b(*this, N)); }
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
struct simd8x64 {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<T>);
|
||||
static_assert(NUM_CHUNKS == 2, "LASX kernel should use two registers per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
|
||||
simd8x64(const simd8x64<T>& o) = delete; // no copy allowed
|
||||
simd8x64<T>& operator=(const simd8<T>& other) = delete; // no assignment allowed
|
||||
simd8x64() = delete; // no default constructor allowed
|
||||
|
||||
simdjson_inline simd8x64(const simd8<T> chunk0, const simd8<T> chunk1) : chunks{chunk0, chunk1} {}
|
||||
simdjson_inline simd8x64(const T ptr[64]) : chunks{simd8<T>::load(ptr), simd8<T>::load(ptr+32)} {}
|
||||
|
||||
simdjson_inline uint64_t compress(uint64_t mask, T * output) const {
|
||||
uint32_t mask1 = uint32_t(mask);
|
||||
uint32_t mask2 = uint32_t(mask >> 32);
|
||||
__m256i zcnt = __lasx_xvpcnt_w(__m256i(v4u64{~mask, 0, 0, 0}));
|
||||
uint64_t zcnt1 = __lasx_xvpickve2gr_wu(zcnt, 0);
|
||||
uint64_t zcnt2 = __lasx_xvpickve2gr_wu(zcnt, 1);
|
||||
// There should be a critical value which processes in scaler is faster.
|
||||
if (zcnt1)
|
||||
this->chunks[0].compress(mask1, output);
|
||||
if (zcnt2)
|
||||
this->chunks[1].compress(mask2, output + zcnt1);
|
||||
return zcnt1 + zcnt2;
|
||||
}
|
||||
|
||||
simdjson_inline void store(T ptr[64]) const {
|
||||
this->chunks[0].store(ptr+sizeof(simd8<T>)*0);
|
||||
this->chunks[1].store(ptr+sizeof(simd8<T>)*1);
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t to_bitmask() const {
|
||||
__m256i mask0 = __lasx_xvmskltz_b(this->chunks[0]);
|
||||
__m256i mask1 = __lasx_xvmskltz_b(this->chunks[1]);
|
||||
__m256i mask_tmp = __lasx_xvpickve_w(mask0, 4);
|
||||
__m256i tmp = __lasx_xvpickve_w(mask1, 4);
|
||||
mask0 = __lasx_xvinsve0_w(mask0, mask1, 1);
|
||||
mask_tmp = __lasx_xvinsve0_w(mask_tmp, tmp, 1);
|
||||
return __lasx_xvpickve2gr_du(__lasx_xvpackev_h(mask_tmp, mask0), 0);
|
||||
}
|
||||
|
||||
simdjson_inline simd8<T> reduce_or() const {
|
||||
return this->chunks[0] | this->chunks[1];
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t eq(const T m) const {
|
||||
const simd8<T> mask = simd8<T>::splat(m);
|
||||
return simd8x64<bool>(
|
||||
this->chunks[0] == mask,
|
||||
this->chunks[1] == mask
|
||||
).to_bitmask();
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t eq(const simd8x64<uint8_t> &other) const {
|
||||
return simd8x64<bool>(
|
||||
this->chunks[0] == other.chunks[0],
|
||||
this->chunks[1] == other.chunks[1]
|
||||
).to_bitmask();
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t lteq(const T m) const {
|
||||
const simd8<T> mask = simd8<T>::splat(m);
|
||||
return simd8x64<bool>(
|
||||
this->chunks[0] <= mask,
|
||||
this->chunks[1] <= mask
|
||||
).to_bitmask();
|
||||
}
|
||||
}; // struct simd8x64<T>
|
||||
|
||||
} // namespace simd
|
||||
} // unnamed namespace
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LASX_SIMD_H
|
||||
@@ -0,0 +1,47 @@
|
||||
#ifndef SIMDJSON_LASX_STRINGPARSING_DEFS_H
|
||||
#define SIMDJSON_LASX_STRINGPARSING_DEFS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lasx/base.h"
|
||||
#include "simdjson/lasx/simd.h"
|
||||
#include "simdjson/lasx/bitmanipulation.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lasx {
|
||||
namespace {
|
||||
|
||||
using namespace simd;
|
||||
|
||||
// Holds backslashes and quotes locations.
|
||||
struct backslash_and_quote {
|
||||
public:
|
||||
static constexpr uint32_t BYTES_PROCESSED = 32;
|
||||
simdjson_inline static backslash_and_quote copy_and_find(const uint8_t *src, uint8_t *dst);
|
||||
|
||||
simdjson_inline bool has_quote_first() { return ((bs_bits - 1) & quote_bits) != 0; }
|
||||
simdjson_inline bool has_backslash() { return bs_bits != 0; }
|
||||
simdjson_inline int quote_index() { return trailing_zeroes(quote_bits); }
|
||||
simdjson_inline int backslash_index() { return trailing_zeroes(bs_bits); }
|
||||
|
||||
uint32_t bs_bits;
|
||||
uint32_t quote_bits;
|
||||
}; // struct backslash_and_quote
|
||||
|
||||
simdjson_inline backslash_and_quote backslash_and_quote::copy_and_find(const uint8_t *src, uint8_t *dst) {
|
||||
// this can read up to 31 bytes beyond the buffer size, but we require
|
||||
// SIMDJSON_PADDING of padding
|
||||
static_assert(SIMDJSON_PADDING >= (BYTES_PROCESSED - 1), "backslash and quote finder must process fewer than SIMDJSON_PADDING bytes");
|
||||
simd8<uint8_t> v(src);
|
||||
v.store(dst);
|
||||
return {
|
||||
static_cast<uint32_t>((v == '\\').to_bitmask()), // bs_bits
|
||||
static_cast<uint32_t>((v == '"').to_bitmask()), // quote_bits
|
||||
};
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LASX_STRINGPARSING_DEFS_H
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SIMDJSON_LSX_H
|
||||
#define SIMDJSON_LSX_H
|
||||
|
||||
#include "simdjson/lsx/begin.h"
|
||||
#include "simdjson/generic/amalgamated.h"
|
||||
#include "simdjson/lsx/end.h"
|
||||
|
||||
#endif // SIMDJSON_LSX_H
|
||||
@@ -0,0 +1,26 @@
|
||||
#ifndef SIMDJSON_LSX_BASE_H
|
||||
#define SIMDJSON_LSX_BASE_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
/**
|
||||
* Implementation for LSX.
|
||||
*/
|
||||
namespace lsx {
|
||||
|
||||
class implementation;
|
||||
|
||||
namespace {
|
||||
namespace simd {
|
||||
template <typename T> struct simd8;
|
||||
template <typename T> struct simd8x64;
|
||||
} // namespace simd
|
||||
} // unnamed namespace
|
||||
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LSX_BASE_H
|
||||
@@ -0,0 +1,10 @@
|
||||
#define SIMDJSON_IMPLEMENTATION lsx
|
||||
#include "simdjson/lsx/base.h"
|
||||
#include "simdjson/lsx/intrinsics.h"
|
||||
#include "simdjson/lsx/bitmanipulation.h"
|
||||
#include "simdjson/lsx/bitmask.h"
|
||||
#include "simdjson/lsx/numberparsing_defs.h"
|
||||
#include "simdjson/lsx/simd.h"
|
||||
#include "simdjson/lsx/stringparsing_defs.h"
|
||||
|
||||
#define SIMDJSON_SKIP_BACKSLASH_SHORT_CIRCUIT 1
|
||||
@@ -0,0 +1,50 @@
|
||||
#ifndef SIMDJSON_LSX_BITMANIPULATION_H
|
||||
#define SIMDJSON_LSX_BITMANIPULATION_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lsx/base.h"
|
||||
#include "simdjson/lsx/intrinsics.h"
|
||||
#include "simdjson/lsx/bitmask.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lsx {
|
||||
namespace {
|
||||
|
||||
// We sometimes call trailing_zero on inputs that are zero,
|
||||
// but the algorithms do not end up using the returned value.
|
||||
// Sadly, sanitizers are not smart enough to figure it out.
|
||||
SIMDJSON_NO_SANITIZE_UNDEFINED
|
||||
// This function can be used safely even if not all bytes have been
|
||||
// initialized.
|
||||
// See issue https://github.com/simdjson/simdjson/issues/1965
|
||||
SIMDJSON_NO_SANITIZE_MEMORY
|
||||
simdjson_inline int trailing_zeroes(uint64_t input_num) {
|
||||
return __builtin_ctzll(input_num);
|
||||
}
|
||||
|
||||
/* result might be undefined when input_num is zero */
|
||||
simdjson_inline uint64_t clear_lowest_bit(uint64_t input_num) {
|
||||
return input_num & (input_num-1);
|
||||
}
|
||||
|
||||
/* result might be undefined when input_num is zero */
|
||||
simdjson_inline int leading_zeroes(uint64_t input_num) {
|
||||
return __builtin_clzll(input_num);
|
||||
}
|
||||
|
||||
/* result might be undefined when input_num is zero */
|
||||
simdjson_inline int count_ones(uint64_t input_num) {
|
||||
return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
|
||||
}
|
||||
|
||||
simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
|
||||
return __builtin_uaddll_overflow(value1, value2,
|
||||
reinterpret_cast<unsigned long long *>(result));
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LSX_BITMANIPULATION_H
|
||||
@@ -0,0 +1,31 @@
|
||||
#ifndef SIMDJSON_LSX_BITMASK_H
|
||||
#define SIMDJSON_LSX_BITMASK_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lsx/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lsx {
|
||||
namespace {
|
||||
|
||||
//
|
||||
// Perform a "cumulative bitwise xor," flipping bits each time a 1 is encountered.
|
||||
//
|
||||
// For example, prefix_xor(00100100) == 00011100
|
||||
//
|
||||
simdjson_inline uint64_t prefix_xor(uint64_t bitmask) {
|
||||
bitmask ^= bitmask << 1;
|
||||
bitmask ^= bitmask << 2;
|
||||
bitmask ^= bitmask << 4;
|
||||
bitmask ^= bitmask << 8;
|
||||
bitmask ^= bitmask << 16;
|
||||
bitmask ^= bitmask << 32;
|
||||
return bitmask;
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,6 @@
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lsx/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#undef SIMDJSON_SKIP_BACKSLASH_SHORT_CIRCUIT
|
||||
#undef SIMDJSON_IMPLEMENTATION
|
||||
@@ -0,0 +1,31 @@
|
||||
#ifndef SIMDJSON_LSX_IMPLEMENTATION_H
|
||||
#define SIMDJSON_LSX_IMPLEMENTATION_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/base.h"
|
||||
#include "simdjson/implementation.h"
|
||||
#include "simdjson/internal/instruction_set.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lsx {
|
||||
|
||||
/**
|
||||
* @private
|
||||
*/
|
||||
class implementation final : public simdjson::implementation {
|
||||
public:
|
||||
simdjson_inline implementation() : simdjson::implementation("lsx", "LoongArch SX", internal::instruction_set::LSX) {}
|
||||
simdjson_warn_unused error_code create_dom_parser_implementation(
|
||||
size_t capacity,
|
||||
size_t max_length,
|
||||
std::unique_ptr<internal::dom_parser_implementation>& dst
|
||||
) const noexcept final;
|
||||
simdjson_warn_unused error_code minify(const uint8_t *buf, size_t len, uint8_t *dst, size_t &dst_len) const noexcept final;
|
||||
simdjson_warn_unused bool validate_utf8(const char *buf, size_t len) const noexcept final;
|
||||
};
|
||||
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LSX_IMPLEMENTATION_H
|
||||
@@ -0,0 +1,14 @@
|
||||
#ifndef SIMDJSON_LSX_INTRINSICS_H
|
||||
#define SIMDJSON_LSX_INTRINSICS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lsx/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
// This should be the correct header whether
|
||||
// you use visual studio or other compilers.
|
||||
#include <lsxintrin.h>
|
||||
|
||||
static_assert(sizeof(__m128i) <= simdjson::SIMDJSON_PADDING, "insufficient padding for LoongArch SX");
|
||||
|
||||
#endif // SIMDJSON_LSX_INTRINSICS_H
|
||||
@@ -0,0 +1,41 @@
|
||||
#ifndef SIMDJSON_LSX_NUMBERPARSING_DEFS_H
|
||||
#define SIMDJSON_LSX_NUMBERPARSING_DEFS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lsx/base.h"
|
||||
#include "simdjson/lsx/intrinsics.h"
|
||||
#include "simdjson/internal/numberparsing_tables.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace simdjson {
|
||||
namespace lsx {
|
||||
namespace numberparsing {
|
||||
|
||||
// we don't have appropriate instructions, so let us use a scalar function
|
||||
// credit: https://johnnylee-sde.github.io/Fast-numeric-string-to-int/
|
||||
/** @private */
|
||||
static simdjson_inline uint32_t parse_eight_digits_unrolled(const uint8_t *chars) {
|
||||
uint64_t val;
|
||||
std::memcpy(&val, chars, sizeof(uint64_t));
|
||||
val = (val & 0x0F0F0F0F0F0F0F0F) * 2561 >> 8;
|
||||
val = (val & 0x00FF00FF00FF00FF) * 6553601 >> 16;
|
||||
return uint32_t((val & 0x0000FFFF0000FFFF) * 42949672960001 >> 32);
|
||||
}
|
||||
|
||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||
internal::value128 answer;
|
||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||
answer.low = uint64_t(r);
|
||||
answer.high = uint64_t(r >> 64);
|
||||
return answer;
|
||||
}
|
||||
|
||||
} // namespace numberparsing
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
||||
|
||||
#endif // SIMDJSON_LSX_NUMBERPARSING_DEFS_H
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SIMDJSON_LSX_ONDEMAND_H
|
||||
#define SIMDJSON_LSX_ONDEMAND_H
|
||||
|
||||
#include "simdjson/lsx/begin.h"
|
||||
#include "simdjson/generic/ondemand/amalgamated.h"
|
||||
#include "simdjson/lsx/end.h"
|
||||
|
||||
#endif // SIMDJSON_LSX_ONDEMAND_H
|
||||
@@ -0,0 +1,354 @@
|
||||
#ifndef SIMDJSON_LSX_SIMD_H
|
||||
#define SIMDJSON_LSX_SIMD_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lsx/base.h"
|
||||
#include "simdjson/lsx/bitmanipulation.h"
|
||||
#include "simdjson/internal/simdprune_tables.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lsx {
|
||||
namespace {
|
||||
namespace simd {
|
||||
|
||||
// Forward-declared so they can be used by splat and friends.
|
||||
template<typename Child>
|
||||
struct base {
|
||||
__m128i value;
|
||||
|
||||
// Zero constructor
|
||||
simdjson_inline base() : value{__m128i()} {}
|
||||
|
||||
// Conversion from SIMD register
|
||||
simdjson_inline base(const __m128i _value) : value(_value) {}
|
||||
|
||||
// Conversion to SIMD register
|
||||
simdjson_inline operator const __m128i&() const { return this->value; }
|
||||
simdjson_inline operator __m128i&() { return this->value; }
|
||||
simdjson_inline operator const v16i8&() const { return (v16i8&)this->value; }
|
||||
simdjson_inline operator v16i8&() { return (v16i8&)this->value; }
|
||||
|
||||
// Bit operations
|
||||
simdjson_inline Child operator|(const Child other) const { return __lsx_vor_v(*this, other); }
|
||||
simdjson_inline Child operator&(const Child other) const { return __lsx_vand_v(*this, other); }
|
||||
simdjson_inline Child operator^(const Child other) const { return __lsx_vxor_v(*this, other); }
|
||||
simdjson_inline Child bit_andnot(const Child other) const { return __lsx_vandn_v(other, *this); }
|
||||
simdjson_inline Child& operator|=(const Child other) { auto this_cast = static_cast<Child*>(this); *this_cast = *this_cast | other; return *this_cast; }
|
||||
simdjson_inline Child& operator&=(const Child other) { auto this_cast = static_cast<Child*>(this); *this_cast = *this_cast & other; return *this_cast; }
|
||||
simdjson_inline Child& operator^=(const Child other) { auto this_cast = static_cast<Child*>(this); *this_cast = *this_cast ^ other; return *this_cast; }
|
||||
};
|
||||
|
||||
// Forward-declared so they can be used by splat and friends.
|
||||
template<typename T>
|
||||
struct simd8;
|
||||
|
||||
template<typename T, typename Mask=simd8<bool>>
|
||||
struct base8: base<simd8<T>> {
|
||||
simdjson_inline base8() : base<simd8<T>>() {}
|
||||
simdjson_inline base8(const __m128i _value) : base<simd8<T>>(_value) {}
|
||||
|
||||
friend simdjson_really_inline Mask operator==(const simd8<T> lhs, const simd8<T> rhs) { return __lsx_vseq_b(lhs, rhs); }
|
||||
|
||||
static const int SIZE = sizeof(base<simd8<T>>::value);
|
||||
|
||||
template<int N=1>
|
||||
simdjson_inline simd8<T> prev(const simd8<T> prev_chunk) const {
|
||||
return __lsx_vor_v(__lsx_vbsll_v(*this, N), __lsx_vbsrl_v(prev_chunk, 16 - N));
|
||||
}
|
||||
};
|
||||
|
||||
// SIMD byte mask type (returned by things like eq and gt)
|
||||
template<>
|
||||
struct simd8<bool>: base8<bool> {
|
||||
static simdjson_inline simd8<bool> splat(bool _value) {
|
||||
return __lsx_vreplgr2vr_b(uint8_t(-(!!_value)));
|
||||
}
|
||||
|
||||
simdjson_inline simd8<bool>() : base8() {}
|
||||
simdjson_inline simd8<bool>(const __m128i _value) : base8<bool>(_value) {}
|
||||
// Splat constructor
|
||||
simdjson_inline simd8<bool>(bool _value) : base8<bool>(splat(_value)) {}
|
||||
|
||||
simdjson_inline int to_bitmask() const { return __lsx_vpickve2gr_w(__lsx_vmskltz_b(*this), 0); }
|
||||
simdjson_inline bool any() const { return 0 == __lsx_vpickve2gr_hu(__lsx_vmsknz_b(*this), 0); }
|
||||
simdjson_inline simd8<bool> operator~() const { return *this ^ true; }
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
struct base8_numeric: base8<T> {
|
||||
static simdjson_inline simd8<T> splat(T _value) { return __lsx_vreplgr2vr_b(_value); }
|
||||
static simdjson_inline simd8<T> zero() { return __lsx_vldi(0); }
|
||||
static simdjson_inline simd8<T> load(const T values[16]) {
|
||||
return __lsx_vld(reinterpret_cast<const __m128i *>(values), 0);
|
||||
}
|
||||
// Repeat 16 values as many times as necessary (usually for lookup tables)
|
||||
static simdjson_inline simd8<T> repeat_16(
|
||||
T v0, T v1, T v2, T v3, T v4, T v5, T v6, T v7,
|
||||
T v8, T v9, T v10, T v11, T v12, T v13, T v14, T v15
|
||||
) {
|
||||
return simd8<T>(
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
);
|
||||
}
|
||||
|
||||
simdjson_inline base8_numeric() : base8<T>() {}
|
||||
simdjson_inline base8_numeric(const __m128i _value) : base8<T>(_value) {}
|
||||
|
||||
// Store to array
|
||||
simdjson_inline void store(T dst[16]) const {
|
||||
return __lsx_vst(*this, reinterpret_cast<__m128i *>(dst), 0);
|
||||
}
|
||||
|
||||
// Addition/subtraction are the same for signed and unsigned
|
||||
simdjson_inline simd8<T> operator+(const simd8<T> other) const { return __lsx_vadd_b(*this, other); }
|
||||
simdjson_inline simd8<T> operator-(const simd8<T> other) const { return __lsx_vsub_b(*this, other); }
|
||||
simdjson_inline simd8<T>& operator+=(const simd8<T> other) { *this = *this + other; return *static_cast<simd8<T>*>(this); }
|
||||
simdjson_inline simd8<T>& operator-=(const simd8<T> other) { *this = *this - other; return *static_cast<simd8<T>*>(this); }
|
||||
|
||||
// Override to distinguish from bool version
|
||||
simdjson_inline simd8<T> operator~() const { return *this ^ 0xFFu; }
|
||||
|
||||
// Perform a lookup assuming the value is between 0 and 16 (undefined behavior for out of range values)
|
||||
template<typename L>
|
||||
simdjson_inline simd8<L> lookup_16(simd8<L> lookup_table) const {
|
||||
return __lsx_vshuf_b(lookup_table, lookup_table, *this);
|
||||
}
|
||||
|
||||
// Copies to 'output" all bytes corresponding to a 0 in the mask (interpreted as a bitset).
|
||||
// Passing a 0 value for mask would be equivalent to writing out every byte to output.
|
||||
// Only the first 16 - count_ones(mask) bytes of the result are significant but 16 bytes
|
||||
// get written.
|
||||
template<typename L>
|
||||
simdjson_inline void compress(uint16_t mask, L * output) const {
|
||||
using internal::thintable_epi8;
|
||||
using internal::BitsSetTable256mul2;
|
||||
using internal::pshufb_combine_table;
|
||||
// this particular implementation was inspired by haswell
|
||||
// lsx do it in 2 steps, first 8 bytes and then second 8 bytes...
|
||||
uint8_t mask1 = uint8_t(mask); // least significant 8 bits
|
||||
uint8_t mask2 = uint8_t(mask >> 8); // second least significant 8 bits
|
||||
// next line just loads the 64-bit values thintable_epi8[mask1] and
|
||||
// thintable_epi8[mask2] into a 128-bit register.
|
||||
__m128i shufmask = {int64_t(thintable_epi8[mask1]), int64_t(thintable_epi8[mask2]) + 0x0808080808080808};
|
||||
// this is the version "nearly pruned"
|
||||
__m128i pruned = __lsx_vshuf_b(*this, *this, shufmask);
|
||||
// we still need to put the pieces back together.
|
||||
// we compute the popcount of the first words:
|
||||
int pop1 = BitsSetTable256mul2[mask1];
|
||||
// then load the corresponding mask
|
||||
__m128i compactmask = __lsx_vldx(reinterpret_cast<void*>(reinterpret_cast<unsigned long>(pshufb_combine_table)), pop1 * 8);
|
||||
__m128i answer = __lsx_vshuf_b(pruned, pruned, compactmask);
|
||||
__lsx_vst(answer, reinterpret_cast<uint8_t*>(output), 0);
|
||||
}
|
||||
|
||||
template<typename L>
|
||||
simdjson_inline simd8<L> lookup_16(
|
||||
L replace0, L replace1, L replace2, L replace3,
|
||||
L replace4, L replace5, L replace6, L replace7,
|
||||
L replace8, L replace9, L replace10, L replace11,
|
||||
L replace12, L replace13, L replace14, L replace15) const {
|
||||
return lookup_16(simd8<L>::repeat_16(
|
||||
replace0, replace1, replace2, replace3,
|
||||
replace4, replace5, replace6, replace7,
|
||||
replace8, replace9, replace10, replace11,
|
||||
replace12, replace13, replace14, replace15
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
// Signed bytes
|
||||
template<>
|
||||
struct simd8<int8_t> : base8_numeric<int8_t> {
|
||||
simdjson_inline simd8() : base8_numeric<int8_t>() {}
|
||||
simdjson_inline simd8(const __m128i _value) : base8_numeric<int8_t>(_value) {}
|
||||
// Splat constructor
|
||||
simdjson_inline simd8(int8_t _value) : simd8(splat(_value)) {}
|
||||
// Array constructor
|
||||
simdjson_inline simd8(const int8_t values[16]) : simd8(load(values)) {}
|
||||
// Member-by-member initialization
|
||||
simdjson_inline simd8(
|
||||
int8_t v0, int8_t v1, int8_t v2, int8_t v3, int8_t v4, int8_t v5, int8_t v6, int8_t v7,
|
||||
int8_t v8, int8_t v9, int8_t v10, int8_t v11, int8_t v12, int8_t v13, int8_t v14, int8_t v15
|
||||
) : simd8({
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
}) {}
|
||||
// Repeat 16 values as many times as necessary (usually for lookup tables)
|
||||
simdjson_inline static simd8<int8_t> repeat_16(
|
||||
int8_t v0, int8_t v1, int8_t v2, int8_t v3, int8_t v4, int8_t v5, int8_t v6, int8_t v7,
|
||||
int8_t v8, int8_t v9, int8_t v10, int8_t v11, int8_t v12, int8_t v13, int8_t v14, int8_t v15
|
||||
) {
|
||||
return simd8<int8_t>(
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
);
|
||||
}
|
||||
|
||||
// Order-sensitive comparisons
|
||||
simdjson_inline simd8<int8_t> max_val(const simd8<int8_t> other) const { return __lsx_vmax_b(*this, other); }
|
||||
simdjson_inline simd8<int8_t> min_val(const simd8<int8_t> other) const { return __lsx_vmin_b(*this, other); }
|
||||
simdjson_inline simd8<bool> operator>(const simd8<int8_t> other) const { return __lsx_vslt_b(other, *this); }
|
||||
simdjson_inline simd8<bool> operator<(const simd8<int8_t> other) const { return __lsx_vslt_b(*this, other); }
|
||||
};
|
||||
|
||||
// Unsigned bytes
|
||||
template<>
|
||||
struct simd8<uint8_t>: base8_numeric<uint8_t> {
|
||||
simdjson_inline simd8() : base8_numeric<uint8_t>() {}
|
||||
simdjson_inline simd8(const __m128i _value) : base8_numeric<uint8_t>(_value) {}
|
||||
// Splat constructor
|
||||
simdjson_inline simd8(uint8_t _value) : simd8(splat(_value)) {}
|
||||
// Array constructor
|
||||
simdjson_inline simd8(const uint8_t values[16]) : simd8(load(values)) {}
|
||||
// Member-by-member initialization
|
||||
simdjson_inline simd8(
|
||||
uint8_t v0, uint8_t v1, uint8_t v2, uint8_t v3, uint8_t v4, uint8_t v5, uint8_t v6, uint8_t v7,
|
||||
uint8_t v8, uint8_t v9, uint8_t v10, uint8_t v11, uint8_t v12, uint8_t v13, uint8_t v14, uint8_t v15
|
||||
) : simd8(__m128i(v16u8{
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
})) {}
|
||||
// Repeat 16 values as many times as necessary (usually for lookup tables)
|
||||
simdjson_inline static simd8<uint8_t> repeat_16(
|
||||
uint8_t v0, uint8_t v1, uint8_t v2, uint8_t v3, uint8_t v4, uint8_t v5, uint8_t v6, uint8_t v7,
|
||||
uint8_t v8, uint8_t v9, uint8_t v10, uint8_t v11, uint8_t v12, uint8_t v13, uint8_t v14, uint8_t v15
|
||||
) {
|
||||
return simd8<uint8_t>(
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
);
|
||||
}
|
||||
|
||||
// Saturated math
|
||||
simdjson_inline simd8<uint8_t> saturating_add(const simd8<uint8_t> other) const { return __lsx_vsadd_bu(*this, other); }
|
||||
simdjson_inline simd8<uint8_t> saturating_sub(const simd8<uint8_t> other) const { return __lsx_vssub_bu(*this, other); }
|
||||
|
||||
// Order-specific operations
|
||||
simdjson_inline simd8<uint8_t> max_val(const simd8<uint8_t> other) const { return __lsx_vmax_bu(*this, other); }
|
||||
simdjson_inline simd8<uint8_t> min_val(const simd8<uint8_t> other) const { return __lsx_vmin_bu(other, *this); }
|
||||
// Same as >, but only guarantees true is nonzero (< guarantees true = -1)
|
||||
simdjson_inline simd8<uint8_t> gt_bits(const simd8<uint8_t> other) const { return this->saturating_sub(other); }
|
||||
// Same as <, but only guarantees true is nonzero (< guarantees true = -1)
|
||||
simdjson_inline simd8<uint8_t> lt_bits(const simd8<uint8_t> other) const { return other.saturating_sub(*this); }
|
||||
simdjson_inline simd8<bool> operator<=(const simd8<uint8_t> other) const { return other.max_val(*this) == other; }
|
||||
simdjson_inline simd8<bool> operator>=(const simd8<uint8_t> other) const { return other.min_val(*this) == other; }
|
||||
simdjson_inline simd8<bool> operator>(const simd8<uint8_t> other) const { return this->gt_bits(other).any_bits_set(); }
|
||||
simdjson_inline simd8<bool> operator<(const simd8<uint8_t> other) const { return this->lt_bits(other).any_bits_set(); }
|
||||
|
||||
// Bit-specific operations
|
||||
simdjson_inline simd8<bool> bits_not_set() const { return *this == uint8_t(0); }
|
||||
simdjson_inline simd8<bool> bits_not_set(simd8<uint8_t> bits) const { return (*this & bits).bits_not_set(); }
|
||||
simdjson_inline simd8<bool> any_bits_set() const { return ~this->bits_not_set(); }
|
||||
simdjson_inline simd8<bool> any_bits_set(simd8<uint8_t> bits) const { return ~this->bits_not_set(bits); }
|
||||
simdjson_inline bool is_ascii() const { return 0 == __lsx_vpickve2gr_w(__lsx_vmskltz_b(*this), 0); }
|
||||
simdjson_inline bool bits_not_set_anywhere() const { return 0 == __lsx_vpickve2gr_hu(__lsx_vmsknz_b(*this), 0); }
|
||||
simdjson_inline bool any_bits_set_anywhere() const { return !bits_not_set_anywhere(); }
|
||||
simdjson_inline bool bits_not_set_anywhere(simd8<uint8_t> bits) const {
|
||||
return 0 == __lsx_vpickve2gr_hu(__lsx_vmsknz_b(__lsx_vand_v(*this, bits)), 0);
|
||||
}
|
||||
simdjson_inline bool any_bits_set_anywhere(simd8<uint8_t> bits) const { return !bits_not_set_anywhere(bits); }
|
||||
template<int N>
|
||||
simdjson_inline simd8<uint8_t> shr() const { return simd8<uint8_t>(__lsx_vsrli_b(*this, N)); }
|
||||
template<int N>
|
||||
simdjson_inline simd8<uint8_t> shl() const { return simd8<uint8_t>(__lsx_vslli_b(*this, N)); }
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
struct simd8x64 {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<T>);
|
||||
static_assert(NUM_CHUNKS == 4, "LSX kernel should use four registers per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
|
||||
simd8x64(const simd8x64<T>& o) = delete; // no copy allowed
|
||||
simd8x64<T>& operator=(const simd8<T>& other) = delete; // no assignment allowed
|
||||
simd8x64() = delete; // no default constructor allowed
|
||||
|
||||
simdjson_inline simd8x64(const simd8<T> chunk0, const simd8<T> chunk1, const simd8<T> chunk2, const simd8<T> chunk3) : chunks{chunk0, chunk1, chunk2, chunk3} {}
|
||||
simdjson_inline simd8x64(const T ptr[64]) : chunks{simd8<T>::load(ptr), simd8<T>::load(ptr+16), simd8<T>::load(ptr+32), simd8<T>::load(ptr+48)} {}
|
||||
|
||||
simdjson_inline uint64_t compress(uint64_t mask, T * output) const {
|
||||
uint16_t mask1 = uint16_t(mask);
|
||||
uint16_t mask2 = uint16_t(mask >> 16);
|
||||
uint16_t mask3 = uint16_t(mask >> 32);
|
||||
uint16_t mask4 = uint16_t(mask >> 48);
|
||||
__m128i zcnt = __lsx_vpcnt_h(__m128i(v2u64{~mask, 0}));
|
||||
uint64_t zcnt1 = __lsx_vpickve2gr_hu(zcnt, 0);
|
||||
uint64_t zcnt2 = __lsx_vpickve2gr_hu(zcnt, 1);
|
||||
uint64_t zcnt3 = __lsx_vpickve2gr_hu(zcnt, 2);
|
||||
uint64_t zcnt4 = __lsx_vpickve2gr_hu(zcnt, 3);
|
||||
uint8_t *voutput = reinterpret_cast<uint8_t*>(output);
|
||||
// There should be a critical value which processes in scaler is faster.
|
||||
if (zcnt1)
|
||||
this->chunks[0].compress(mask1, reinterpret_cast<T*>(voutput));
|
||||
voutput += zcnt1;
|
||||
if (zcnt2)
|
||||
this->chunks[1].compress(mask2, reinterpret_cast<T*>(voutput));
|
||||
voutput += zcnt2;
|
||||
if (zcnt3)
|
||||
this->chunks[2].compress(mask3, reinterpret_cast<T*>(voutput));
|
||||
voutput += zcnt3;
|
||||
if (zcnt4)
|
||||
this->chunks[3].compress(mask4, reinterpret_cast<T*>(voutput));
|
||||
voutput += zcnt4;
|
||||
return reinterpret_cast<uint64_t>(voutput) - reinterpret_cast<uint64_t>(output);
|
||||
}
|
||||
|
||||
simdjson_inline void store(T ptr[64]) const {
|
||||
this->chunks[0].store(ptr+sizeof(simd8<T>)*0);
|
||||
this->chunks[1].store(ptr+sizeof(simd8<T>)*1);
|
||||
this->chunks[2].store(ptr+sizeof(simd8<T>)*2);
|
||||
this->chunks[3].store(ptr+sizeof(simd8<T>)*3);
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t to_bitmask() const {
|
||||
__m128i mask1 = __lsx_vmskltz_b(this->chunks[0]);
|
||||
__m128i mask2 = __lsx_vmskltz_b(this->chunks[1]);
|
||||
__m128i mask3 = __lsx_vmskltz_b(this->chunks[2]);
|
||||
__m128i mask4 = __lsx_vmskltz_b(this->chunks[3]);
|
||||
mask1 = __lsx_vilvl_h(mask2, mask1);
|
||||
mask2 = __lsx_vilvl_h(mask4, mask3);
|
||||
return __lsx_vpickve2gr_du(__lsx_vilvl_w(mask2, mask1), 0);
|
||||
}
|
||||
|
||||
simdjson_inline simd8<T> reduce_or() const {
|
||||
return (this->chunks[0] | this->chunks[1]) | (this->chunks[2] | this->chunks[3]);
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t eq(const T m) const {
|
||||
const simd8<T> mask = simd8<T>::splat(m);
|
||||
return simd8x64<bool>(
|
||||
this->chunks[0] == mask,
|
||||
this->chunks[1] == mask,
|
||||
this->chunks[2] == mask,
|
||||
this->chunks[3] == mask
|
||||
).to_bitmask();
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t eq(const simd8x64<uint8_t> &other) const {
|
||||
return simd8x64<bool>(
|
||||
this->chunks[0] == other.chunks[0],
|
||||
this->chunks[1] == other.chunks[1],
|
||||
this->chunks[2] == other.chunks[2],
|
||||
this->chunks[3] == other.chunks[3]
|
||||
).to_bitmask();
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t lteq(const T m) const {
|
||||
const simd8<T> mask = simd8<T>::splat(m);
|
||||
return simd8x64<bool>(
|
||||
this->chunks[0] <= mask,
|
||||
this->chunks[1] <= mask,
|
||||
this->chunks[2] <= mask,
|
||||
this->chunks[3] <= mask
|
||||
).to_bitmask();
|
||||
}
|
||||
}; // struct simd8x64<T>
|
||||
|
||||
} // namespace simd
|
||||
} // unnamed namespace
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LSX_SIMD_H
|
||||
@@ -0,0 +1,53 @@
|
||||
#ifndef SIMDJSON_LSX_STRINGPARSING_DEFS_H
|
||||
#define SIMDJSON_LSX_STRINGPARSING_DEFS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/lsx/base.h"
|
||||
#include "simdjson/lsx/simd.h"
|
||||
#include "simdjson/lsx/bitmanipulation.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace lsx {
|
||||
namespace {
|
||||
|
||||
using namespace simd;
|
||||
|
||||
// Holds backslashes and quotes locations.
|
||||
struct backslash_and_quote {
|
||||
public:
|
||||
static constexpr uint32_t BYTES_PROCESSED = 32;
|
||||
simdjson_inline static backslash_and_quote copy_and_find(const uint8_t *src, uint8_t *dst);
|
||||
|
||||
simdjson_inline bool has_quote_first() { return ((bs_bits - 1) & quote_bits) != 0; }
|
||||
simdjson_inline bool has_backslash() { return bs_bits != 0; }
|
||||
simdjson_inline int quote_index() { return trailing_zeroes(quote_bits); }
|
||||
simdjson_inline int backslash_index() { return trailing_zeroes(bs_bits); }
|
||||
|
||||
uint32_t bs_bits;
|
||||
uint32_t quote_bits;
|
||||
}; // struct backslash_and_quote
|
||||
|
||||
simdjson_inline backslash_and_quote backslash_and_quote::copy_and_find(const uint8_t *src, uint8_t *dst) {
|
||||
// this can read up to 31 bytes beyond the buffer size, but we require
|
||||
// SIMDJSON_PADDING of padding
|
||||
static_assert(SIMDJSON_PADDING >= (BYTES_PROCESSED - 1), "backslash and quote finder must process fewer than SIMDJSON_PADDING bytes");
|
||||
simd8<uint8_t> v0(src);
|
||||
simd8<uint8_t> v1(src + sizeof(v0));
|
||||
v0.store(dst);
|
||||
v1.store(dst + sizeof(v0));
|
||||
|
||||
// Getting a 64-bit bitmask is much cheaper than multiple 16-bit bitmasks on LSX; therefore, we
|
||||
// smash them together into a 64-byte mask and get the bitmask from there.
|
||||
uint64_t bs_and_quote = simd8x64<bool>(v0 == '\\', v1 == '\\', v0 == '"', v1 == '"').to_bitmask();
|
||||
return {
|
||||
uint32_t(bs_and_quote), // bs_bits
|
||||
uint32_t(bs_and_quote >> 32) // quote_bits
|
||||
};
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_LSX_STRINGPARSING_DEFS_H
|
||||
@@ -38,8 +38,8 @@ else:
|
||||
|
||||
RelativeRoot = str # Literal['src','include'] # Literal not supported in Python 3.7 (CI)
|
||||
RELATIVE_ROOTS: List[RelativeRoot] = ['src', 'include' ]
|
||||
Implementation = str # Literal['arm64', 'fallback', 'haswell', 'icelake', 'ppc64', 'westmere'] # Literal not supported in Python 3.7 (CI)
|
||||
IMPLEMENTATIONS: List[Implementation] = [ 'arm64', 'haswell', 'icelake', 'ppc64', 'westmere', 'fallback' ]
|
||||
Implementation = str # Literal['arm64', 'fallback', 'haswell', 'icelake', 'ppc64', 'westmere', 'lsx', 'lasx'] # Literal not supported in Python 3.7 (CI)
|
||||
IMPLEMENTATIONS: List[Implementation] = [ 'arm64', 'haswell', 'icelake', 'lasx', 'lsx', 'ppc64', 'westmere', 'fallback' ]
|
||||
GENERIC_INCLUDE = "simdjson/generic"
|
||||
GENERIC_SRC = "generic"
|
||||
BUILTIN = "simdjson/builtin"
|
||||
|
||||
@@ -93,6 +93,30 @@ static const simdjson::westmere::implementation* get_westmere_singleton() {
|
||||
} // namespace simdjson
|
||||
#endif // SIMDJSON_IMPLEMENTATION_WESTMERE
|
||||
|
||||
#if SIMDJSON_IMPLEMENTATION_LSX
|
||||
#include <simdjson/lsx/implementation.h>
|
||||
namespace simdjson {
|
||||
namespace internal {
|
||||
static const simdjson::lsx::implementation* get_lsx_singleton() {
|
||||
static const simdjson::lsx::implementation lsx_singleton{};
|
||||
return &lsx_singleton;
|
||||
}
|
||||
} // namespace internal
|
||||
} // namespace simdjson
|
||||
#endif // SIMDJSON_IMPLEMENTATION_LSX
|
||||
|
||||
#if SIMDJSON_IMPLEMENTATION_LASX
|
||||
#include <simdjson/lasx/implementation.h>
|
||||
namespace simdjson {
|
||||
namespace internal {
|
||||
static const simdjson::lasx::implementation* get_lasx_singleton() {
|
||||
static const simdjson::lasx::implementation lasx_singleton{};
|
||||
return &lasx_singleton;
|
||||
}
|
||||
} // namespace internal
|
||||
} // namespace simdjson
|
||||
#endif // SIMDJSON_IMPLEMENTATION_LASX
|
||||
|
||||
#undef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
@@ -104,6 +128,7 @@ namespace internal {
|
||||
#define SIMDJSON_SINGLE_IMPLEMENTATION (SIMDJSON_IMPLEMENTATION_ICELAKE \
|
||||
+ SIMDJSON_IMPLEMENTATION_HASWELL + SIMDJSON_IMPLEMENTATION_WESTMERE \
|
||||
+ SIMDJSON_IMPLEMENTATION_ARM64 + SIMDJSON_IMPLEMENTATION_PPC64 \
|
||||
+ SIMDJSON_IMPLEMENTATION_LSX + SIMDJSON_IMPLEMENTATION_LASX \
|
||||
+ SIMDJSON_IMPLEMENTATION_FALLBACK == 1)
|
||||
|
||||
#if SIMDJSON_SINGLE_IMPLEMENTATION
|
||||
@@ -124,6 +149,12 @@ namespace internal {
|
||||
#if SIMDJSON_IMPLEMENTATION_PPC64
|
||||
get_ppc64_singleton();
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_LSX
|
||||
get_lsx_singleton();
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_LASX
|
||||
get_lasx_singleton();
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
get_fallback_singleton();
|
||||
#endif
|
||||
@@ -176,6 +207,12 @@ static const std::initializer_list<const implementation *>& get_available_implem
|
||||
#if SIMDJSON_IMPLEMENTATION_PPC64
|
||||
get_ppc64_singleton(),
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_LSX
|
||||
get_lsx_singleton(),
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_LASX
|
||||
get_lasx_singleton(),
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
get_fallback_singleton(),
|
||||
#endif
|
||||
|
||||
@@ -218,6 +218,19 @@ static inline uint32_t detect_supported_architectures() {
|
||||
|
||||
return host_isa;
|
||||
}
|
||||
|
||||
#elif defined(__loongarch_sx) && !defined(__loongarch_asx)
|
||||
|
||||
static inline uint32_t detect_supported_architectures() {
|
||||
return instruction_set::LSX;
|
||||
}
|
||||
|
||||
#elif defined(__loongarch_asx)
|
||||
|
||||
static inline uint32_t detect_supported_architectures() {
|
||||
return instruction_set::LASX;
|
||||
}
|
||||
|
||||
#else // fallback
|
||||
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
|
||||
#include <simdjson/implementation_detection.h>
|
||||
|
||||
#if SIMDJSON_IMPLEMENTATION_ARM64 || SIMDJSON_IMPLEMENTATION_ICELAKE || SIMDJSON_IMPLEMENTATION_HASWELL || SIMDJSON_IMPLEMENTATION_WESTMERE || SIMDJSON_IMPLEMENTATION_PPC64
|
||||
#if SIMDJSON_IMPLEMENTATION_ARM64 || SIMDJSON_IMPLEMENTATION_ICELAKE || SIMDJSON_IMPLEMENTATION_HASWELL || SIMDJSON_IMPLEMENTATION_WESTMERE || SIMDJSON_IMPLEMENTATION_PPC64 || SIMDJSON_IMPLEMENTATION_LSX || SIMDJSON_IMPLEMENTATION_LASX
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
@@ -133,6 +133,6 @@ SIMDJSON_DLLIMPORTEXPORT const uint64_t thintable_epi8[256] = {
|
||||
} // namespace internal
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_IMPLEMENTATION_ARM64 || SIMDJSON_IMPLEMENTATION_ICELAKE || SIMDJSON_IMPLEMENTATION_HASWELL || SIMDJSON_IMPLEMENTATION_WESTMERE || SIMDJSON_IMPLEMENTATION_PPC64
|
||||
#endif // SIMDJSON_IMPLEMENTATION_ARM64 || SIMDJSON_IMPLEMENTATION_ICELAKE || SIMDJSON_IMPLEMENTATION_HASWELL || SIMDJSON_IMPLEMENTATION_WESTMERE || SIMDJSON_IMPLEMENTATION_PPC64 || SIMDJSON_IMPLEMENTATION_LSX || SIMDJSON_IMPLEMENTATION_LASX
|
||||
|
||||
#endif // SIMDJSON_SRC_SIMDPRUNE_TABLES_CPP
|
||||
|
||||
+132
@@ -0,0 +1,132 @@
|
||||
#ifndef SIMDJSON_SRC_LASX_CPP
|
||||
#define SIMDJSON_SRC_LASX_CPP
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include <base.h>
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#include <simdjson/lasx.h>
|
||||
#include <simdjson/lasx/implementation.h>
|
||||
|
||||
#include <simdjson/lasx/begin.h>
|
||||
#include <generic/amalgamated.h>
|
||||
#include <generic/stage1/amalgamated.h>
|
||||
#include <generic/stage2/amalgamated.h>
|
||||
|
||||
//
|
||||
// Stage 1
|
||||
//
|
||||
namespace simdjson {
|
||||
namespace lasx {
|
||||
|
||||
simdjson_warn_unused error_code implementation::create_dom_parser_implementation(
|
||||
size_t capacity,
|
||||
size_t max_depth,
|
||||
std::unique_ptr<internal::dom_parser_implementation>& dst
|
||||
) const noexcept {
|
||||
dst.reset( new (std::nothrow) dom_parser_implementation() );
|
||||
if (!dst) { return MEMALLOC; }
|
||||
if (auto err = dst->set_capacity(capacity))
|
||||
return err;
|
||||
if (auto err = dst->set_max_depth(max_depth))
|
||||
return err;
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
namespace {
|
||||
|
||||
using namespace simd;
|
||||
|
||||
simdjson_inline json_character_block json_character_block::classify(const simd::simd8x64<uint8_t>& in) {
|
||||
// Inspired by haswell.
|
||||
// LASX use low 5 bits as index. For the 6 operators (:,[]{}), the unique-5bits is [6:2].
|
||||
// The ASCII white-space and operators have these values: (char, hex, unique-5bits)
|
||||
// (' ', 20, 00000) ('\t', 09, 01001) ('\n', 0A, 01010) ('\r', 0D, 01101)
|
||||
// (',', 2C, 01011) (':', 3A, 01110) ('[', 5B, 10110) ('{', 7B, 11110) (']', 5D, 10111) ('}', 7D, 11111)
|
||||
const simd8<uint8_t> ws_table = simd8<uint8_t>::repeat_16(
|
||||
' ', 0, 0, 0, 0, 0, 0, 0, 0, '\t', '\n', 0, 0, '\r', 0, 0
|
||||
);
|
||||
const simd8<uint8_t> op_table_lo = simd8<uint8_t>::repeat_16(
|
||||
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, ',', 0, 0, ':', 0
|
||||
);
|
||||
const simd8<uint8_t> op_table_hi = simd8<uint8_t>::repeat_16(
|
||||
0, 0, 0, 0, 0, 0, '[', ']', 0, 0, 0, 0, 0, 0, '{', '}'
|
||||
);
|
||||
uint64_t ws = in.eq({
|
||||
in.chunks[0].lookup_16(ws_table),
|
||||
in.chunks[1].lookup_16(ws_table),
|
||||
});
|
||||
uint64_t op = in.eq({
|
||||
__lasx_xvshuf_b(op_table_hi, op_table_lo, in.chunks[0].shr<2>()),
|
||||
__lasx_xvshuf_b(op_table_hi, op_table_lo, in.chunks[1].shr<2>()),
|
||||
});
|
||||
|
||||
return { ws, op };
|
||||
}
|
||||
|
||||
simdjson_inline bool is_ascii(const simd8x64<uint8_t>& input) {
|
||||
return input.reduce_or().is_ascii();
|
||||
}
|
||||
|
||||
simdjson_inline simd8<uint8_t> must_be_2_3_continuation(const simd8<uint8_t> prev2, const simd8<uint8_t> prev3) {
|
||||
simd8<uint8_t> is_third_byte = prev2.saturating_sub(0xe0u-0x80); // Only 111_____ will be >= 0x80
|
||||
simd8<uint8_t> is_fourth_byte = prev3.saturating_sub(0xf0u-0x80); // Only 1111____ will be >= 0x80
|
||||
return is_third_byte | is_fourth_byte;
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
//
|
||||
// Stage 2
|
||||
//
|
||||
|
||||
//
|
||||
// Implementation-specific overrides
|
||||
//
|
||||
namespace simdjson {
|
||||
namespace lasx {
|
||||
|
||||
simdjson_warn_unused error_code implementation::minify(const uint8_t *buf, size_t len, uint8_t *dst, size_t &dst_len) const noexcept {
|
||||
return lasx::stage1::json_minifier::minify<64>(buf, len, dst, dst_len);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode streaming) noexcept {
|
||||
this->buf = _buf;
|
||||
this->len = _len;
|
||||
return lasx::stage1::json_structural_indexer::index<64>(buf, len, *this, streaming);
|
||||
}
|
||||
|
||||
simdjson_warn_unused bool implementation::validate_utf8(const char *buf, size_t len) const noexcept {
|
||||
return lasx::stage1::generic_validate_utf8(buf,len);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage2(dom::document &_doc) noexcept {
|
||||
return stage2::tape_builder::parse_document<false>(*this, _doc);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage2_next(dom::document &_doc) noexcept {
|
||||
return stage2::tape_builder::parse_document<true>(*this, _doc);
|
||||
}
|
||||
|
||||
simdjson_warn_unused uint8_t *dom_parser_implementation::parse_string(const uint8_t *src, uint8_t *dst, bool allow_replacement) const noexcept {
|
||||
return lasx::stringparsing::parse_string(src, dst, allow_replacement);
|
||||
}
|
||||
|
||||
simdjson_warn_unused uint8_t *dom_parser_implementation::parse_wobbly_string(const uint8_t *src, uint8_t *dst) const noexcept {
|
||||
return lasx::stringparsing::parse_wobbly_string(src, dst);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::parse(const uint8_t *_buf, size_t _len, dom::document &_doc) noexcept {
|
||||
auto error = stage1(_buf, _len, stage1_mode::regular);
|
||||
if (error) { return error; }
|
||||
return stage2(_doc);
|
||||
}
|
||||
|
||||
} // namespace lasx
|
||||
} // namespace simdjson
|
||||
|
||||
#include <simdjson/lasx/end.h>
|
||||
|
||||
#endif // SIMDJSON_SRC_LASX_CPP
|
||||
+136
@@ -0,0 +1,136 @@
|
||||
#ifndef SIMDJSON_SRC_LSX_CPP
|
||||
#define SIMDJSON_SRC_LSX_CPP
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include <base.h>
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#include <simdjson/lsx.h>
|
||||
#include <simdjson/lsx/implementation.h>
|
||||
|
||||
#include <simdjson/lsx/begin.h>
|
||||
#include <generic/amalgamated.h>
|
||||
#include <generic/stage1/amalgamated.h>
|
||||
#include <generic/stage2/amalgamated.h>
|
||||
|
||||
//
|
||||
// Stage 1
|
||||
//
|
||||
namespace simdjson {
|
||||
namespace lsx {
|
||||
|
||||
simdjson_warn_unused error_code implementation::create_dom_parser_implementation(
|
||||
size_t capacity,
|
||||
size_t max_depth,
|
||||
std::unique_ptr<internal::dom_parser_implementation>& dst
|
||||
) const noexcept {
|
||||
dst.reset( new (std::nothrow) dom_parser_implementation() );
|
||||
if (!dst) { return MEMALLOC; }
|
||||
if (auto err = dst->set_capacity(capacity))
|
||||
return err;
|
||||
if (auto err = dst->set_max_depth(max_depth))
|
||||
return err;
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
namespace {
|
||||
|
||||
using namespace simd;
|
||||
|
||||
simdjson_inline json_character_block json_character_block::classify(const simd::simd8x64<uint8_t>& in) {
|
||||
// Inspired by haswell.
|
||||
// LSX use low 5 bits as index. For the 6 operators (:,[]{}), the unique-5bits is [6:2].
|
||||
// The ASCII white-space and operators have these values: (char, hex, unique-5bits)
|
||||
// (' ', 20, 00000) ('\t', 09, 01001) ('\n', 0A, 01010) ('\r', 0D, 01101)
|
||||
// (',', 2C, 01011) (':', 3A, 01110) ('[', 5B, 10110) ('{', 7B, 11110) (']', 5D, 10111) ('}', 7D, 11111)
|
||||
const simd8<uint8_t> ws_table = simd8<uint8_t>::repeat_16(
|
||||
' ', 0, 0, 0, 0, 0, 0, 0, 0, '\t', '\n', 0, 0, '\r', 0, 0
|
||||
);
|
||||
const simd8<uint8_t> op_table_lo = simd8<uint8_t>::repeat_16(
|
||||
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, ',', 0, 0, ':', 0
|
||||
);
|
||||
const simd8<uint8_t> op_table_hi = simd8<uint8_t>::repeat_16(
|
||||
0, 0, 0, 0, 0, 0, '[', ']', 0, 0, 0, 0, 0, 0, '{', '}'
|
||||
);
|
||||
uint64_t ws = in.eq({
|
||||
in.chunks[0].lookup_16(ws_table),
|
||||
in.chunks[1].lookup_16(ws_table),
|
||||
in.chunks[2].lookup_16(ws_table),
|
||||
in.chunks[3].lookup_16(ws_table)
|
||||
});
|
||||
uint64_t op = in.eq({
|
||||
__lsx_vshuf_b(op_table_hi, op_table_lo, in.chunks[0].shr<2>()),
|
||||
__lsx_vshuf_b(op_table_hi, op_table_lo, in.chunks[1].shr<2>()),
|
||||
__lsx_vshuf_b(op_table_hi, op_table_lo, in.chunks[2].shr<2>()),
|
||||
__lsx_vshuf_b(op_table_hi, op_table_lo, in.chunks[3].shr<2>())
|
||||
});
|
||||
|
||||
return { ws, op };
|
||||
}
|
||||
|
||||
simdjson_inline bool is_ascii(const simd8x64<uint8_t>& input) {
|
||||
return input.reduce_or().is_ascii();
|
||||
}
|
||||
|
||||
simdjson_inline simd8<uint8_t> must_be_2_3_continuation(const simd8<uint8_t> prev2, const simd8<uint8_t> prev3) {
|
||||
simd8<uint8_t> is_third_byte = prev2.saturating_sub(0xe0u-0x80); // Only 111_____ will be >= 0x80
|
||||
simd8<uint8_t> is_fourth_byte = prev3.saturating_sub(0xf0u-0x80); // Only 1111____ will be >= 0x80
|
||||
return is_third_byte | is_fourth_byte;
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
//
|
||||
// Stage 2
|
||||
//
|
||||
|
||||
//
|
||||
// Implementation-specific overrides
|
||||
//
|
||||
namespace simdjson {
|
||||
namespace lsx {
|
||||
|
||||
simdjson_warn_unused error_code implementation::minify(const uint8_t *buf, size_t len, uint8_t *dst, size_t &dst_len) const noexcept {
|
||||
return lsx::stage1::json_minifier::minify<64>(buf, len, dst, dst_len);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode streaming) noexcept {
|
||||
this->buf = _buf;
|
||||
this->len = _len;
|
||||
return lsx::stage1::json_structural_indexer::index<64>(buf, len, *this, streaming);
|
||||
}
|
||||
|
||||
simdjson_warn_unused bool implementation::validate_utf8(const char *buf, size_t len) const noexcept {
|
||||
return lsx::stage1::generic_validate_utf8(buf,len);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage2(dom::document &_doc) noexcept {
|
||||
return stage2::tape_builder::parse_document<false>(*this, _doc);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage2_next(dom::document &_doc) noexcept {
|
||||
return stage2::tape_builder::parse_document<true>(*this, _doc);
|
||||
}
|
||||
|
||||
simdjson_warn_unused uint8_t *dom_parser_implementation::parse_string(const uint8_t *src, uint8_t *dst, bool allow_replacement) const noexcept {
|
||||
return lsx::stringparsing::parse_string(src, dst, allow_replacement);
|
||||
}
|
||||
|
||||
simdjson_warn_unused uint8_t *dom_parser_implementation::parse_wobbly_string(const uint8_t *src, uint8_t *dst) const noexcept {
|
||||
return lsx::stringparsing::parse_wobbly_string(src, dst);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::parse(const uint8_t *_buf, size_t _len, dom::document &_doc) noexcept {
|
||||
auto error = stage1(_buf, _len, stage1_mode::regular);
|
||||
if (error) { return error; }
|
||||
return stage2(_doc);
|
||||
}
|
||||
|
||||
} // namespace lsx
|
||||
} // namespace simdjson
|
||||
|
||||
#include <simdjson/lsx/end.h>
|
||||
|
||||
#endif // SIMDJSON_SRC_LSX_CPP
|
||||
@@ -35,6 +35,12 @@ SIMDJSON_PUSH_DISABLE_UNUSED_WARNINGS
|
||||
#if SIMDJSON_IMPLEMENTATION_WESTMERE
|
||||
#include <westmere.cpp>
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_LSX
|
||||
#include <lsx.cpp>
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_LASX
|
||||
#include <lasx.cpp>
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
#include <fallback.cpp>
|
||||
#endif
|
||||
|
||||
Reference in New Issue
Block a user