/// @file dpf/utils.hpp /// @brief miscellaneous helper functions, structs, preprocessor directives /// @details Type traits, bit lengths, and small tuple helpers shared by the headers. /// @author Ryan Henry /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license; /// see [LICENSE.md](@ref license) for details. #ifndef LIBDPF_INCLUDE_DPF_UTILS_HPP__ #define LIBDPF_INCLUDE_DPF_UTILS_HPP__ #include #include #include #include #include #include #include #include #include #include #include #include #include "hedley/hedley.h" #include "simde/simde/x86/avx2.h" #include "portable-snippets/exact-int/exact-int.h" #include "portable-snippets/builtin/builtin.h" #include "portable-snippets/endian/endian.h" HEDLEY_PRAGMA(GCC diagnostic push) HEDLEY_PRAGMA(GCC diagnostic ignored "-Waggressive-loop-optimizations") #include "hash-library/sha256.cpp" // NOLINT(build/include) HEDLEY_PRAGMA(GCC diagnostic pop) #include "uint256_t/uint256_t.hpp" #define DPF_UNROLL_LOOP_N(N) HEDLEY_PRAGMA(GCC unroll N) #define DPF_UNROLL_LOOP DPF_UNROLL_LOOP_N(16) #define DPF_ALWAYS_VECTORIZE (#pragma GCC ivdep) namespace std { /// @details specializes `std::numeric_limits` for `uint128_t` template<> class numeric_limits<::uint128_t> { public: static constexpr bool is_specialized = true; static constexpr bool is_signed = false; static constexpr bool is_integer = true; static constexpr bool is_exact = true; static constexpr bool has_infinity = false; static constexpr bool has_quiet_NaN = false; static constexpr bool has_signaling_NaN = false; static constexpr std::float_denorm_style has_denorm = std::denorm_absent; static constexpr bool has_denorm_loss = false; static constexpr std::float_round_style round_style = std::round_toward_zero; static constexpr bool is_iec559 = false; static constexpr bool is_bounded = true; static constexpr bool is_modulo = true; static constexpr int digits = 128; static constexpr int digits10 = 38; static constexpr int max_digits10 = 0; static constexpr int radix = 2; static constexpr int min_exponent = 0; static constexpr int max_exponent = 0; static constexpr int min_exponent10 = 0; static constexpr int max_exponent10 = 0; static constexpr bool traps = false; static constexpr bool tinyness_before = false; HEDLEY_NO_THROW static constexpr uint128_t min() noexcept { return uint128_t{0ul, 0ul}; } HEDLEY_NO_THROW static constexpr uint128_t lowest() noexcept { return uint128_t{0ul, 0ul}; } HEDLEY_NO_THROW static constexpr uint128_t max() noexcept { return uint128_t{-1ul, -1ul}; } HEDLEY_NO_THROW static constexpr uint128_t epsilon() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint128_t round_error() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint128_t infinity() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint128_t quiet_NaN() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint128_t signaling_NaN() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint128_t denorm_min() noexcept { return 0; } }; /// @details specializes `std::numeric_limits` for `uint128_t const` template<> class numeric_limits : public numeric_limits {}; /// @details specializes `std::numeric_limits` for /// @brief `uint128_t volatile` template<> class numeric_limits : public numeric_limits {}; /// @details specializes `std::numeric_limits` for /// @brief `uint128_t const volatile` template<> class numeric_limits : public numeric_limits {}; /// @details specializes `std::numeric_limits` for `uint256_t` template<> class numeric_limits<::uint256_t> { public: static constexpr bool is_specialized = true; static constexpr bool is_signed = false; static constexpr bool is_integer = true; static constexpr bool is_exact = true; static constexpr bool has_infinity = false; static constexpr bool has_quiet_NaN = false; static constexpr bool has_signaling_NaN = false; static constexpr std::float_denorm_style has_denorm = std::denorm_absent; static constexpr bool has_denorm_loss = false; static constexpr std::float_round_style round_style = std::round_toward_zero; static constexpr bool is_iec559 = false; static constexpr bool is_bounded = true; static constexpr bool is_modulo = true; static constexpr int digits = 256; static constexpr int digits10 = 77; static constexpr int max_digits10 = 0; static constexpr int radix = 2; static constexpr int min_exponent = 0; static constexpr int max_exponent = 0; static constexpr int min_exponent10 = 0; static constexpr int max_exponent10 = 0; static constexpr bool traps = false; static constexpr bool tinyness_before = false; HEDLEY_NO_THROW static constexpr uint256_t min() noexcept { return uint256_t{uint128_t{0ul, 0ul}, uint128_t{0ul, 0ul}}; } HEDLEY_NO_THROW static constexpr uint256_t lowest() noexcept { return uint256_t{uint128_t{0ul, 0ul}, uint128_t{0ul, 0ul}}; } HEDLEY_NO_THROW static constexpr uint256_t max() noexcept { return uint256_t{uint128_t{-1ul, -1ul}, uint128_t{-1ul, -1ul}}; } HEDLEY_NO_THROW static constexpr uint256_t epsilon() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint256_t round_error() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint256_t infinity() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint256_t quiet_NaN() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint256_t signaling_NaN() noexcept { return 0; } HEDLEY_NO_THROW static constexpr uint256_t denorm_min() noexcept { return 0; } }; /// @details specializes `std::numeric_limits` for `uint256_t const` template<> class numeric_limits : public numeric_limits {}; /// @details specializes `std::numeric_limits` for /// @brief `uint256_t volatile` template<> class numeric_limits : public numeric_limits {}; /// @details specializes `std::numeric_limits` for /// @brief `uint256_t const volatile` template<> class numeric_limits : public numeric_limits {}; } namespace dpf { using digest_type = std::array; namespace utils { /// @brief Ugly hack to implement `constexpr`-frien`dly conditional `throw` /// @tparam Exception exception /// @param b the `b` /// @param what the diagnostic message /// @return Ugly hack to implement `constexpr`-frien`dly conditional `throw` /// @throws Exception template HEDLEY_ALWAYS_INLINE static constexpr auto constexpr_maybe_throw(bool b, std::string_view what) -> void { (b ? throw Exception{std::data(what)} : 0); } struct max_align { static constexpr std::size_t value = 64; // alignof(__m512i) }; static constexpr std::size_t max_align_v = max_align::value; struct max_integral_bits { static constexpr std::size_t value = 256; // sizeof(uint256_t) * CHAR_BIT }; static constexpr std::size_t max_integral_bits_v = max_integral_bits::value; template struct is_quotient_integer : std::bool_constant || std::is_same_v, simde_int128> || std::is_same_v, simde_uint128>> {}; template static constexpr bool is_quotient_integer_v = is_quotient_integer::value; /// @brief Integer overflow-proof ceiling of division /// @tparam T value type /// @tparam T value type /// @param numerator the `numerator` /// @param denominator the `denominator` /// @return Integer overflow-proof ceiling of division template , bool> = false> HEDLEY_CONST HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE static constexpr T quotient_ceiling(T numerator, T denominator) noexcept { if (numerator == T{0}) return T{0}; return 1 + static_cast(numerator - 1) / denominator; } /// @brief Integer overflow-proof floor of division /// @tparam T value type /// @tparam T value type /// @param numerator the `numerator` /// @param denominator the `denominator` /// @return Integer overflow-proof floor of division template , bool> = false> HEDLEY_CONST HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE static constexpr T quotient_floor(T numerator, T denominator) noexcept { return numerator / denominator; } template struct is_signed_integral : public std::conjunction, std::is_signed> { }; template static constexpr bool is_signed_integral_v = is_signed_integral::value; /// @brief Whether DPF keygen/eval flip the input MSB (two's-complement domains). /// @details Distinct from `is_signed_integral`: wrappers such as signed `fixedpoint` /// are not `std::is_integral`, and treating them as such would break /// `make_unsigned`. Sequence recipe construction and breadth-first eval /// must use this trait, not `is_signed_integral_v`. /// @tparam T value type template struct uses_signed_msb : std::bool_constant< is_signed_integral_v> || std::is_same_v, simde_int128>> { }; template static constexpr bool uses_signed_msb_v = uses_signed_msb::value; template struct make_unsigned { using type = T; }; template struct make_unsigned>> { using type = std::make_unsigned_t; }; template <> struct make_unsigned{ using type = simde_uint128; }; template <> struct make_unsigned{ using type = simde_uint128; }; template <> struct make_unsigned{ using type = uint128_t; }; template <> struct make_unsigned{ using type = uint256_t; }; template using make_unsigned_t = typename make_unsigned::type; /// @brief Make an `std::bitset` from a variadic list of `bool`s /// @tparam Bools bools /// @param bs the `bs` /// @return Make an `std::bitset` from a variadic list of `bool`s template auto make_bitset(Bools ...bs) { std::bitset ret; std::size_t i = 0; (ret.set(i++, bs), ...); return ret; } template HEDLEY_NO_THROW static NodeT single_bit_mask(std::size_t i) noexcept; template <> HEDLEY_ALWAYS_INLINE HEDLEY_PURE HEDLEY_NO_THROW simde__m128i single_bit_mask(std::size_t i) noexcept { return simde_mm_slli_epi64(simde_mm_set_epi64x(i >= 64, i <= 63), i % 64); } template <> HEDLEY_ALWAYS_INLINE HEDLEY_PURE HEDLEY_NO_THROW simde__m256i single_bit_mask(std::size_t i) noexcept { return simde_mm256_slli_epi64(simde_mm256_set_epi64x(i >= 192, i >= 128 && i <= 191, i >= 64 && i <= 127, i <= 63), i % 64); } template ExteriorT to_exterior_node(InteriorT seed); template <> simde__m128i to_exterior_node(simde__m128i seed) { return seed; } template <> simde__m256i to_exterior_node(simde__m128i seed) { return _mm256_zextsi128_si256(seed); } template <> simde__m256i to_exterior_node(simde__m256i seed) { return seed; } template <> simde__m128i to_exterior_node(simde__m256i seed) { return _mm256_castsi256_si128(seed); } template struct bitlength_of : public std::integral_constant ? // we cannot leverage `std::make_unsigned_t` here since `T` might // not be an integral type at all (i.e., this ternary might take // the other path); thus, we resort to a bit of a hack. // If `T==bool`, then `digits==1`; otherwise, if `T` is a signed // integral type, then `digits==bitlength-1`; otherwise if, `T` is // an unsigned integral type, then `digits==bitlength`; otherwise, // we take the other branch. // // We want each of these to return the `bitlength`. So we add // `CHAR_BIT-1` so that `bool` maps to `8`, signed types with // `8l-1` digits map to `8l+6`, and unsigned types with `8l` // digits map to `8l+7`. We then use integer division and // multiplication to round down to the nearest multiple of `8`. ((static_cast(std::numeric_limits::digits)+CHAR_BIT-1)/CHAR_BIT)*CHAR_BIT : CHAR_BIT * sizeof(T)> { }; template static constexpr std::size_t bitlength_of_v = bitlength_of::value; HEDLEY_PRAGMA(GCC diagnostic push) HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") template <> struct bitlength_of : public std::integral_constant { }; template <> struct bitlength_of : public std::integral_constant { }; template <> struct bitlength_of : public std::integral_constant { }; template <> struct bitlength_of : public std::integral_constant { }; HEDLEY_PRAGMA(GCC diagnostic pop) // template <> // struct bitlength_of // : public std::integral_constant { }; template struct bitlength_of> : public std::integral_constant * N> { }; template struct bitlength_of_output : public std::integral_constant(std::ceil(std::log2(sizeof(OutputT) * CHAR_BIT))) : quotient_ceiling(sizeof(OutputT), sizeof(NodeT)) * sizeof(NodeT) * CHAR_BIT> { }; template static constexpr std::size_t bitlength_of_output_v = bitlength_of_output::value; /// @brief the primitive integral type used to represent non integral types /// @tparam Nbits width in bits /// @tparam MinBits min bits /// @tparam MaxBits max bits template struct integral_type_from_bitlength { static_assert(MinBits <= MaxBits); static constexpr auto effective_Nbits = std::min(std::max(Nbits, MinBits), MaxBits); static constexpr auto less_equal = std::less_equal{}; using type = std::conditional_t, psnip_uint32_t>, psnip_uint64_t>, simde_uint128>, uint256_t>, void>; }; template using integral_type_from_bitlength_t = typename integral_type_from_bitlength::type; /// @brief the primitive integral type used to represent non integral types /// @tparam Nbits width in bits /// @tparam MinBits min bits /// @tparam MaxBits max bits template struct nonvoid_integral_type_from_bitlength : public integral_type_from_bitlength { static_assert(Nbits && Nbits <= MaxBits, "representation must fit in 256 bits"); }; template using nonvoid_integral_type_from_bitlength_t = typename nonvoid_integral_type_from_bitlength::type; template struct to_integral_type_base { static constexpr std::size_t bits = bitlength_of_v; // Select integer type larger than or equal to size of std::size_t using integral_type = nonvoid_integral_type_from_bitlength_t>; }; template struct to_integral_type : public to_integral_type_base { using parent = to_integral_type_base; using parent::bits; using typename parent::integral_type; using T_integral_type = integral_type_from_bitlength_t; HEDLEY_PURE HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr integral_type operator()(const T & input) const noexcept { return static_cast(static_cast(input)); } }; template struct make_signed_if { using type = T; }; template struct make_signed_if { using type = std::make_signed_t; }; template struct make_from_integral_value { using T_integral_type = integral_type_from_bitlength_t>; using S_integral_type = std::conditional_t, simde_uint128, T_integral_type>; // `std::conditional_t<..., make_signed_t, U>` instantiates `make_signed` // even when the condition is false, and libstdc++ has no `make_signed` // for `unsigned __int128`. using integral_type = typename make_signed_if && sizeof(S_integral_type) <= 8>::type; HEDLEY_NO_THROW constexpr T operator()(integral_type val) const noexcept { return static_cast(val); } }; /// @brief Reconstruct `x0 XOR x1` via the integral bridge. Prefer this over /// `static_cast(x0 ^ x1)`: for `keyword`, `operator^` yields the parent /// `modint`, which cannot convert back through the private keyword ctor. /// @tparam T value type /// @param x0 the `x0` /// @param x1 the `x1` /// @return Reconstruct `x0 XOR x1` via the integral bridge template HEDLEY_NO_THROW constexpr T xor_input_shares(T x0, T x1) noexcept { constexpr auto to_int = to_integral_type{}; using FromI = typename make_from_integral_value::integral_type; return make_from_integral_value{}( static_cast(to_int(x0) ^ to_int(x1))); } template struct make_default { static constexpr T value = make_from_integral_value{}(1); }; template static constexpr T make_default_v = make_default::value; template static constexpr IntegralT get_node_mask(InputT mask, std::size_t level_index) { using dpf_type = DpfKey; constexpr auto to_int = to_integral_type{}; return static_cast(to_int(mask) >> (level_index-1 + dpf_type::lg_outputs_per_leaf)); } /// @brief Logical right shift. Offsets at or past the width yield 0 (a `>>` of that /// width is undefined for the native unsigned types). /// @tparam IntegralT integral type /// @param value the value to convert or store /// @param offset the public offset /// @return Logical right shift template HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW constexpr IntegralT shift_right(IntegralT value, std::size_t offset) noexcept { if (offset >= bitlength_of_v) return IntegralT{0}; return static_cast(value >> offset); } /// @brief Floor of `from_inclusive / 2^lg_opl`. `lg_opl` is `log2(outputs_per_leaf)`. /// @tparam IntegralT integral type /// @param from_inclusive the `from_inclusive` /// @param lg_opl the `lg_opl` /// @return Floor of `from_inclusive / 2^lg_opl` template HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW constexpr IntegralT leaf_node_floor(IntegralT from_inclusive, std::size_t lg_opl) noexcept { if (lg_opl == 0) return from_inclusive; return shift_right(from_inclusive, lg_opl); } /// @brief Exclusive leaf index of an inclusive input `to_inclusive`. /// @details `2^lg_opl` outputs share a leaf. When `to_inclusive + 1` does not fit in /// `IntegralT`, the exclusive node index is `2^(width - lg_opl)`. That value /// itself does not fit when `lg_opl == 0`; the returned 0 is that saturated /// end (`[from, 2^width)`), which `split_leaf_nodes` interprets. /// @tparam IntegralT integral type /// @param to_inclusive the `to_inclusive` /// @param lg_opl the `lg_opl` /// @return Exclusive leaf index of an inclusive input `to_inclusive` template HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW constexpr IntegralT leaf_node_ceil_exclusive(IntegralT to_inclusive, std::size_t lg_opl) noexcept { constexpr std::size_t width = bitlength_of_v; const IntegralT one{1}; const IntegralT next = static_cast(to_inclusive + one); if (next < to_inclusive) { if (lg_opl == 0 || lg_opl >= width) return IntegralT{0}; return static_cast(one << (width - lg_opl)); } const IntegralT opl = (lg_opl == 0) ? one : static_cast(one << lg_opl); return quotient_ceiling(next, opl); } /// @brief Multi-level flavor: caller passes the slot's `lg(outputs-per-leaf)` (and, /// for interval splitting, its `tree_level`) explicitly. The classic wrappers /// below forward the deepest-slot packing (`DpfKey::lg_outputs_per_leaf`). /// @tparam InputT input domain type /// @tparam size_t size type /// @param from the inclusive start of the range /// @param lg_opl the `lg_opl` /// @return Multi-level flavor: caller passes the slot's `lg(outputs-per-leaf)` (and, for interval /// splitting, its `tree_level`) explicitly template , bitlength_of_v>> static constexpr IntegralT get_from_node_at(InputT from, std::size_t lg_opl) { constexpr auto to_int = to_integral_type{}; return leaf_node_floor(static_cast(to_int(from)), lg_opl); } template , bitlength_of_v>> static constexpr IntegralT get_to_node_at(InputT to, std::size_t lg_opl) { constexpr auto to_int = to_integral_type{}; return leaf_node_ceil_exclusive(static_cast(to_int(to)), lg_opl); } template static constexpr IntegralT get_from_node(InputT from) { return get_from_node_at(from, DpfKey::lg_outputs_per_leaf); } template static constexpr IntegralT get_to_node(InputT to) { return get_to_node_at(to, DpfKey::lg_outputs_per_leaf); } /// @brief One half-open leaf-node range. `to_node == 0` with a nonzero `count` is the /// saturated end `[from_node, 2^width)`. /// @tparam IntegralT integral type template struct node_segment { IntegralT from_node{}; IntegralT to_node{}; std::size_t count = 0; }; template struct node_segments { node_segment seg[2]{}; std::size_t n = 0; std::size_t total = 0; }; /// @brief True when the inclusive walk `[from, to]` wraps the low `bits` of the /// domain. Comparison is on the post-MSB-flip bit pattern. Leaf ids alone /// cannot carry this: packing can put a wrapping pair into `from_node <= to_node`. /// @tparam IntegralT integral type /// @param from the inclusive start of the range /// @param to the `to` /// @param bits the packed bits /// @return True when the inclusive walk `[from, to]` wraps the low `bits` of the domain template inline bool interval_wraps(IntegralT from, IntegralT to, std::size_t bits) { if (bits == 0) return false; if (bits < bitlength_of_v) { const IntegralT mask = static_cast( (IntegralT{1} << bits) - IntegralT{1}); return (from & mask) > (to & mask); } return from > to; } /// @brief Split an inclusive output interval, already reduced to leaf ids, into one /// or two half-open walks. A linearized `from_node > to_node` wraps the node /// id space `[0, 2^depth)`. A saturated `to_node == 0` means the exclusive end /// is `2^{bitwidth(IntegralT)}`, which is the whole id space when `depth` is /// that width. /// /// `input_wraps` is the order of the original inputs, before leaf coarsening. /// @details The buffer is still two runs, `[from_node, 2^depth)` then `[0, to_node)`, /// even when packing makes `from_node <= to_node`. In that case the runs /// overlap on the shared leaf: the iterable's preclip consumes the start of /// the first copy and its length stops inside the second. Collapsing the /// overlap into one forward segment writes the wrong leaves. /// @tparam IntegralT integral type /// @param from_node the `from_node` /// @param to_node the `to_node` /// @param depth the tree depth /// @param input_wraps the `input_wraps` /// @return the returned `node_segments` /// @throws std::length_error if `DPF leaf domain does not fit in size_t` template inline node_segments split_leaf_nodes(IntegralT from_node, IntegralT to_node, std::size_t depth, bool input_wraps = false) { node_segments out; constexpr std::size_t width = bitlength_of_v; constexpr std::size_t size_digits = bitlength_of_v; auto push = [&](IntegralT lo, IntegralT hi) { std::size_t count = 0; if (hi == IntegralT{0} && lo != IntegralT{0}) { // Exclusive end is 2^width. The count fits in size_t only when // that power is one past size_t's maximum and lo is nonzero. if (width > size_digits) throw std::length_error("DPF leaf domain does not fit in size_t"); count = static_cast(0) - static_cast(lo); } else { const auto wide = hi - lo; if (wide > IntegralT(std::numeric_limits::max())) throw std::length_error("DPF leaf domain does not fit in size_t"); count = static_cast(wide); } if (count == 0) return; if (out.total > std::numeric_limits::max() - count) throw std::length_error("DPF leaf domain does not fit in size_t"); out.seg[out.n++] = node_segment{lo, hi, count}; out.total += count; }; if (input_wraps) { if (depth >= width) { if (from_node == IntegralT{0}) throw std::length_error("DPF leaf domain does not fit in size_t"); push(from_node, IntegralT{0}); if (to_node != IntegralT{0}) push(IntegralT{0}, to_node); return out; } const IntegralT domain_end = static_cast(IntegralT{1} << depth); if (from_node < domain_end) push(from_node, domain_end); if (to_node != IntegralT{0}) push(IntegralT{0}, to_node); return out; } if (to_node != IntegralT{0} && from_node < to_node) { push(from_node, to_node); return out; } if (to_node != IntegralT{0} && from_node == to_node) return out; if (from_node == IntegralT{0} && to_node == IntegralT{0}) throw std::length_error("DPF leaf domain does not fit in size_t"); if (depth >= width) { push(from_node, IntegralT{0}); if (to_node != IntegralT{0}) push(IntegralT{0}, to_node); return out; } const IntegralT domain_end = static_cast(IntegralT{1} << depth); if (from_node < domain_end) push(from_node, domain_end); if (to_node != IntegralT{0}) push(IntegralT{0}, to_node); return out; } template static constexpr std::size_t get_leafnodes_in_node_interval(IntegralT from_node, IntegralT to_node) { return static_cast(to_node - from_node); } template inline void flip_msb_if_signed_integral(T & x); template static std::size_t get_leafnodes_in_output_interval(InputT from, InputT to) { // Match eval: the walk order is the bit pattern after the sign flip. InputT flipped_from = from; InputT flipped_to = to; flip_msb_if_signed_integral(flipped_from); flip_msb_if_signed_integral(flipped_to); constexpr auto to_int = to_integral_type{}; const auto from_i = static_cast(to_int(flipped_from)); const auto to_i = static_cast(to_int(flipped_to)); const bool wraps = interval_wraps(from_i, to_i, bitlength_of_v); return split_leaf_nodes(get_from_node(flipped_from), get_to_node(flipped_to), static_cast(DpfKey::depth), wraps).total; } /// @brief Historical name used by the test suite. /// @tparam DpfKey DPF key type /// @tparam InputT input domain type /// @tparam IntegralT integral type /// @param from the inclusive start of the range /// @param to the `to` /// @return Historical name used by the test suite template static std::size_t get_nodes_in_interval(InputT from, InputT to) { return get_leafnodes_in_output_interval(from, to); } template struct mod_pow_2 { HEDLEY_NO_THROW std::size_t operator()(T val, std::size_t n) const noexcept { if (n == 0) { return 0; } // `n >= 64` would shift a uint64 by its width or more. The result is // a `size_t`, so keep the low 64 bits of `val` (the whole residue // when it fits, otherwise the low limb of a wider residue). if (n >= bitlength_of_v) { return static_cast(val & static_cast(~uint64_t{0})); } const auto shift = bitlength_of_v - n; const uint64_t modulo_mask = static_cast(~uint64_t{0}) >> shift; return static_cast(val & modulo_mask); } }; template struct msb_of : public std::integral_constant - 1ul> { }; template struct msb_of, void>> : public msb_of> { }; template static constexpr auto msb_of_v = msb_of::value; template struct countl_zero { HEDLEY_NO_THROW HEDLEY_CONST HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(T val) const noexcept { constexpr auto to_int = to_integral_type{}; constexpr auto bits = bitlength_of_v; using I = typename to_integral_type::integral_type; const I val_ = to_int(val); if constexpr (bitlength_of_v <= 64) { const psnip_uint64_t word = static_cast(val_); if (word == 0) { return bits; } return psnip_builtin_clz64(word) - (64 - bits); } else { constexpr auto i_bits = bitlength_of_v; return countl_zero{}(val_) - (i_bits - bits); } } }; template struct countr_zero { HEDLEY_NO_THROW HEDLEY_CONST HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(T val) const noexcept { constexpr auto to_int = to_integral_type{}; constexpr auto bits = bitlength_of_v; using I = typename to_integral_type::integral_type; const I val_ = to_int(val); if constexpr (bitlength_of_v <= 64) { const uint64_t word = static_cast(val_); if (word == 0) { return bits; } // Zero-extension into a 64-bit word does not change trailing zeros. return psnip_builtin_ctz64(word); } else { if (val_ == I{}) { return bits; } return countr_zero{}(val_); } } }; template struct countl_zero_symmetric_difference { HEDLEY_NO_THROW HEDLEY_CONST HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(T lhs, T rhs) const noexcept { constexpr auto xor_op = std::bit_xor{}; constexpr auto clz = countl_zero{}; return clz(xor_op(lhs, rhs)); } }; HEDLEY_PRAGMA(GCC diagnostic push) HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") template <> struct countl_zero { using T = simde_int128; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { if (!val) return 128; auto limb1 = static_cast(val >> 64); auto limb0 = static_cast(val); return limb1 ? psnip_builtin_clz64(limb1) : 64 + psnip_builtin_clz64(limb0); } }; template struct has_characteristic_two : public std::false_type {}; template static constexpr auto has_characteristic_two_v = has_characteristic_two::value; template <> struct countl_zero { using T = simde_uint128; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { if (!val) return 128; auto limb1 = static_cast(val >> 64); auto limb0 = static_cast(val); return limb1 ? psnip_builtin_clz64(limb1) : 64 + psnip_builtin_clz64(limb0); } }; template <> struct countl_zero { using T = uint128_t; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { if (!val) return 128; auto hi = static_cast(val.upper()); auto lo = static_cast(val.lower()); return hi ? psnip_builtin_clz64(hi) : 64 + psnip_builtin_clz64(lo); } }; template <> struct countl_zero { using T = uint256_t; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { if (!val) return 256; if (val.upper()) return countl_zero{}(val.upper()); return 128 + countl_zero{}(val.lower()); } }; template <> struct countl_zero { using T = simde__m128i; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE std::size_t operator()(const T & val) const noexcept { auto limb1 = static_cast(val[1]); auto limb0 = static_cast(val[0]); if (!limb0 && !limb1) return 128; return limb1 ? psnip_builtin_clz64(limb1) : 64 + psnip_builtin_clz64(limb0); } }; template <> struct countl_zero { using T = simde__m256i; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { std::size_t prefix_len = 0; for (int i = 3; i >= 0; --i, prefix_len += 64) { auto limbi = static_cast(val[i]); if (limbi) { return prefix_len + psnip_builtin_clz64(limbi); } } return prefix_len; } }; template <> struct countr_zero { using T = simde_int128; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { if (!val) return 128; auto limb0 = static_cast(val); if (limb0) return psnip_builtin_ctz64(limb0); return 64 + psnip_builtin_ctz64(static_cast(val >> 64)); } }; template <> struct countr_zero { using T = simde_uint128; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { if (!val) return 128; auto limb0 = static_cast(val); if (limb0) return psnip_builtin_ctz64(limb0); return 64 + psnip_builtin_ctz64(static_cast(val >> 64)); } }; template <> struct countr_zero { using T = uint128_t; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { if (!val) return 128; auto lo = static_cast(val.lower()); if (lo) return psnip_builtin_ctz64(lo); return 64 + psnip_builtin_ctz64(static_cast(val.upper())); } }; template <> struct countr_zero { using T = uint256_t; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { if (!val) return 256; if (val.lower()) return countr_zero{}(val.lower()); return 128 + countr_zero{}(val.upper()); } }; template <> struct countr_zero { using T = simde__m128i; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE std::size_t operator()(const T & val) const noexcept { auto limb0 = static_cast(val[0]); auto limb1 = static_cast(val[1]); if (!limb0 && !limb1) return 128; if (limb0) return psnip_builtin_ctz64(limb0); return 64 + psnip_builtin_ctz64(limb1); } }; template <> struct countr_zero { using T = simde__m256i; HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept { std::size_t suffix_len = 0; for (int i = 0; i < 4; ++i, suffix_len += 64) { auto limbi = static_cast(val[i]); if (limbi) { return suffix_len + psnip_builtin_ctz64(limbi); } } return suffix_len; } }; HEDLEY_PRAGMA(GCC diagnostic pop) // template <> // struct countl_zero // { // using T = simde__m512i; // HEDLEY_CONST // HEDLEY_ALWAYS_INLINE // constexpr std::size_t operator()(const T & val) const noexcept // { // std::size_t prefix_len = 0; // for (int i = 7; i >= 0; --i, prefix_len += 64) // { // auto limbi = static_cast(val[i]); // if (limbi) // { // return prefix_len + psnip_builtin_clz64(limbi); // } // } // return prefix_len; // } // }; template struct is_xor_wrapper : std::false_type {}; template static constexpr bool is_xor_wrapper_v = is_xor_wrapper::value; /// @brief Sub-byte DPF outputs whose lanes are packed inside a leaf node /// (`dpf::bit` is 1, `dpf::twobit` is 2, `dpf::nyble` is 4). The leaf /// image is the buffer image: interval eval memcpy's the node. /// @tparam T value type template struct is_packed_subbyte : std::false_type {}; template static constexpr bool is_packed_subbyte_v = is_packed_subbyte>::value; template struct packed_lane_bits : std::integral_constant {}; template static constexpr std::size_t packed_lane_bits_v = packed_lane_bits>::value; template HEDLEY_PURE HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr auto data(T & bar) noexcept // NOLINT(runtime/references) { return std::data(bar); } /// @brief Pointer overload. Constness of `bar` is the constness of `T`. /// @tparam T value type /// @param bar the `bar` /// @return Pointer overload template HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr T * data(T * bar) noexcept { return bar; } template HEDLEY_ALWAYS_INLINE constexpr auto make_tuple(T && t, Ts && ...ts) noexcept { if constexpr(sizeof...(Ts) == 0) { return std::forward(t); } else { return std::make_tuple(std::forward(t), std::forward(ts)...); } } template struct is_tuple : std::false_type {}; template struct is_tuple> : std::true_type { }; template static constexpr bool is_tuple_v = is_tuple::value; template HEDLEY_PURE HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr auto & get(T & t) noexcept // NOLINT(runtime/references) { if constexpr(I == 0 && is_tuple_v == false) { // if 0th value requested, return it --- even if `t` isn't a tuple return t; } else { // otherwise, just invoke `std::get(t)` and let it succeed or fail // as it may return std::get(t); } } template struct is_bit_array : std::false_type {}; template static constexpr bool is_bit_array_v = is_bit_array::value; template HEDLEY_PURE HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr auto size(const T & t) noexcept { if constexpr(is_bit_array_v == false) { return std::size(t); } else { return t.data_length(); } } template struct flip_msb_for_input { constexpr void operator()(T & x) const { if constexpr (uses_signed_msb_v) { x ^= static_cast(msb_of_v); } } }; template inline void flip_msb_if_signed_integral(T & x) { flip_msb_for_input{}(x); } template auto get_common_part_hash(const std::array & correction_words, const std::array & correction_advice, const LeafTupleT & leaf_tuple, const WildcardMaskT & wildcard_mask) { using zero_type = unsigned char; static constexpr zero_type zero{}; SHA256 h; digest_type digest; h.add(&correction_words, sizeof(correction_words)); h.add(&correction_advice, sizeof(correction_advice)); std::apply([&h, &wildcard_mask](auto const & ...leaf) { std::apply([&h, &leaf...](auto ...is_wildcard) { (h.add(!is_wildcard ? reinterpret_cast(&leaf.get()) : &zero, !is_wildcard ? sizeof(leaf.get()) : sizeof(zero)), ...); }, wildcard_mask); }, leaf_tuple); h.getHash(digest.data()); return digest; } template auto get_common_part_hash(const std::array & correction_words, const std::array & correction_advice, const LeafTupleT & leaf_tuple, const WildcardMaskT & wildcard_mask, const ExtraT & extra) { using zero_type = unsigned char; static constexpr zero_type zero{}; SHA256 h; digest_type digest; h.add(&correction_words, sizeof(correction_words)); h.add(&correction_advice, sizeof(correction_advice)); if constexpr (std::tuple_size_v > 0) h.add(&extra, sizeof(extra)); std::apply([&h, &wildcard_mask](auto const & ...leaf) { std::apply([&h, &leaf...](auto ...is_wildcard) { (h.add(!is_wildcard ? reinterpret_cast(&leaf.get()) : &zero, !is_wildcard ? sizeof(leaf.get()) : sizeof(zero)), ...); }, wildcard_mask); }, leaf_tuple); h.getHash(digest.data()); return digest; } template auto get_common_part_hash(const DpfKey & dpf) { return get_common_part_hash(dpf.correction_words(), dpf.correction_advice(), dpf.leaves(), dpf.wildcard_mask); } template struct has_operators_plus_minus : public std::false_type { }; /// @brief True when `a + b` and `a - b` are valid expressions. /// Overload sets are accepted; taking the address of `operator+` /// is not, because that fails when `+` or `-` is overloaded. /// @tparam OutputT output type template struct has_operators_plus_minus() + std::declval()), decltype(std::declval() - std::declval())>> : public std::true_type { }; template static constexpr bool has_operators_plus_minus_v = has_operators_plus_minus::value; std::size_t parity(psnip_uint8_t x) { return psnip_builtin_parity32(static_cast(x)); } std::size_t parity(psnip_uint16_t x) { return psnip_builtin_parity32(static_cast(x)); } std::size_t parity(psnip_uint32_t x) { return psnip_builtin_parity32(x); } std::size_t parity(psnip_uint64_t x) { return psnip_builtin_parity64(x); } std::size_t parity(simde_uint128 x) { return (psnip_builtin_parity64(static_cast(x)) + psnip_builtin_parity64(static_cast(x >> 64))) & 1; } std::size_t popcount(psnip_uint8_t x) { return psnip_builtin_popcount32(static_cast(x)); } std::size_t popcount(psnip_uint16_t x) { return psnip_builtin_popcount32(static_cast(x)); } std::size_t popcount(psnip_uint32_t x) { return psnip_builtin_popcount32(x); } std::size_t popcount(psnip_uint64_t x) { return psnip_builtin_popcount64(x); } std::size_t popcount(simde_uint128 x) { return psnip_builtin_popcount64(static_cast(x)) + psnip_builtin_popcount64(static_cast(x >> 64)); } std::size_t clz(psnip_uint8_t x) { return x ? psnip_builtin_clz32(static_cast(x)) - 24 : 8; } std::size_t clz(psnip_uint16_t x) { return x ? psnip_builtin_clz32(static_cast(x)) - 16 : 16; } std::size_t clz(psnip_uint32_t x) { return x ? psnip_builtin_clz32(x) : 32; } std::size_t clz(psnip_uint64_t x) { return x ? psnip_builtin_clz64(x) : 64; } std::size_t clz(simde_uint128 x) { if (!x) return 128; return (x > UINT64_MAX) ? psnip_builtin_clz64(x >> 64) : 64 + psnip_builtin_clz64(static_cast(x)); } std::size_t clz(uint128_t x) { return countl_zero{}(x); } std::size_t clz(uint256_t x) { return countl_zero{}(x); } std::size_t ctz(psnip_uint8_t x) { return psnip_builtin_ctz32(x | 0x100); } std::size_t ctz(psnip_uint16_t x) { return psnip_builtin_ctz32(x | 0x10000); } std::size_t ctz(psnip_uint32_t x) { return psnip_builtin_ctz32(x); } std::size_t ctz(psnip_uint64_t x) { return psnip_builtin_ctz64(x); } std::size_t ctz(simde_uint128 x) { return !(x & UINT64_MAX) ? 64 + psnip_builtin_ctz64(x >> 64) : psnip_builtin_ctz64(x); } psnip_uint8_t le(psnip_uint8_t x) { return x; } psnip_uint16_t le(psnip_uint16_t x) { return psnip_endian_le16(x); } psnip_uint32_t le(psnip_uint32_t x) { return psnip_endian_le32(x); } psnip_uint64_t le(psnip_uint64_t x) { return psnip_endian_le64(x); } simde_uint128 le(simde_uint128 x) { #if PSNIP_ENDIAN_ORDER == PSNIP_ENDIAN_LITTLE return x; #else return simde_uint128(psnip_endian_le64(x >> 64)) << 64 | psnip_endian_le64(x); #endif } } // namespace utils } // namespace dpf #endif // LIBDPF_INCLUDE_DPF_UTILS_HPP__