/// @file grotto/dyadic_lut.hpp /// @brief Exact dyadic step functions from the Grotto gadget list. /// @details Integer results and boolean 1s are raw fixed-point values: /// an integer n is stored as `n << fractional_bits`. `ilogb(0)` and /// `ilog10(0)` return `ilog_of_zero`. /// /// `make_msb_lut(i)` is bit `i` counting from the most significant /// bit. It is constant on `2^{i+1}` intervals, so `i` must be less /// than `msb_bit_limit`. /// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license. #ifndef LIBDPF_INCLUDE_GROTTO_DYADIC_LUT_HPP__ #define LIBDPF_INCLUDE_GROTTO_DYADIC_LUT_HPP__ #include "hedley/hedley.h" #include "grotto/easy_lut.hpp" #include #include #include #include #include namespace grotto { /// @brief Sentinel raw value for `ilogb(0)` and `ilog10(0)`. inline constexpr std::int64_t ilog_of_zero = std::numeric_limits::min(); /// @brief `make_msb_lut(i)` allows `i` in `[0, msb_bit_limit)`. /// @details Bit 0 is two intervals; bit 7 is 256. inline constexpr unsigned msb_bit_limit = 8; namespace detail { using u128 = unsigned __int128; template HEDLEY_CONST HEDLEY_NO_THROW constexpr int raw_width() noexcept { return std::numeric_limits::digits + 1; } inline std::int64_t encode_units(std::int64_t units, unsigned fractional_bits) { if (fractional_bits >= 63) throw std::invalid_argument("dyadic lut: fractional width does not fit"); const __int128 scaled = static_cast<__int128>(units) << fractional_bits; if (scaled > std::numeric_limits::max() || scaled < std::numeric_limits::min()) throw std::overflow_error("dyadic lut: encoded value does not fit int64"); return static_cast(scaled); } inline easy_poly unit_poly(std::int64_t units, unsigned fractional_bits) { return easy_poly{encode_units(units, fractional_bits), 0, 0, 1}; } inline easy_poly indicator_poly(bool on, unsigned fractional_bits) { return unit_poly(on ? 1 : 0, fractional_bits); } template HEDLEY_CONST HEDLEY_NO_THROW constexpr std::uint64_t raw_bits(std::int64_t raw) noexcept { constexpr int width = raw_width(); const auto masked = static_cast(raw); if constexpr (width >= 64) return masked; else return masked & ((std::uint64_t{1} << width) - 1); } inline int countl_zero_width(std::uint64_t bits, int width) { if (width <= 0 || width > 64) throw std::invalid_argument("dyadic lut: width must be 1..64"); if (width < 64) bits &= (std::uint64_t{1} << width) - 1; if (bits == 0) return width; return __builtin_clzll(bits) - (64 - width); } inline int countl_one_width(std::uint64_t bits, int width) { const std::uint64_t flipped = width >= 64 ? ~bits : (~bits & ((std::uint64_t{1} << width) - 1)); return countl_zero_width(flipped, width); } template int clz_of(std::int64_t raw) { return countl_zero_width(raw_bits(raw), raw_width()); } template int clrsb_of(std::int64_t raw) { constexpr int width = raw_width(); const std::uint64_t bits = raw_bits(raw); const bool neg = ((bits >> (width - 1)) & 1u) != 0; const int matched = neg ? countl_one_width(bits, width) : countl_zero_width(bits, width); return matched - 1; } HEDLEY_CONST HEDLEY_NO_THROW constexpr int floor_log2_u128(u128 mag) noexcept { if (mag == 0) return -1; int n = 0; while (mag > 1) { mag >>= 1; ++n; } return n; } template HEDLEY_CONST HEDLEY_NO_THROW constexpr u128 magnitude(std::int64_t raw) noexcept { using lim = std::numeric_limits; if (raw == static_cast(lim::min())) return u128{1} << (raw_width() - 1); const auto abs = raw < 0 ? -raw : raw; return static_cast(abs); } template HEDLEY_CONST HEDLEY_NO_THROW constexpr int ilogb_units(std::int64_t raw, unsigned fractional_bits) noexcept { if (raw == 0) return 0; return floor_log2_u128(magnitude(raw)) - static_cast(fractional_bits); } inline u128 pow10_u128(int exponent) { if (exponent < 0 || exponent > 38) throw std::invalid_argument("dyadic lut: power of ten is out of range"); u128 p = 1; for (int i = 0; i < exponent; ++i) p *= 10; return p; } /// @brief `mag / 2^k >= 10^e`. /// @param mag the magnitude /// @param fractional_bits the number of fractional bits /// @param exponent the exponent /// @return `mag / 2^k >= 10^e` inline bool magnitude_ge_pow10(u128 mag, unsigned fractional_bits, int exponent) { if (mag == 0) return false; if (exponent >= 0) { const u128 decade = pow10_u128(exponent); if (fractional_bits > 0 && decade > (~u128{0} >> fractional_bits)) return false; return mag >= (decade << fractional_bits); } const u128 decade = pow10_u128(-exponent); const u128 scale = u128{1} << fractional_bits; u128 threshold = scale / decade; if (scale % decade != 0) ++threshold; return mag >= threshold; } template int ilog10_units(std::int64_t raw, unsigned fractional_bits) { if (raw == 0) return 0; const u128 mag = magnitude(raw); int lo = -static_cast(fractional_bits) - 2; int hi = raw_width(); while (lo < hi) { const int mid = lo + (hi - lo + 1) / 2; if (magnitude_ge_pow10(mag, fractional_bits, mid)) lo = mid; else hi = mid - 1; } return lo; } template HEDLEY_WARN_UNUSED_RESULT easy_lut steps_from_cuts(std::vector cuts, At && at) { return assemble_easy(std::move(cuts), [&](std::int64_t raw) { return at(raw); }); } template HEDLEY_WARN_UNUSED_RESULT easy_lut predicate_lut(unsigned fractional_bits, bool at_or_below_zero, bool at_zero, bool above_zero) { std::vector cuts{0, 1}; return steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { const bool on = raw < 0 ? at_or_below_zero : (raw == 0 ? at_zero : above_zero); return indicator_poly(on, fractional_bits); }); } } // namespace detail template HEDLEY_WARN_UNUSED_RESULT easy_lut make_positive_lut(unsigned fractional_bits = 0) { return detail::predicate_lut(fractional_bits, false, false, true); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_negative_lut(unsigned fractional_bits = 0) { return detail::predicate_lut(fractional_bits, true, false, false); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_nonnegative_lut(unsigned fractional_bits = 0) { return detail::predicate_lut(fractional_bits, false, true, true); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_nonpositive_lut(unsigned fractional_bits = 0) { return detail::predicate_lut(fractional_bits, true, true, false); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_zero_lut(unsigned fractional_bits = 0) { return detail::predicate_lut(fractional_bits, false, true, false); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_nonzero_lut(unsigned fractional_bits = 0) { return detail::predicate_lut(fractional_bits, true, false, true); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_signum_lut(unsigned fractional_bits = 0) { const std::int64_t one = detail::encode_units(1, fractional_bits); return detail::steps_from_cuts({0, 1}, [=](std::int64_t raw) { if (raw < 0) return detail::easy_poly{-one, 0, 0, 1}; if (raw == 0) return detail::kZero; return detail::easy_poly{one, 0, 0, 1}; }); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_clz_lut(unsigned fractional_bits = 0) { constexpr int width = detail::raw_width(); std::vector cuts{0, 1}; for (int b = 1; b <= width - 2; ++b) cuts.push_back(std::int64_t{1} << b); return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { return detail::unit_poly(detail::clz_of(raw), fractional_bits); }); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_clrsb_lut(unsigned fractional_bits = 0) { constexpr int width = detail::raw_width(); std::vector cuts{0, -1, -2}; for (int exp = 0; exp <= width - 2; ++exp) cuts.push_back(std::int64_t{1} << exp); for (int exp = 2; exp <= width - 1; ++exp) { if (exp >= 63) cuts.push_back(static_cast(std::numeric_limits::min())); else cuts.push_back(-(std::int64_t{1} << exp)); } return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { return detail::unit_poly(detail::clrsb_of(raw), fractional_bits); }); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_ilogb_lut(unsigned fractional_bits = 0) { constexpr int width = detail::raw_width(); std::vector cuts{0, 1}; for (int b = 1; b <= width - 2; ++b) cuts.push_back(std::int64_t{1} << b); for (int exp = 1; exp <= width - 1; ++exp) { if (exp >= width - 1) cuts.push_back(static_cast(std::numeric_limits::min()) + 1); else cuts.push_back(-(std::int64_t{1} << exp) + 1); } return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { if (raw == 0) return detail::easy_poly{ilog_of_zero, 0, 0, 1}; return detail::unit_poly(detail::ilogb_units(raw, fractional_bits), fractional_bits); }); } template HEDLEY_WARN_UNUSED_RESULT easy_lut make_ilog10_lut(unsigned fractional_bits = 0) { using lim = std::numeric_limits; const std::int64_t maxv = static_cast(lim::max()); const int lo = detail::ilog10_units(1, fractional_bits); const int hi_pos = detail::ilog10_units(maxv, fractional_bits); const int hi_neg = detail::ilog10_units(static_cast(lim::min()), fractional_bits); const int hi = hi_pos > hi_neg ? hi_pos : hi_neg; std::vector cuts{0, 1}; std::int64_t prev = 0; for (int e = lo; e <= hi; ++e) { std::int64_t left = 1; std::int64_t right = maxv; while (left < right) { const std::int64_t mid = left + (right - left) / 2; if (detail::ilog10_units(mid, fractional_bits) >= e) right = mid; else left = mid + 1; } if (left == prev) continue; prev = left; cuts.push_back(left); if (left < maxv) { std::int64_t end = left; std::int64_t scan_left = left; std::int64_t scan_right = maxv; while (scan_left < scan_right) { const std::int64_t mid = scan_left + (scan_right - scan_left + 1) / 2; if (detail::ilog10_units(mid, fractional_bits) == e) scan_left = mid; else scan_right = mid - 1; } end = scan_left; const std::int64_t neg = -end; if (neg > static_cast(lim::min())) cuts.push_back(neg); } } return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { if (raw == 0) return detail::easy_poly{ilog_of_zero, 0, 0, 1}; return detail::unit_poly(detail::ilog10_units(raw, fractional_bits), fractional_bits); }); } /// @brief Bit `index` counting down from the most significant bit of `Raw`. /// @details Index 0 is the sign bit. Larger indexes are refused: the bit is constant /// on `2^{index+1}` intervals. /// @tparam Raw underlying representation /// @param index the index /// @param fractional_bits the number of fractional bits /// @return Bit `index` counting down from the most significant bit of `Raw` /// @throws std::invalid_argument if `only the most significant bits are piecewise-cheap` template HEDLEY_WARN_UNUSED_RESULT easy_lut make_msb_lut(unsigned index, unsigned fractional_bits = 0) { constexpr int width = detail::raw_width(); if (index >= msb_bit_limit || static_cast(index) >= width) throw std::invalid_argument( "msb lut: only the most significant bits are piecewise-cheap"); const int shift = width - 1 - static_cast(index); if (shift >= 63) { return detail::steps_from_cuts({0}, [=](std::int64_t raw) { const bool on = ((detail::raw_bits(raw) >> shift) & 1u) != 0; return detail::indicator_poly(on, fractional_bits); }); } const std::int64_t step = std::int64_t{1} << shift; std::vector cuts; const auto minv = static_cast(std::numeric_limits::min()); for (std::int64_t boundary = minv; ; ) { cuts.push_back(boundary); if (boundary > static_cast(std::numeric_limits::max()) - step) break; boundary += step; } return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { const bool on = ((detail::raw_bits(raw) >> shift) & 1u) != 0; return detail::indicator_poly(on, fractional_bits); }); } } // namespace grotto #endif // LIBDPF_INCLUDE_GROTTO_DYADIC_LUT_HPP__