/// @file dpf/eval_interval.hpp /// @brief Evaluate every input in a closed interval. /// @details `[from, to]` is inclusive. The returned iterable yields one /// share per input, in that order. Pass a named output buffer; /// this overload binds it as a non-const reference. An interval /// memoizer is optional and comes after the buffer. /// @snippet evaluation/eval_interval.cpp eval-interval /// @author Ryan Henry /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license; /// see [LICENSE.md](@ref license) for details. #ifndef LIBDPF_INCLUDE_DPF_EVAL_INTERVAL_HPP__ #define LIBDPF_INCLUDE_DPF_EVAL_INTERVAL_HPP__ #include #include #include "hedley/hedley.h" #include #include #include #include #include #include #include #include #include "dpf/dpf_key.hpp" #include "dpf/eval_common.hpp" #include "dpf/eval_target.hpp" #include "dpf/output_buffer.hpp" #include "dpf/interval_memoizer.hpp" #include "dpf/subinterval_iterable.hpp" namespace dpf { namespace internal { template inline auto eval_interval_interior(const DpfKey & dpf, IntegralT from_node, IntegralT to_node, IntervalMemoizer & memoizer, // NOLINT(runtime/references) std::size_t to_level = DpfKey::depth) { using dpf_type = DpfKey; using integral_type = typename DpfKey::integral_type; using node_type = typename DpfKey::interior_node; // level_index represents the current level being built // level_index = 0 => root // level_index = depth => last layer of interior nodes std::size_t level_index = memoizer.assign_interval(dpf, from_node, to_node); std::size_t nodes_at_level = memoizer.get_nodes_at_level(); integral_type mask = utils::get_node_mask(dpf.msb_mask, level_index); for (; level_index <= to_level; level_index = memoizer.advance_level(), nodes_at_level = memoizer.get_nodes_at_level(), mask>>=1) { std::size_t i = 0, j = 0; bool from_offset = mask & from_node, to_offset = from_offset ^ (nodes_at_level & 1); const node_type cw[2] = { dpf.correction_word(level_index-1, 0), dpf.correction_word(level_index-1, 1) }; const bool is_last = dpf_type::tree::is_last_level(level_index - 1, dpf.depth); auto *prev = memoizer[level_index-1]; auto *curr = memoizer[level_index]; // process node which only requires a right traversal if (from_offset == true) { curr[i++] = dpf_type::traverse_interior(prev[j++], cw[1], 1, is_last); } // process all nodes which require both a left traversal and a right traversal const std::size_t both_end = nodes_at_level - to_offset; while (i + 8 <= both_end) { alignas(node_type) node_type parents[4]; alignas(node_type) node_type left[4]; alignas(node_type) node_type right[4]; DPF_UNROLL_LOOP for (std::size_t t = 0; t < 4; ++t) { parents[t] = prev[j + t]; } dpf_type::traverse_interior01_x4(parents, cw[0], cw[1], left, right, is_last); DPF_UNROLL_LOOP for (std::size_t t = 0; t < 4; ++t) { curr[i + 2 * t] = left[t]; curr[i + 2 * t + 1] = right[t]; } i += 8; j += 4; } DPF_UNROLL_LOOP for (; i < both_end;) { auto cur_node = prev[j++]; auto kids = dpf_type::traverse_interior01(cur_node, cw[0], cw[1], is_last); curr[i++] = kids[0]; curr[i++] = kids[1]; } // process node which only requires a left traversal if (to_offset == true) { curr[i] = dpf_type::traverse_interior(prev[j], cw[0], 0, is_last); } } } template inline auto eval_interval_exterior(const DpfKey & dpf, IntegralT from_node, IntegralT to_node, OutputBuffer && outbuf, IntervalMemoizer && memoizer, std::size_t start = 0) { assert_not_wildcard_output(dpf); if (HEDLEY_UNLIKELY(to_node < from_node && to_node != IntegralT{0})) throw std::runtime_error("to_node; std::size_t nodes_in_interval = static_cast(to_node - from_node); HEDLEY_PRAGMA(GCC diagnostic push) HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") auto cw = std::get(dpf.leaf_nodes).get(); HEDLEY_PRAGMA(GCC diagnostic pop) auto *nodes = memoizer[dpf_type::depth]; DPF_UNROLL_LOOP for (std::size_t j = 0, k = start; j < nodes_in_interval; ++j, ++k) { auto leaf = dpf.template traverse_exterior(nodes[j], get_if_lo_bit(cw, nodes[j])); if constexpr (utils::is_packed_subbyte_v) { store_leaf_bytes(outbuf, k, leaf); } else { std::memcpy(&outbuf[k*dpf_type::outputs_per_leaf], &leaf, sizeof(output_type) * dpf_type::outputs_per_leaf); } } } template HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW void store_interval_leaf(OutputBuffer && outbuf, std::size_t k, const LeafT & leaf) noexcept { using dpf_type = DpfKey; using output_type = typename DpfKey::concrete_output_type; if constexpr (utils::is_packed_subbyte_v) { store_leaf_bytes(outbuf, k, leaf); } else { std::memcpy(&outbuf[k * dpf_type::outputs_per_leaf], &leaf, sizeof(output_type) * dpf_type::outputs_per_leaf); } } /// @brief One pass over the leaf-level interior nodes. When the selected output /// indices occupy a contiguous PRG-position range, a single batched /// `ExteriorPRG::eval` produces every output's leaf mask. /// @tparam Is is /// @tparam DpfKey DPF key type /// @tparam OutputBuffers tuple of output buffers /// @tparam IntervalMemoizer interval memoizer type /// @tparam IntegralT integral type /// @tparam IIs iis /// @param dpf the DPF key /// @param from_node the `from_node` /// @param to_node the `to_node` /// @param outbufs the named output buffers /// @param memoizer the memoizer built for this key /// @param start the start of the range /// @throws std::runtime_error if `to_node inline void eval_interval_exterior_fused(const DpfKey & dpf, IntegralT from_node, IntegralT to_node, OutputBuffers && outbufs, IntervalMemoizer && memoizer, std::index_sequence, std::size_t start = 0) { assert_not_wildcard_output(dpf); if (HEDLEY_UNLIKELY(to_node < from_node && to_node != IntegralT{0})) throw std::runtime_error("to_node; HEDLEY_PRAGMA(GCC diagnostic pop) std::size_t nodes_in_interval = static_cast(to_node - from_node); auto *nodes = memoizer[DpfKey::depth]; auto cws = std::make_tuple(std::get(dpf.leaf_nodes).get()...); HEDLEY_PRAGMA(GCC diagnostic push) HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") auto apply_masks = [&](std::size_t k, const node_type & node, const node_type * HEDLEY_RESTRICT masks) { auto apply_output = [&](auto out_index, auto buf_index) { constexpr std::size_t out_i = decltype(out_index)::value; constexpr std::size_t buf_i = decltype(buf_index)::value; using output_type = typename DpfKey::concrete_output_type; using leaf_type = dpf::leaf_node_t; constexpr auto pos = block_offset_of_leaf_v; leaf_type mask; std::memcpy(&mask, masks + (pos - range::pos_min), sizeof(leaf_type)); // Subtractive share: CW_if_t − mask so reconstruct(y0, y1) = y0 − y1 = β. auto leaf = dpf::subtract_leaf( get_if_lo_bit(std::get(cws), node), mask); store_interval_leaf(utils::get(outbufs), k, leaf); }; (apply_output(std::integral_constant{}, std::integral_constant{}), ...); }; std::size_t j = 0, k = start; if constexpr (range::count == 2 && range::pos_min == 0) { for (; j + 4 <= nodes_in_interval; j += 4, k += 4) { alignas(node_type) node_type seeds[4]; alignas(node_type) node_type left[4]; alignas(node_type) node_type right[4]; DPF_UNROLL_LOOP for (std::size_t t = 0; t < 4; ++t) { seeds[t] = utils::to_exterior_node( unset_lo_2bits(nodes[j + t])); } DpfKey::exterior_prg::eval01_x4(seeds, left, right); DPF_UNROLL_LOOP for (std::size_t t = 0; t < 4; ++t) { node_type masks[2] = {left[t], right[t]}; apply_masks(k + t, nodes[j + t], masks); } } } else if constexpr (range::count == 1) { const auto pos = static_cast(range::pos_min); for (; j + 8 <= nodes_in_interval; j += 8, k += 8) { alignas(node_type) node_type seeds[8]; alignas(node_type) node_type masks[8]; DPF_UNROLL_LOOP for (std::size_t t = 0; t < 8; ++t) { seeds[t] = utils::to_exterior_node( unset_lo_2bits(nodes[j + t])); } DpfKey::exterior_prg::eval_x8(seeds, masks, pos); DPF_UNROLL_LOOP for (std::size_t t = 0; t < 8; ++t) { apply_masks(k + t, nodes[j + t], &masks[t]); } } for (; j + 4 <= nodes_in_interval; j += 4, k += 4) { alignas(node_type) node_type seeds[4]; alignas(node_type) node_type masks[4]; DPF_UNROLL_LOOP for (std::size_t t = 0; t < 4; ++t) { seeds[t] = utils::to_exterior_node( unset_lo_2bits(nodes[j + t])); } DpfKey::exterior_prg::eval_x4(seeds, masks, pos); DPF_UNROLL_LOOP for (std::size_t t = 0; t < 4; ++t) { apply_masks(k + t, nodes[j + t], &masks[t]); } } } DPF_UNROLL_LOOP for (; j < nodes_in_interval; ++j, ++k) { const auto & node = nodes[j]; auto seed = utils::to_exterior_node(unset_lo_2bits(node)); std::array masks; DpfKey::exterior_prg::eval(seed, masks.data(), static_cast(range::count), static_cast(range::pos_min)); apply_masks(k, node, masks.data()); } HEDLEY_PRAGMA(GCC diagnostic pop) } template HEDLEY_ALWAYS_INLINE void eval_interval_exterior_all(const DpfKey & dpf, IntegralT from_node, IntegralT to_node, OutputBuffers && outbufs, IntervalMemoizer && memoizer, std::index_sequence idxs, std::size_t start = 0) { using node_type = typename DpfKey::exterior_node; using outputs_tuple = typename DpfKey::concrete_outputs_tuple; HEDLEY_PRAGMA(GCC diagnostic push) HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") using range = leaf_prg_range; HEDLEY_PRAGMA(GCC diagnostic pop) if constexpr (range::is_contiguous) { eval_interval_exterior_fused(dpf, from_node, to_node, outbufs, memoizer, idxs, start); } else { (eval_interval_exterior(dpf, from_node, to_node, utils::get(outbufs), memoizer, start), ...); } } template auto eval_interval_impl(const DpfKey & dpf, InputT from, InputT to, OutputBuffers && outbufs, IntervalMemoizer && memoizer, std::index_sequence) { using dpf_type = DpfKey; using integral_type = typename DpfKey::integral_type; utils::flip_msb_if_signed_integral(from); utils::flip_msb_if_signed_integral(to); integral_type from_node = utils::get_from_node(from), to_node = utils::get_to_node(to); constexpr auto to_int = utils::to_integral_type{}; const bool wraps = utils::interval_wraps( static_cast(to_int(from)), static_cast(to_int(to)), utils::bitlength_of_v); auto segs = utils::split_leaf_nodes(from_node, to_node, dpf.depth, wraps); auto idxs = std::index_sequence{}; std::size_t start = 0; for (std::size_t s = 0; s < segs.n; ++s) { const auto & seg = segs.seg[s]; internal::eval_interval_interior(dpf, seg.from_node, seg.to_node, memoizer); eval_interval_exterior_all(dpf, seg.from_node, seg.to_node, outbufs, memoizer, idxs, start); start += seg.count; } } template auto eval_interval(const DpfKey & dpf, InputT from, InputT to, OutputBuffers && outbufs, IntervalMemoizer && memoizer, std::index_sequence) { using dpf_type = DpfKey; constexpr auto mod_pow_2 = utils::mod_pow_2{}; constexpr auto to_integral_t = utils::to_integral_type{}; constexpr auto bits = utils::bitlength_of_v; eval_interval_impl(dpf, from, to, outbufs, memoizer, std::make_index_sequence()); // `to_integral_type` widens to at least `size_t`. Subtracting in that // wider type loses wrap-around of a narrower input domain (e.g. int16 // intervals that increment across 0). Mask back to the domain width so // `subinterval_iterable` length matches the inclusive [from, to] walk. auto from_i = to_integral_t(from); auto span = to_integral_t(to) - from_i; if constexpr (bits < utils::bitlength_of_v) { span &= (decltype(span){1} << bits) - 1; } auto from_sz = static_cast(from_i); auto to_sz = from_sz + static_cast(span); return utils::make_tuple(subinterval_iterable(std::begin(utils::get(outbufs)), utils::size(utils::get(outbufs)), from_sz, to_sz, mod_pow_2(from, dpf_type::lg_outputs_per_leaf), dpf_type::outputs_per_leaf)...); } } // namespace internal /// @name Closed-interval evaluation /// @tparam I output index /// @tparam Is the remaining output indices /// @tparam DpfKey DPF key type /// @tparam InputT input domain type /// @param dpf the DPF key /// @param from the inclusive start of the range /// @param to the inclusive end of the range /// @{ /// @brief Write outputs `I, Is...` for `[from, to]` into `outbufs`. /// @tparam OutputBuffers tuple of output buffers /// @tparam IntervalMemoizer interval memoizer type /// @param dpf the DPF key /// @param from the inclusive start of the range /// @param to the inclusive end of the range /// @param outbufs named buffer, or a tuple of buffers when several outputs /// are selected. Must outlive the returned iterable. /// @param memoizer workspace sized for at least this interval /// @return an iterable over the written outputs template , std::enable_if_t && !is_multilevel_key_v, bool> = true> HEDLEY_ALWAYS_INLINE auto eval_interval(const DpfKey & dpf, InputT from, InputT to, OutputBuffers & outbufs, IntervalMemoizer && memoizer) // NOLINT(runtime/references) { assert_not_wildcard_output(dpf); return internal::eval_interval(dpf, dpf.offset_x(from), dpf.offset_x(to), outbufs, memoizer, std::make_index_sequence<1+sizeof...(Is)>()); } /// @brief Evaluate `[from, to]` into `outbufs`, allocating a basic interval memoizer. /// @tparam OutputBuffers tuple of output buffers /// @param dpf the DPF key /// @param from the inclusive start of the range /// @param to the inclusive end of the range /// @param outbufs the named output buffers /// @return an iterable over the written outputs template && !is_multilevel_key_v, bool> = true, std::enable_if_t>, std::decay_t>, bool> = true> HEDLEY_ALWAYS_INLINE auto eval_interval(const DpfKey & dpf, InputT from, InputT to, OutputBuffers & outbufs) // NOLINT(runtime/references) { return eval_interval(dpf, from, to, outbufs, dpf::make_basic_interval_memoizer(from, to)); } /// @brief Evaluate `[from, to]` with a caller-supplied memoizer. /// @tparam IntervalMemoizer interval memoizer type /// @param dpf the DPF key /// @param from the inclusive start of the range /// @param to the inclusive end of the range /// @param memoizer the memoizer built for this key /// @return `std::pair` of a new buffer (or tuple of buffers) and an iterable /// into that buffer. template && !is_multilevel_key_v, bool> = true, std::enable_if_t>, std::decay_t>, bool> = true> HEDLEY_ALWAYS_INLINE auto eval_interval(const DpfKey & dpf, InputT from, InputT to, IntervalMemoizer && memoizer) { auto outbufs = utils::make_tuple( make_output_buffer_for_interval(dpf, from, to), make_output_buffer_for_interval(dpf, from, to)...); // moving `outbufs` is allowed as the `outbufs` are `std::vectors` // the underlying data remains on the heap // and thus the data the iterable refers to is still valid auto iterable = eval_interval(dpf, from, to, outbufs, memoizer); return std::make_pair(std::move(outbufs), std::move(iterable)); } /// @brief Evaluate `[from, to]`, allocating a basic interval memoizer and a buffer. /// @return `std::pair` of a new buffer (or tuple of buffers) and an iterable /// into that buffer. template && !is_multilevel_key_v, bool> = true> HEDLEY_ALWAYS_INLINE auto eval_interval(const DpfKey & dpf, InputT from, InputT to) { return eval_interval(dpf, from, to, dpf::make_basic_interval_memoizer(from, to)); } /// @} } // namespace dpf #endif // LIBDPF_INCLUDE_DPF_EVAL_INTERVAL_HPP__