libdpf/include/dpf/eval_unified.hpp
Ryan Henry 0d22946a0e Checkpoint the party/runtime stack before share-program and malicious-mode work.
Ship the TLS mesh, composer, Beaver/Yao/leaf MPC, prep/online paths, apps, and docs so the tree is pushable before elevating share_expr, security_mode, and prep resume.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-28 05:59:19 -06:00

964 lines
46 KiB
C++
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

/// @file dpf/eval_unified.hpp
/// @brief Target-first eval surface for DPF / iDPF / DCF channels.
/// @details `eval_*(out<I>, …)` selects point-output slot `I`.
/// `eval_*(cmp, …)` selects the comparison channel. Memoizer and
/// buffer arguments match the classic overloads: a path memoizer
/// on `eval_point`, an output buffer then an interval memoizer on
/// `eval_interval`. `make_output_buffer(out<I>, key, from, to)` and
/// `make_output_buffer(cmp, key, n)` size the buffer for that channel.
/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors)
/// @license Released under a GNU General Public v2.0 (GPLv2) license.
#ifndef LIBDPF_INCLUDE_DPF_EVAL_UNIFIED_HPP__
#define LIBDPF_INCLUDE_DPF_EVAL_UNIFIED_HPP__
#include "hedley/hedley.h"
#include <cstddef>
#include <cstring>
#include <algorithm>
#include <iterator>
#include <limits>
#include <list>
#include <stdexcept>
#include <type_traits>
#include <utility>
#include <portable-snippets/exact-int/exact-int.h>
#include "dpf/eval_target.hpp"
#include "dpf/eval_point.hpp"
#include "dpf/eval_interval.hpp"
#include "dpf/eval_full.hpp"
#include "dpf/eval_sequence.hpp"
#include "dpf/sequence_recipe.hpp"
#include "dpf/incremental.hpp"
#include "dpf/path_memoizer.hpp"
#include "dpf/interval_memoizer.hpp"
#include "dpf/aligned_allocator.hpp"
#include "dpf/leaf_node.hpp"
#include "dpf/verifiable.hpp"
namespace dpf
{
namespace detail
{
template <std::size_t I, std::size_t N, typename KeyT>
HEDLEY_NO_THROW
constexpr std::size_t resolved_out_prefix() noexcept
{
if constexpr (is_multilevel_key_v<KeyT>)
{
if constexpr (N != prefix_deduce)
{
static_assert(KeyT::meta[I].prefix == N,
"out<I,N>: N does not match key::meta[I].prefix");
return N;
}
else
return KeyT::meta[I].prefix;
}
else
{
(void)N;
return utils::bitlength_of_v<typename KeyT::input_type>;
}
}
} // namespace detail
// ---------------------------------------------------------------------------
// eval_point(target, key, x [, path])
// ---------------------------------------------------------------------------
/// \complexity O(n) time. n is `depth`. One interior traversal per level from the memoizer resume index through the leaf. Extra space is the path memoizer (O(n) nodes, or one node if it does not memoize). A proof token adds one fold per level walked.
template <std::size_t I, std::size_t N, typename KeyT, typename QueryT,
typename PathMemoizer = nonmemoizing_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_point(out_t<I, N>, const KeyT & key, QueryT && x,
PathMemoizer && path = PathMemoizer{})
{
if constexpr (is_multilevel_key_v<KeyT>)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_point_impl<pref, I>(key,
std::forward<QueryT>(x), std::forward<PathMemoizer>(path));
}
else
{
return eval_point<I>(key, std::forward<QueryT>(x),
std::forward<PathMemoizer>(path));
}
}
/// \complexity O(n) time. n is `depth`. One interior traversal per level from the memoizer resume index through the leaf. Extra space is the path memoizer (O(n) nodes, or one node if it does not memoize). A proof token adds one fold per level walked.
template <typename Beta = uint64_t, typename KeyT, typename QueryT,
typename PathMemoizer = basic_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_point(cmp_t, const KeyT & key, QueryT && x,
PathMemoizer && path = PathMemoizer{})
{
return detail::incr::eval_cmp_point_impl<Beta>(key, std::forward<QueryT>(x),
std::forward<PathMemoizer>(path));
}
/// \complexity O(n) time. n is `depth`. One interior traversal per level from the memoizer resume index through the leaf. Extra space is the path memoizer (O(n) nodes, or one node if it does not memoize). A proof token adds one fold per level walked.
template <typename Beta = uint64_t, typename KeyT, typename QueryT,
typename PathMemoizer = basic_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_point(cmp_t, const KeyT & key, QueryT && x, prove_ref pr,
PathMemoizer && path = PathMemoizer{})
{
static_assert(KeyT::is_verifiable,
"eval_point(cmp, ..., prove(π)): key must carry dpf::verifiable");
detail::vdpf::init_proof(pr.token, key);
auto out = detail::incr::eval_cmp_point_impl<Beta>(key, std::forward<QueryT>(x),
std::forward<PathMemoizer>(path), &pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
return out;
}
/// \complexity O(n) time. n is `depth`. One interior traversal per level from the memoizer resume index through the leaf. Extra space is the path memoizer (O(n) nodes, or one node if it does not memoize). A proof token adds one fold per level walked.
template <typename Beta = uint64_t, typename KeyT, typename QueryT,
typename PathMemoizer = basic_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_point(cmp_t, const KeyT & key, QueryT && x, sketch_ref & sk,
PathMemoizer && path = PathMemoizer{})
{
static_assert(KeyT::is_extractable,
"eval_point(cmp, ..., sketch(σ)): key must carry dpf::extractable");
auto y = detail::incr::eval_cmp_point_impl<Beta>(key, std::forward<QueryT>(x),
std::forward<PathMemoizer>(path));
sk.absorb(y);
return y;
}
/// \complexity O(n) time. n is `depth`. One interior traversal per level from the memoizer resume index through the leaf. Extra space is the path memoizer (O(n) nodes, or one node if it does not memoize). A proof token adds one fold per level walked.
template <std::size_t L, typename Beta = uint64_t, typename KeyT, typename QueryT,
typename PathMemoizer = basic_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_point(cmp_prefix_t<L>, const KeyT & key, QueryT && x,
PathMemoizer && path = PathMemoizer{})
{
return detail::incr::eval_cmp_prefix_point_impl<L, Beta>(key,
std::forward<QueryT>(x), std::forward<PathMemoizer>(path));
}
/// \complexity O(n) time. n is `depth`. One interior traversal per level from the memoizer resume index through the leaf. Extra space is the path memoizer (O(n) nodes, or one node if it does not memoize). A proof token adds one fold per level walked.
template <std::size_t L, typename Beta = uint64_t, typename KeyT, typename QueryT,
typename PathMemoizer = basic_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_point(cmp_prefix_t<L>, const KeyT & key, QueryT && x, prove_ref pr,
PathMemoizer && path = PathMemoizer{})
{
static_assert(KeyT::is_verifiable,
"eval_point(cmp_prefix, ..., prove(π)): key must carry dpf::verifiable");
detail::vdpf::init_proof(pr.token, key);
auto out = detail::incr::eval_cmp_prefix_point_impl<L, Beta>(key,
std::forward<QueryT>(x), std::forward<PathMemoizer>(path), &pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
return out;
}
// ---------------------------------------------------------------------------
// eval_interval(target, key, from, to [, buf [, memo]])
// ---------------------------------------------------------------------------
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <std::size_t I, std::size_t N, typename KeyT, typename LaneT,
typename OutputBuffer, typename IntervalMemoizer,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_interval(out_t<I, N>, const KeyT & key, LaneT from, LaneT to,
OutputBuffer && outbuf, IntervalMemoizer && memo)
{
if constexpr (is_multilevel_key_v<KeyT>)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_interval_impl<pref, I>(key, from, to,
std::forward<OutputBuffer>(outbuf),
std::forward<IntervalMemoizer>(memo));
}
else
{
return eval_interval<I>(key, from, to,
std::forward<OutputBuffer>(outbuf),
std::forward<IntervalMemoizer>(memo));
}
}
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <std::size_t I, std::size_t N, typename KeyT, typename LaneT,
typename OutputBuffer,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_interval(out_t<I, N>, const KeyT & key, LaneT from, LaneT to,
OutputBuffer && outbuf)
{
if constexpr (is_multilevel_key_v<KeyT>)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_interval_impl<pref, I>(key, from, to,
std::forward<OutputBuffer>(outbuf));
}
else
{
return eval_interval<I>(key, from, to,
std::forward<OutputBuffer>(outbuf));
}
}
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <std::size_t I, std::size_t N, typename KeyT, typename LaneT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_interval(out_t<I, N>, const KeyT & key, LaneT from, LaneT to)
{
if constexpr (is_multilevel_key_v<KeyT>)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_interval_impl<pref, I>(key, from, to);
}
else
{
return eval_interval<I>(key, from, to);
}
}
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
typename OutputBuffer,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void eval_interval(cmp_t, const KeyT & key, LaneT from, LaneT to,
OutputBuffer && outbuf)
{
detail::incr::eval_cmp_interval_impl<Beta>(key, from, to,
std::forward<OutputBuffer>(outbuf));
}
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
typename OutputBuffer,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void eval_interval(cmp_t, const KeyT & key, LaneT from, LaneT to,
OutputBuffer && outbuf, prove_ref pr)
{
static_assert(KeyT::is_verifiable,
"eval_interval(cmp, ..., prove(π)): key must carry dpf::verifiable");
detail::vdpf::init_proof(pr.token, key);
detail::incr::eval_cmp_interval_impl<Beta>(key, from, to,
std::forward<OutputBuffer>(outbuf), &pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
}
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
typename OutputBuffer, typename IntervalMemoizer,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void eval_interval(cmp_t, const KeyT & key, LaneT from, LaneT to,
OutputBuffer && outbuf, IntervalMemoizer && memo)
{
detail::incr::eval_cmp_interval_impl<Beta>(key, from, to,
std::forward<OutputBuffer>(outbuf),
std::forward<IntervalMemoizer>(memo));
}
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
typename OutputBuffer, typename IntervalMemoizer,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void eval_interval(cmp_t, const KeyT & key, LaneT from, LaneT to,
OutputBuffer && outbuf, IntervalMemoizer && memo, prove_ref pr)
{
static_assert(KeyT::is_verifiable,
"eval_interval(cmp, ..., prove(π)): key must carry dpf::verifiable");
detail::vdpf::init_proof(pr.token, key);
detail::incr::eval_cmp_interval_impl<Beta>(key, from, to,
std::forward<OutputBuffer>(outbuf),
std::forward<IntervalMemoizer>(memo), &pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
}
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_interval(cmp_t, const KeyT & key, LaneT from, LaneT to)
{
return detail::incr::eval_cmp_interval_impl<Beta>(key, from, to);
}
/// \complexity O(L) interior traversals and O(L) workspace in the basic memoizer. L is the number of leaf nodes covering the closed interval (`get_nodes_at_level` at `depth`). Level k expands `(to >> (n-k)) - (from >> (n-k)) + 1` nodes; those counts sum to Θ(L). The output buffer holds one slot per input in the interval. n is `depth`.
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_interval(cmp_t, const KeyT & key, LaneT from, LaneT to, prove_ref pr)
{
static_assert(KeyT::is_verifiable,
"eval_interval(cmp, ..., prove(π)): key must carry dpf::verifiable");
detail::vdpf::init_proof(pr.token, key);
auto out = detail::incr::eval_cmp_interval_impl<Beta>(key, from, to, &pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
return out;
}
// ---------------------------------------------------------------------------
// eval_full(target, key [, …])
// ---------------------------------------------------------------------------
/// \complexity Same expansion as `eval_interval` on the whole domain. L = 2^{n - lg(outputs_per_leaf)} leaf nodes, n = `depth`. Time Θ(L) interior traversals. The output buffer stores one slot per domain point (2^n).
template <std::size_t I, std::size_t N, typename KeyT,
typename OutputBuffer, typename IntervalMemoizer,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_full(out_t<I, N>, const KeyT & key, OutputBuffer && outbuf,
IntervalMemoizer && memo)
{
if constexpr (is_multilevel_key_v<KeyT>)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_full_impl<pref, I>(key,
std::forward<OutputBuffer>(outbuf),
std::forward<IntervalMemoizer>(memo));
}
else
{
return eval_full<I>(key, std::forward<OutputBuffer>(outbuf),
std::forward<IntervalMemoizer>(memo));
}
}
/// \complexity Same expansion as `eval_interval` on the whole domain. L = 2^{n - lg(outputs_per_leaf)} leaf nodes, n = `depth`. Time Θ(L) interior traversals. The output buffer stores one slot per domain point (2^n).
template <std::size_t I, std::size_t N, typename KeyT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_full(out_t<I, N>, const KeyT & key)
{
if constexpr (is_multilevel_key_v<KeyT>)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_full_impl<pref, I>(key);
}
else
{
return eval_full<I>(key);
}
}
/// \complexity Same expansion as `eval_interval` on the whole domain. L = 2^{n - lg(outputs_per_leaf)} leaf nodes, n = `depth`. Time Θ(L) interior traversals. The output buffer stores one slot per domain point (2^n).
template <typename Beta = uint64_t, typename KeyT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_full(cmp_t, const KeyT & key)
{
if (!key.has_cmp())
throw std::invalid_argument("eval_full(cmp): no comparison channel");
using lane_t = typename KeyT::integral_type;
const auto nbits = static_cast<std::size_t>(key.cmp().nbits);
const lane_t lo = 0;
const lane_t hi = (nbits >= 8 * sizeof(lane_t))
? static_cast<lane_t>(~lane_t{0})
: static_cast<lane_t>((lane_t{1} << nbits) - 1);
return detail::incr::eval_cmp_interval_impl<Beta>(key, lo, hi);
}
/// \complexity Same expansion as `eval_interval` on the whole domain. L = 2^{n - lg(outputs_per_leaf)} leaf nodes, n = `depth`. Time Θ(L) interior traversals. The output buffer stores one slot per domain point (2^n).
template <typename Beta = uint64_t, typename KeyT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_full(cmp_t, const KeyT & key, prove_ref pr)
{
static_assert(KeyT::is_verifiable,
"eval_full(cmp, ..., prove(π)): key must carry dpf::verifiable");
if (!key.has_cmp())
throw std::invalid_argument("eval_full(cmp): no comparison channel");
using lane_t = typename KeyT::integral_type;
const auto nbits = static_cast<std::size_t>(key.cmp().nbits);
const lane_t lo = 0;
const lane_t hi = (nbits >= 8 * sizeof(lane_t))
? static_cast<lane_t>(~lane_t{0})
: static_cast<lane_t>((lane_t{1} << nbits) - 1);
detail::vdpf::init_proof(pr.token, key);
auto out = detail::incr::eval_cmp_interval_impl<Beta>(key, lo, hi, &pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
return out;
}
// ---------------------------------------------------------------------------
// eval_sequence(target, key, begin, end, buf [, path])
// ---------------------------------------------------------------------------
/// \complexity O(n k) interior traversals in the worst case and O(k) node workspace. k is the number of listed points and n is `depth`. The breadth-first buffer is 2k nodes, so each level traverses at most one node per point. Shared prefixes do fewer traversals. A recipe memoizer instead stores O(recipe leaf nodes) (see that memoizer).
template <std::size_t I, std::size_t N, typename KeyT, typename ForwardIterator,
typename OutputBuffer,
typename PathMemoizer = basic_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_sequence(out_t<I, N>, const KeyT & key, ForwardIterator begin,
ForwardIterator end, OutputBuffer && outbuf,
PathMemoizer && path = PathMemoizer{})
{
if constexpr (is_multilevel_key_v<KeyT>)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_sequence_impl<pref, I>(key, begin, end,
std::forward<OutputBuffer>(outbuf),
std::forward<PathMemoizer>(path));
}
else
{
return eval_sequence<I>(key, begin, end,
std::forward<OutputBuffer>(outbuf));
}
}
/// \complexity O(n k) interior traversals in the worst case and O(k) node workspace. k is the number of listed points and n is `depth`. The breadth-first buffer is 2k nodes, so each level traverses at most one node per point. Shared prefixes do fewer traversals. A recipe memoizer instead stores O(recipe leaf nodes) (see that memoizer).
template <typename Beta = uint64_t, typename KeyT, typename ForwardIterator,
typename OutputBuffer,
typename PathMemoizer = basic_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void eval_sequence(cmp_t, const KeyT & key, ForwardIterator begin,
ForwardIterator end, OutputBuffer && outbuf,
PathMemoizer && path = PathMemoizer{})
{
detail::incr::eval_cmp_sequence_impl<Beta>(key, begin, end,
std::forward<OutputBuffer>(outbuf),
std::forward<PathMemoizer>(path));
}
/// \complexity O(n k) interior traversals in the worst case and O(k) node workspace. k is the number of listed points and n is `depth`. The breadth-first buffer is 2k nodes, so each level traverses at most one node per point. Shared prefixes do fewer traversals. A recipe memoizer instead stores O(recipe leaf nodes) (see that memoizer).
template <typename Beta = uint64_t, typename KeyT, typename ForwardIterator,
typename OutputBuffer,
typename PathMemoizer = basic_path_memoizer<KeyT>,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void eval_sequence(cmp_t, const KeyT & key, ForwardIterator begin,
ForwardIterator end, OutputBuffer && outbuf, prove_ref pr,
PathMemoizer && path = PathMemoizer{})
{
static_assert(KeyT::is_verifiable,
"eval_sequence(cmp, ..., prove(π)): key must carry dpf::verifiable");
detail::vdpf::init_proof(pr.token, key);
detail::incr::eval_cmp_sequence_impl<Beta>(key, begin, end,
std::forward<OutputBuffer>(outbuf),
std::forward<PathMemoizer>(path), &pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
}
// ---------------------------------------------------------------------------
// make_output_buffer(target, …)
// ---------------------------------------------------------------------------
template <typename Beta = uint64_t, typename KeyT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto make_output_buffer(cmp_t, const KeyT & key, std::size_t n)
{
return detail::incr::make_output_buffer_for_cmp_impl<Beta>(key, n);
}
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto make_output_buffer(cmp_t, const KeyT & key, LaneT from, LaneT to)
{
return detail::incr::make_output_buffer_for_cmp_interval_impl<Beta>(
key, from, to);
}
template <std::size_t I, std::size_t N, typename KeyT, typename LaneT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto make_output_buffer(out_t<I, N>, const KeyT & key, LaneT from, LaneT to)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::make_output_buffer_for_out_interval_impl<pref, I>(
key, from, to);
}
// ---------------------------------------------------------------------------
// eval_inner_product(target, key, from, to, weights [, memo])
//
// Point-slot inner product: same interior walk as `eval_interval(out<I>, …)`
// but each packed leaf is multiply-accumulated against a public weight vector
// instead of being materialized. Additive outputs sum `DPF_I(x)·w[x]`; XOR
// outputs (`bit` / `xor_wrapper`) xor `DPF_I(x) & w[x]`. Weights are indexed in
// the slot's lane domain, matching `eval_interval`'s destination layout.
//
// Cmp inner product: dot of the per-point comparison path-sum shares with the
// weights (no leaf MAC); the two parties' results reconstruct to the true dot.
// ---------------------------------------------------------------------------
namespace detail
{
namespace incr
{
template <typename OutputT, typename NodeT>
struct ml_ip_accum
{
static constexpr bool xor_mode =
std::is_same_v<OutputT, dpf::bit> || utils::is_xor_wrapper_v<OutputT>;
psnip_uint64_t acc = 0;
template <typename LeafT, typename W>
void mac(const LeafT & leaf, std::size_t base, std::size_t opl, W && w)
{
for (std::size_t p = 0; p < opl; ++p)
{
psnip_uint64_t val;
if constexpr (utils::is_packed_subbyte_v<OutputT>)
{
val = static_cast<psnip_uint64_t>(
dpf::extract_leaf<NodeT, OutputT>(leaf, p));
}
else
{
OutputT v;
std::memcpy(&v,
reinterpret_cast<const unsigned char *>(std::addressof(leaf))
+ p * sizeof(OutputT),
sizeof(v));
if constexpr (utils::is_xor_wrapper_v<OutputT>)
{
// `static_cast<psnip_uint64_t>(v)` is ambiguous for
// `xor_wrapper` (both `operator bool` and `operator T`
// are viable). Go through the concrete underlying bits.
val = static_cast<psnip_uint64_t>(v.data());
}
else
{
val = static_cast<psnip_uint64_t>(v);
}
}
const auto wt = static_cast<psnip_uint64_t>(w[base + p]);
if constexpr (xor_mode)
acc ^= (val & wt);
else if constexpr (utils::is_packed_subbyte_v<OutputT>)
{
constexpr auto mask
= (static_cast<psnip_uint64_t>(1)
<< utils::packed_lane_bits_v<OutputT>)
- 1;
acc = (acc + (val & mask) * (wt & mask)) & mask;
}
else
acc += val * wt;
}
}
OutputT finish() const
{
if constexpr (std::is_same_v<OutputT, dpf::bit>)
return OutputT{static_cast<bool>(acc & 1)};
else if constexpr (utils::is_packed_subbyte_v<OutputT>)
return static_cast<OutputT>(acc);
else if constexpr (utils::is_xor_wrapper_v<OutputT>)
return OutputT{static_cast<typename OutputT::value_type>(acc)};
else
return static_cast<OutputT>(acc);
}
};
template <std::size_t N, std::size_t I, typename KeyT, typename LaneT,
typename Weights, typename IntervalMemoizer>
auto eval_out_inner_product_impl(const KeyT & dpf, LaneT from, LaneT to,
Weights && weights, IntervalMemoizer && memoizer)
{
using key_type = KeyT;
static_assert(key_type::meta[I].prefix == N,
"out inner product: N does not match output I");
using output_type = typename key_type::template concrete_output_type<I>;
using exterior_node = typename key_type::exterior_node;
using integral_type = typename key_type::integral_type;
constexpr auto opl = key_type::template outputs_per_leaf_of<I>;
constexpr auto lg_opl = key_type::template lg_outputs_per_leaf_of<I>;
constexpr auto to_level = key_type::meta[I].tree_level;
constexpr auto to_int = utils::to_integral_type<LaneT>{};
utils::flip_msb_if_signed_integral(from);
utils::flip_msb_if_signed_integral(to);
const auto from_i = static_cast<integral_type>(to_int(from));
const auto to_i = static_cast<integral_type>(to_int(to));
integral_type from_node = utils::leaf_node_floor(from_i, lg_opl);
integral_type to_node = utils::leaf_node_ceil_exclusive(to_i, lg_opl);
const bool wraps = utils::interval_wraps(from_i, to_i, N);
const auto segs = utils::split_leaf_nodes(from_node, to_node, to_level, wraps);
HEDLEY_PRAGMA(GCC diagnostic push)
HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes")
ml_ip_accum<output_type, exterior_node> acc{};
HEDLEY_PRAGMA(GCC diagnostic pop)
std::size_t start = 0;
for (std::size_t s = 0; s < segs.n; ++s)
{
const auto & seg = segs.seg[s];
internal::eval_out_interval_interior<N, I>(dpf, seg.from_node,
seg.to_node, memoizer);
auto * nodes = memoizer[to_level];
const std::size_t count =
static_cast<std::size_t>(seg.to_node - seg.from_node);
for (std::size_t j = 0; j < count; ++j)
{
auto leaf = dpf.template traverse_exterior<I>(nodes[j]);
detail::incr::absorb_public_addend_all_lanes<I>(dpf, leaf);
acc.mac(leaf, (start + j) * opl, opl, weights);
}
start += seg.count;
}
return acc.finish();
}
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
typename Weights>
Beta eval_cmp_inner_product_impl(const KeyT & dpf, LaneT from, LaneT to,
Weights && weights, proof_token * pi = nullptr)
{
if (!dpf.has_cmp())
throw std::invalid_argument("cmp inner product: no comparison channel");
if (!dpf.cmp_assigned())
throw std::invalid_argument(
"cmp inner product: wildcard payload not assigned (call assign_cmp)");
constexpr auto to_int = utils::to_integral_type<LaneT>{};
utils::flip_msb_if_signed_integral(from);
utils::flip_msb_if_signed_integral(to);
const auto nbits = static_cast<std::size_t>(dpf.cmp().nbits);
const uint64_t mask = dpf.cmp().mask;
using integral = typename KeyT::integral_type;
const auto a = static_cast<integral>(to_int(from));
const auto b = static_cast<integral>(to_int(to));
const auto count = cmp_inclusive_count(a, b);
constexpr std::size_t stop =
KeyT::cmp_depth == 0 ? KeyT::depth : KeyT::cmp_depth;
detail::incr::cmp_full_interval_memo<KeyT, stop> memo{count};
const std::size_t levels = unwrap_party_key_t<KeyT>::cmp_block > 0
? unwrap_party_key_t<KeyT>::cmp_h : nbits;
detail::incr::eval_cmp_interval_impl_interior(dpf, a, cmp_exclusive_end(b),
nbits, memo, levels, pi);
uint64_t dot = 0;
for (std::size_t i = 0; i < count; ++i)
{
const auto q = static_cast<integral>(a + static_cast<integral>(i));
const uint64_t raw = [&] {
if constexpr (unwrap_party_key_t<KeyT>::cmp_block > 0)
{
return detail::blocked::eval_share_memo(dpf, q, a,
cmp_exclusive_end(b), memo);
}
else
{
return detail::incr::eval_cmp_from_interval_memo(
dpf, q, a, nbits, memo);
}
}();
const uint64_t wt = static_cast<uint64_t>(weights[i]) & mask;
dot = (dot + ((raw & mask) * wt)) & mask;
}
return detail::dcf_impl::u64_to_beta<Beta>(dot);
}
} // namespace incr
} // namespace detail
/// \complexity Same interior expansion as `eval_interval` on `[from, to]` (Θ(L) nodes, L = leaf nodes in the interval) plus a multiply-add per output slot into an O(1) accumulator. The basic memoizer still holds O(L) nodes.
template <std::size_t I, std::size_t N, typename KeyT, typename LaneT,
typename Weights, typename IntervalMemoizer,
std::enable_if_t<is_multilevel_key_v<KeyT>, bool> = true,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_inner_product(out_t<I, N>, const KeyT & key, LaneT from, LaneT to,
Weights && weights, IntervalMemoizer && memo)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_inner_product_impl<pref, I>(key, from, to,
std::forward<Weights>(weights),
std::forward<IntervalMemoizer>(memo));
}
/// \complexity Same interior expansion as `eval_interval` on `[from, to]` (Θ(L) nodes, L = leaf nodes in the interval) plus a multiply-add per output slot into an O(1) accumulator. The basic memoizer still holds O(L) nodes.
template <std::size_t I, std::size_t N, typename KeyT, typename LaneT,
typename Weights,
std::enable_if_t<is_multilevel_key_v<KeyT>, bool> = true,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_inner_product(out_t<I, N>, const KeyT & key, LaneT from, LaneT to,
Weights && weights)
{
auto memo = make_basic_interval_memoizer<KeyT, I>(key, from, to);
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
return detail::incr::eval_out_inner_product_impl<pref, I>(key, from, to,
std::forward<Weights>(weights), memo);
}
/// \complexity Same interior expansion as `eval_interval` on `[from, to]` (Θ(L) nodes, L = leaf nodes in the interval) plus a multiply-add per output slot into an O(1) accumulator. The basic memoizer still holds O(L) nodes.
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
typename Weights,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
Beta eval_inner_product(cmp_t, const KeyT & key, LaneT from, LaneT to,
Weights && weights)
{
return detail::incr::eval_cmp_inner_product_impl<Beta>(key, from, to,
std::forward<Weights>(weights));
}
/// @brief Comparison inner product over the whole comparison domain.
/// @details `sum_x [x satisfies cmp] * weights[x]` (as complementary halves),
/// the full-domain form of `eval_inner_product(cmp, key, lo, hi, w)`.
/// Pair the two parties' results with `reconstruct_cmp_halves`.
/// Waldo's private-threshold aggregate is this one call.
/// \complexity Same expansion as `eval_full` on the comparison domain, plus a
/// multiply-add per point into an `O(1)` accumulator.
template <typename Beta = uint64_t, typename KeyT, typename Weights,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
Beta eval_full_inner_product(cmp_t, const KeyT & key, Weights && weights)
{
if (!key.has_cmp())
throw std::invalid_argument(
"eval_full_inner_product(cmp): no comparison channel");
using lane_t = typename KeyT::integral_type;
const auto nbits = static_cast<std::size_t>(key.cmp().nbits);
const lane_t lo = 0;
const lane_t hi = (nbits >= 8 * sizeof(lane_t))
? static_cast<lane_t>(~lane_t{0})
: static_cast<lane_t>((lane_t{1} << nbits) - 1);
return eval_inner_product<Beta>(cmp, key, lo, hi,
std::forward<Weights>(weights));
}
/// \complexity Same interior expansion as `eval_interval` on `[from, to]` (Θ(L) nodes, L = leaf nodes in the interval) plus a multiply-add per output slot into an O(1) accumulator. The basic memoizer still holds O(L) nodes.
template <typename Beta = uint64_t, typename KeyT, typename LaneT,
typename Weights,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
Beta eval_inner_product(cmp_t, const KeyT & key, LaneT from, LaneT to,
Weights && weights, prove_ref pr)
{
static_assert(KeyT::is_verifiable,
"eval_inner_product(cmp, ..., prove(π)): key must carry dpf::verifiable");
detail::vdpf::init_proof(pr.token, key);
auto out = detail::incr::eval_cmp_inner_product_impl<Beta>(key, from, to,
std::forward<Weights>(weights), &pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
return out;
}
/// @brief Initialise `pr.token` and fold `[from, to]` once per cmp BFS node.
template <typename KeyT, typename LaneT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void prove_cmp_interval(const KeyT & key, LaneT from, LaneT to, prove_ref pr)
{
static_assert(KeyT::is_verifiable,
"prove_cmp_interval: key must carry dpf::verifiable");
detail::vdpf::init_proof(pr.token, key);
detail::incr::prove_fold_cmp_interval(key, from, to, pr.token);
detail::vdpf::fold_output_binding(pr.token, key);
}
/// @brief Initialise `pr.token` and fold the full comparison domain.
template <typename KeyT,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void prove_cmp_full(const KeyT & key, prove_ref pr)
{
static_assert(KeyT::is_verifiable,
"prove_cmp_full: key must carry dpf::verifiable");
if (!key.has_cmp())
throw std::invalid_argument("prove_cmp_full: no comparison channel");
using lane_t = typename KeyT::integral_type;
const auto nbits = static_cast<std::size_t>(key.cmp().nbits);
const lane_t lo = 0;
const lane_t hi = (nbits >= 8 * sizeof(lane_t))
? static_cast<lane_t>(~lane_t{0})
: static_cast<lane_t>((lane_t{1} << nbits) - 1);
prove_cmp_interval(key, lo, hi, pr);
}
/// @brief Fold a sorted sequence into `pr` via cmp interval-run covers.
template <typename KeyT, typename ForwardIterator,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void prove_cmp_sequence(const KeyT & key, ForwardIterator begin,
ForwardIterator end, prove_ref pr)
{
static_assert(KeyT::is_verifiable,
"prove_cmp_sequence: key must carry dpf::verifiable");
if (HEDLEY_UNLIKELY(begin != end && !std::is_sorted(begin, end)))
throw std::runtime_error("list must be sorted");
detail::vdpf::init_proof(pr.token, key);
using lane_t = typename KeyT::integral_type;
for (auto it = begin; it != end; )
{
const auto run_from = static_cast<lane_t>(*it);
auto run_to = run_from;
++it;
while (it != end)
{
const auto next = static_cast<lane_t>(*it);
if (next != static_cast<lane_t>(run_to + lane_t{1}))
break;
run_to = next;
++it;
}
detail::incr::prove_fold_cmp_interval(key, run_from, run_to, pr.token);
}
detail::vdpf::fold_output_binding(pr.token, key);
}
// ---------------------------------------------------------------------------
// eval_sequence_breadth_first(out<I>, key, begin, end [, outbuf])
//
// Breadth-first sequence eval that stops the interior walk at `meta[I]
// .tree_level` (the leaf level of slot `I`) instead of the full key depth.
// `begin`/`end` are a *sorted* range of lane points in `[0, 2^N)` (top-N-bit
// prefixes); the result is written output-only, one value per query point in
// query order (`outbuf[i]` is the output for the `i`-th query).
// ---------------------------------------------------------------------------
namespace detail
{
namespace incr
{
template <std::size_t N, std::size_t I, typename KeyT,
typename ForwardIterator, typename OutputBuffer>
void eval_out_sequence_breadth_first_impl(const KeyT & dpf,
ForwardIterator begin, ForwardIterator end, OutputBuffer && outbuf)
{
using key_type = KeyT;
static_assert(key_type::meta[I].prefix == N,
"breadth-first out sequence: N does not match output I");
using input_type = typename key_type::input_type;
using node_type = typename key_type::interior_node;
using exterior_node = typename key_type::exterior_node;
using output_type = typename key_type::template concrete_output_type<I>;
constexpr std::size_t stop = key_type::meta[I].tree_level;
constexpr std::size_t lg_opl = key_type::template lg_outputs_per_leaf_of<I>;
constexpr std::size_t opl = std::size_t{1} << lg_opl;
if (HEDLEY_UNLIKELY(!std::is_sorted(begin, end)))
throw std::runtime_error("breadth-first sequence: list must be sorted");
if (begin == end)
return;
HEDLEY_PRAGMA(GCC diagnostic push)
HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes")
using allocator = aligned_allocator<node_type>;
HEDLEY_PRAGMA(GCC diagnostic pop)
allocator alloc{};
const std::size_t nseq = static_cast<std::size_t>(std::distance(begin, end));
auto memo = alloc.allocate_unique_ptr(nseq * 2);
if (HEDLEY_UNLIKELY(memo == nullptr))
throw std::bad_alloc{};
input_type mask = static_cast<input_type>(input_type{1} << (N - 1));
bool curhalf = (stop ^ 1) & 1;
memo[static_cast<std::size_t>(!curhalf) * nseq + 0] = dpf.root();
std::list<ForwardIterator> splits{begin, end};
std::size_t level_index = 1;
auto step = [&]() {
std::size_t i = 0, j = 0;
const node_type cw[2] = {
dpf.correction_word(level_index - 1, 0),
dpf.correction_word(level_index - 1, 1)};
const bool is_last = key_type::tree::is_last_level(level_index - 1,
key_type::depth);
const std::size_t cur = static_cast<std::size_t>(curhalf) * nseq;
const std::size_t prv = static_cast<std::size_t>(!curhalf) * nseq;
for (auto upper = std::begin(splits), lower = upper++;
upper != std::end(splits); lower = upper++)
{
auto it = std::upper_bound(*lower, *upper, mask,
[](auto a, auto b) { return static_cast<bool>(a & b); });
if (it == *lower)
{
memo[cur + i++] = key_type::traverse_interior(
memo[prv + j++], cw[1], 1, is_last);
}
else if (it == *upper)
{
memo[cur + i++] = key_type::traverse_interior(
memo[prv + j++], cw[0], 0, is_last);
}
else
{
auto kids = key_type::traverse_interior01(memo[prv + j++],
cw[0], cw[1], is_last);
memo[cur + i++] = kids[0];
memo[cur + i++] = kids[1];
splits.insert(upper, it);
}
}
};
for (; level_index <= stop;
++level_index, mask >>= 1, curhalf = !curhalf)
step();
auto * buf = memo.get(); // deepest built level (stop) lands in half 0
auto curr = begin, prev = begin;
std::size_t j = 0;
for (std::size_t i = 0; i < nseq; ++i)
{
if (i > 0
&& (static_cast<input_type>(*curr) >> lg_opl)
!= (static_cast<input_type>(*prev) >> lg_opl))
++j;
auto leaf = dpf.template traverse_exterior<I>(buf[j]);
const std::size_t off =
static_cast<std::size_t>(static_cast<input_type>(*curr) & (opl - 1));
detail::incr::absorb_public_addend_lane<I>(dpf, leaf,
static_cast<input_type>(off));
auto v = dpf::extract_leaf<exterior_node, output_type>(leaf, off);
if constexpr (is_party_key_v<KeyT>)
outbuf[i] = subtractive_share<output_type, party_of_v<KeyT>>::from_raw(v);
else
outbuf[i] = v;
prev = curr++;
}
}
} // namespace incr
} // namespace detail
/// \complexity O(n k) interior traversals in the worst case and O(k) node workspace. k is the number of listed points and n is `depth`. The breadth-first buffer is 2k nodes, so each level traverses at most one node per point. Shared prefixes do fewer traversals. A recipe memoizer instead stores O(recipe leaf nodes) (see that memoizer).
template <std::size_t I, std::size_t N, typename KeyT,
typename ForwardIterator, typename OutputBuffer,
std::enable_if_t<is_multilevel_key_v<KeyT>, bool> = true,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
void eval_sequence_breadth_first(out_t<I, N>, const KeyT & key,
ForwardIterator begin, ForwardIterator end, OutputBuffer && outbuf)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
detail::incr::eval_out_sequence_breadth_first_impl<pref, I>(key, begin, end,
std::forward<OutputBuffer>(outbuf));
}
/// \complexity O(n k) interior traversals in the worst case and O(k) node workspace. k is the number of listed points and n is `depth`. The breadth-first buffer is 2k nodes, so each level traverses at most one node per point. Shared prefixes do fewer traversals. A recipe memoizer instead stores O(recipe leaf nodes) (see that memoizer).
template <std::size_t I, std::size_t N, typename KeyT,
typename ForwardIterator,
std::enable_if_t<is_multilevel_key_v<KeyT>, bool> = true,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto eval_sequence_breadth_first(out_t<I, N>, const KeyT & key,
ForwardIterator begin, ForwardIterator end)
{
using output_type = typename KeyT::template concrete_output_type<I>;
const std::size_t n = static_cast<std::size_t>(std::distance(begin, end));
dpf::output_buffer<leaf_buffer_elem_t<KeyT, output_type>> buf(n);
eval_sequence_breadth_first(out_t<I, N>{}, key, begin, end, buf);
return buf;
}
/// @brief Build a sequence recipe stopped at slot `I`'s tree level (prefix domain).
/// @tparam I output index
/// @tparam N width in bits
/// @tparam KeyT key type
/// @tparam ForwardIterator forward iterator type
/// @tparam KeyT key type
/// @param key the `key`
/// @param begin the iterator to the first query
/// @param end the iterator past the last query
/// @return the constructed object
template <std::size_t I, std::size_t N, typename KeyT, typename ForwardIterator,
std::enable_if_t<is_multilevel_key_v<KeyT>, bool> = true,
std::enable_if_t<!has_embedded_dpf_key_v<std::decay_t<KeyT>>, int> = 0>
auto make_sequence_recipe(out_t<I, N>, const KeyT & key, ForwardIterator begin,
ForwardIterator end)
{
constexpr auto pref = detail::resolved_out_prefix<I, N, KeyT>();
using input_type = typename KeyT::input_type;
constexpr auto stop = KeyT::meta[I].tree_level;
constexpr auto lg = KeyT::template lg_outputs_per_leaf_of<I>;
const input_type lane_msb =
static_cast<input_type>(input_type{1} << (pref - 1));
(void)key;
return make_sequence_recipe_at<stop, lg, input_type>(lane_msb, begin, end);
}
} // namespace dpf
#endif // LIBDPF_INCLUDE_DPF_EVAL_UNIFIED_HPP__