libdpf/test/profile/eval_profile.cpp
Ryan Henry 0d22946a0e Checkpoint the party/runtime stack before share-program and malicious-mode work.
Ship the TLS mesh, composer, Beaver/Yao/leaf MPC, prep/online paths, apps, and docs so the tree is pushable before elevating share_expr, security_mode, and prep resume.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-28 05:59:19 -06:00

907 lines
42 KiB
C++

/// @file test/profile/eval_profile.cpp
/// @brief Workload for sequence and interval evaluation.
///
/// Both party keys are evaluated inside each sample. Key generation is its
/// own case, not part of the eval samples. `out_bytes` is the output buffer
/// volume of one sample (both parties). `per_item_ns` divides by the point
/// count, not by the number of parties.
///
/// profile_eval --list
/// profile_eval --case interval_u32_L4096 --repeat 50 --warmup 3
///
/// Profile-guided build, separate from COVERAGE:
/// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release
/// cmake --build build-pgo --target profile_eval
/// build-pgo/bin/profile_eval --repeat 40 --warmup 2
/// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release
/// cmake --build build-pgo --target profile_eval
#include "harness.hpp"
#include "dpf.hpp"
#include <cstdint>
#include <initializer_list>
#include <memory>
#include <string>
#include <utility>
#include <vector>
#include <algorithm>
#include <cstring>
namespace
{
using profile::sample;
using profile::touch_buf;
using profile::touch_word;
using profile::work;
template <typename Input>
std::shared_ptr<std::vector<Input>> clustered(Input start, std::size_t n)
{
auto pts = std::make_shared<std::vector<Input>>(n);
for (std::size_t i = 0; i < n; ++i)
(*pts)[i] = static_cast<Input>(start + static_cast<Input>(i));
return pts;
}
template <typename Input>
std::shared_ptr<std::vector<Input>> strided(Input start, Input step, std::size_t n)
{
auto pts = std::make_shared<std::vector<Input>>(n);
for (std::size_t i = 0; i < n; ++i)
(*pts)[i] = static_cast<Input>(start + step * static_cast<Input>(i));
return pts;
}
template <typename Keys>
sample hash_roots(const Keys & keys)
{
std::uint64_t w0 = 0;
std::uint64_t w1 = 0;
const auto r0 = std::get<0>(keys).root();
const auto r1 = std::get<1>(keys).root();
const std::size_t n0 = sizeof(r0) < sizeof(w0) ? sizeof(r0) : sizeof(w0);
const std::size_t n1 = sizeof(r1) < sizeof(w1) ? sizeof(r1) : sizeof(w1);
std::memcpy(&w0, &r0, n0);
std::memcpy(&w1, &r1, n1);
return touch_word(w0 ^ w1, sizeof(r0) + sizeof(r1));
}
template <typename Keys, typename Input>
sample interval_packed(const Keys & keys, Input from, Input to, unsigned lane_bits)
{
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(keys), from, to);
auto b = dpf::eval_interval(std::get<1>(keys), from, to);
auto fold = [lane_bits](const auto & buf) {
std::uint64_t w = 0;
const std::size_t bytes = (buf.size() * lane_bits + 7u) / 8u;
if (bytes != 0 && buf.data() != nullptr)
{
const std::size_t n = bytes < sizeof(w) ? bytes : sizeof(w);
std::memcpy(&w, buf.data(), n);
}
return touch_word(w, bytes);
};
return fold(a.first) + fold(b.first);
});
}
template <typename Keys, typename Input>
sample interval_both(const Keys & keys, Input from, Input to)
{
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(keys), from, to);
auto b = dpf::eval_interval(std::get<1>(keys), from, to);
return touch_buf(a.first) + touch_buf(b.first);
});
}
template <typename Keys, typename Input>
sample interval_reuse(const Keys & keys, Input from, Input to,
decltype(dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to)) & buf0,
decltype(dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to)) & buf1)
{
return profile::with_prg([&] {
auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0);
auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1);
(void)i0;
(void)i1;
return touch_buf(buf0) + touch_buf(buf1);
});
}
template <typename Keys, typename Input>
sample interval_prove(const Keys & keys, Input from, Input to)
{
auto buf0 = dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to);
auto buf1 = dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to);
dpf::proof_token p0{};
dpf::proof_token p1{};
auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0, dpf::prove(p0));
auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1, dpf::prove(p1));
(void)i0;
(void)i1;
std::uint64_t extra = 0;
std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0));
return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0);
}
template <typename Keys, typename Pts>
sample sequence_both(const Keys & keys, const Pts & pts, bool output_only)
{
return profile::with_prg([&] {
sample s;
if (output_only)
{
auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(),
dpf::return_output_only_tag_{});
auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(),
dpf::return_output_only_tag_{});
s = touch_buf(a.first) + touch_buf(b.first);
}
else
{
auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end());
auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end());
s = touch_buf(a.first) + touch_buf(b.first);
}
return s;
});
}
template <typename Keys, typename Pts>
sample sequence_breadth(const Keys & keys, const Pts & pts)
{
return profile::with_prg([&] {
auto a = dpf::eval_sequence_breadth_first(std::get<0>(keys), pts.begin(), pts.end());
auto b = dpf::eval_sequence_breadth_first(std::get<1>(keys), pts.begin(), pts.end());
return touch_buf(a.first) + touch_buf(b.first);
});
}
template <typename Keys, typename Recipe0, typename Recipe1>
sample sequence_recipe(const Keys & keys, const Recipe0 & r0, const Recipe1 & r1)
{
return profile::with_prg([&] {
auto a = dpf::eval_sequence(std::get<0>(keys), r0);
auto b = dpf::eval_sequence(std::get<1>(keys), r1);
return touch_buf(a.first) + touch_buf(b.first);
});
}
template <typename Keys, typename Pts>
sample sequence_prove(const Keys & keys, const Pts & pts)
{
auto buf0 = dpf::make_output_buffer_for_subsequence(std::get<0>(keys),
pts.begin(), pts.end(), dpf::return_output_only_tag_{});
auto buf1 = dpf::make_output_buffer_for_subsequence(std::get<1>(keys),
pts.begin(), pts.end(), dpf::return_output_only_tag_{});
dpf::proof_token p0{};
dpf::proof_token p1{};
auto i0 = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(), buf0,
dpf::prove(p0), dpf::return_output_only_tag_{});
auto i1 = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(), buf1,
dpf::prove(p1), dpf::return_output_only_tag_{});
(void)i0;
(void)i1;
std::uint64_t extra = 0;
std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0));
return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0);
}
template <typename Input, typename... Extra>
auto make_keys(Input alpha, Extra && ...extra)
{
auto keys = dpf::make_dpf(alpha, std::uint64_t{0x9e3779b97f4a7c15ull},
std::forward<Extra>(extra)...);
using keys_t = std::decay_t<decltype(keys)>;
return std::make_shared<keys_t>(std::move(keys));
}
template <typename Input, typename Output, typename... Extra>
auto make_out_keys(Input alpha, Output beta, Extra && ...extra)
{
auto keys = dpf::make_dpf(alpha, beta, std::forward<Extra>(extra)...);
using keys_t = std::decay_t<decltype(keys)>;
return std::make_shared<keys_t>(std::move(keys));
}
const char kUsage[] =
"profile_eval [--list] [--family F] [--slice S] [--case NAME] [--tier std]\n"
" [--repeat N=20] [--warmup W=2]\n"
"Slices: keygen, interval, interval-shape, sequence, sequence-algo,\n"
" memoizer, bits, gf, inner-product, helpers, defer.\n"
"Times sequence and interval evaluation on both party keys.\n";
template <typename Input, typename KeysPtr>
void add_interval_at(std::vector<work> & works, const char * slice,
const std::string & name, Input from, Input to, const KeysPtr & keys)
{
const auto items = static_cast<std::uint64_t>(to) - static_cast<std::uint64_t>(from) + 1;
works.push_back(profile::make_work("eval", slice, name, items, [keys, from, to] {
return interval_both(*keys, from, to);
}));
}
} // namespace
int main(int argc, char ** argv)
{
const auto opt = profile::parse_args(argc, argv, 20, 2, kUsage);
using u8 = std::uint8_t;
using u16 = std::uint16_t;
using u32 = std::uint32_t;
const auto k8 = make_keys(u8{40});
const auto k16 = make_keys(u16{1512});
const auto k16v = make_keys(u16{1512}, dpf::verifiable{});
const auto k32 = make_keys(u32{1002048u});
const auto k32v = make_keys(u32{1002048u}, dpf::verifiable{});
std::vector<work> works;
works.push_back(profile::make_work("eval", "keygen", "keygen_u8", 1, [] {
return hash_roots(*make_keys(u8{40}));
}));
works.push_back(profile::make_work("eval", "keygen", "keygen_u16", 1, [] {
return hash_roots(*make_keys(u16{1512}));
}));
works.push_back(profile::make_work("eval", "keygen", "keygen_u32", 1, [] {
return hash_roots(*make_keys(u32{0x01000000u}));
}));
works.push_back(profile::make_work("eval", "keygen", "keygen_u32_verifiable", 1, [] {
return hash_roots(*make_keys(u32{0x01000000u}, dpf::verifiable{}));
}));
const auto add_gf = [&](const char * name, auto beta) {
using out = std::decay_t<decltype(beta)>;
works.push_back(profile::make_work("eval", "gf",
std::string("keygen_") + name, 1, [beta] {
return hash_roots(*make_out_keys(u8{40}, beta));
}));
works.push_back(profile::make_work("eval", "gf",
std::string("keygen_") + name + "_verifiable", 1, [beta] {
return hash_roots(*make_out_keys(u8{40}, beta, dpf::verifiable{}));
}));
const auto keys = make_out_keys(u8{40}, beta);
if constexpr (dpf::utils::is_packed_subbyte_v<out>)
{
constexpr unsigned bits = dpf::utils::packed_lane_bits_v<out>;
works.push_back(profile::make_work("eval", "gf",
std::string("interval_") + name + "_u8_L256", 256,
[keys, bits] {
return interval_packed(*keys, u8{0}, u8{255}, bits);
}));
}
else
{
add_interval_at(works, "gf", std::string("interval_") + name + "_u8_L256",
u8{0}, u8{255}, keys);
}
};
add_gf("gf2", dpf::gf2{1});
add_gf("gf22", dpf::gf22{3});
add_gf("gf24", dpf::gf24{0xa});
add_gf("gf28", dpf::gf28{0x1b});
add_gf("gf216", dpf::gf216{0x2d});
add_gf("gf232", dpf::gf232{0x90200001u});
add_gf("gf264", dpf::gf264{0x11});
const auto add_lengths = [&](auto from0, auto keys, const char * width,
std::initializer_list<std::uint64_t> lengths) {
using input = decltype(from0);
for (const std::uint64_t n : lengths)
{
const input from = from0;
const input to = static_cast<input>(from + static_cast<input>(n - 1));
add_interval_at(works, "interval",
std::string("interval_") + width + "_L" + std::to_string(n),
from, to, keys);
}
};
add_lengths(u8{0}, k8, "u8", {1, 16, 64, 256});
add_lengths(u16{1000}, k16, "u16", {1, 16, 256, 1024, 4096});
add_lengths(u32{1000000}, k32, "u32", {1, 16, 64, 256, 1024, 4096, 16384});
add_interval_at(works, "interval-shape", "interval_u32_L256_unaligned",
u32{1000003}, u32{1000258}, k32);
add_interval_at(works, "interval-shape", "interval_u32_L4096_unaligned",
u32{1000003}, u32{1004098}, k32);
const auto add_reuse = [&](u32 from, u32 to, const char * name) {
using buf0_t = std::decay_t<decltype(dpf::make_output_buffer_for_interval(
std::get<0>(*k32), from, to))>;
using buf1_t = std::decay_t<decltype(dpf::make_output_buffer_for_interval(
std::get<1>(*k32), from, to))>;
auto buf0 = std::make_shared<buf0_t>(
dpf::make_output_buffer_for_interval(std::get<0>(*k32), from, to));
auto buf1 = std::make_shared<buf1_t>(
dpf::make_output_buffer_for_interval(std::get<1>(*k32), from, to));
const auto items = static_cast<std::uint64_t>(to - from + 1);
works.push_back(profile::make_work("eval", "interval-shape", name, items,
[k32, buf0, buf1, from, to] {
return interval_reuse(*k32, from, to, *buf0, *buf1);
}));
};
add_reuse(1000000, 1000255, "interval_u32_L256_reuse");
add_reuse(1000000, 1004095, "interval_u32_L4096_reuse");
works.push_back(profile::make_work("eval", "interval-shape",
"interval_u16_L1024_verifiable", 1024, [k16v] {
return interval_prove(*k16v, u16{1000}, u16{2023});
}));
works.push_back(profile::make_work("eval", "interval-shape",
"interval_u32_L256_verifiable", 256, [k32v] {
return interval_prove(*k32v, u32{1000000}, u32{1000255});
}));
const auto add_seq = [&](auto keys, const char * width, const char * shape,
auto pts) {
const auto n = static_cast<std::uint64_t>(pts->size());
const std::string name = std::string("sequence_") + width + "_" + shape
+ "_" + std::to_string(n);
works.push_back(profile::make_work("eval", "sequence", name, n,
[keys, pts] {
return sequence_both(*keys, *pts, false);
}));
};
const std::size_t seq32[] = {8, 32, 64, 128, 256, 512, 1024, 2048};
for (const std::size_t n : seq32)
{
add_seq(k32, "u32", "cluster", clustered<u32>(1u << 20, n));
const u32 step = n <= 64 ? u32{1u << 20} : u32{1u << 12};
add_seq(k32, "u32", "stride", strided<u32>(16u, step, n));
}
const std::size_t seq16[] = {8, 64, 256, 1024};
for (const std::size_t n : seq16)
add_seq(k16, "u16", "cluster", clustered<u16>(1000, n));
const std::size_t seq8[] = {8, 32, 64};
for (const std::size_t n : seq8)
{
add_seq(k8, "u8", "cluster", clustered<u8>(0, n));
add_seq(k8, "u8", "stride", strided<u8>(0, 3, n));
}
const auto c256 = clustered<u32>(1u << 20, 256);
const auto s64 = strided<u32>(16u, 1u << 20, 64);
const auto c16s = clustered<u16>(1000, 64);
const auto c32s = clustered<u32>(1u << 20, 64);
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_cluster_256_output_only", 256, [k32, c256] {
return sequence_both(*k32, *c256, true);
}));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_cluster_256_breadth", 256, [k32, c256] {
return sequence_breadth(*k32, *c256);
}));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_stride_64_breadth", 64, [k32, s64] {
return sequence_breadth(*k32, *s64);
}));
using recipe0_t = std::decay_t<decltype(dpf::make_sequence_recipe(
std::get<0>(*k32), c256->begin(), c256->end()))>;
using recipe1_t = std::decay_t<decltype(dpf::make_sequence_recipe(
std::get<1>(*k32), c256->begin(), c256->end()))>;
auto recipe0 = std::make_shared<recipe0_t>(dpf::make_sequence_recipe(
std::get<0>(*k32), c256->begin(), c256->end()));
auto recipe1 = std::make_shared<recipe1_t>(dpf::make_sequence_recipe(
std::get<1>(*k32), c256->begin(), c256->end()));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_cluster_256_recipe", 256, [k32, recipe0, recipe1] {
return sequence_recipe(*k32, *recipe0, *recipe1);
}));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u16_cluster_64_verifiable", 64, [k16v, c16s] {
return sequence_prove(*k16v, *c16s);
}));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_cluster_64_verifiable", 64, [k32v, c32s] {
return sequence_prove(*k32v, *c32s);
}));
// --- memoizer: built-in reuse vs hand-rolled pointwise / fresh memo ---
{
using u32 = std::uint32_t;
const u32 from = 1000000u;
const u32 to = 1000255u; // L256
const auto items = static_cast<std::uint64_t>(to - from + 1);
const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32));
using dpf_t = std::decay_t<decltype(std::get<0>(*k32))>;
using node_t = typename dpf_t::interior_node;
const auto out_elem = sizeof(std::uint64_t);
const auto logical = items * out_elem * 2;
// Match `basic_interval_memoizer` capacity (leaf nodes, not output slots).
const auto leaf_nodes =
dpf::utils::get_leafnodes_in_output_interval<dpf_t>(from, to);
const auto slots = leaf_nodes == 0 ? std::size_t{1} : leaf_nodes;
const auto pivot = std::max((slots >> 1) + (slots & 1) - 1,
(slots + 6) >> 2);
const auto memo_nodes = pivot + ((slots + 2) >> 1);
const auto memo_bytes = 2ull * memo_nodes * sizeof(node_t);
auto memo0 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to))>>(
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to));
auto memo1 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to))>>(
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to));
works.push_back(profile::make_work("eval", "memoizer",
"memo_interval_reuse", items, [k32, memo0, memo1, from, to, prep, memo_bytes, logical] {
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0);
auto b = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1);
auto s = touch_buf(a.first) + touch_buf(b.first);
// Warm reuse: second pass on the same memoizers.
auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0);
auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1);
s = s + touch_buf(a2.first) + touch_buf(b2.first);
return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2);
});
}));
works.push_back(profile::make_work("eval", "memoizer",
"memo_interval_fresh", items, [k32, from, to, prep, logical, memo_bytes] {
return profile::with_prg([&] {
auto m0 = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to);
auto m1 = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to);
auto a = dpf::eval_interval(std::get<0>(*k32), from, to, m0);
auto b = dpf::eval_interval(std::get<1>(*k32), from, to, m1);
auto s = touch_buf(a.first) + touch_buf(b.first);
auto m0b = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to);
auto m1b = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to);
auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, m0b);
auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, m1b);
s = s + touch_buf(a2.first) + touch_buf(b2.first);
return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2);
});
}));
const auto pts = clustered<u32>(1u << 20, 64);
auto path0 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_path_memoizer(std::get<0>(*k32)))>>(
dpf::make_basic_path_memoizer(std::get<0>(*k32)));
auto path1 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_path_memoizer(std::get<1>(*k32)))>>(
dpf::make_basic_path_memoizer(std::get<1>(*k32)));
works.push_back(profile::make_work("eval", "memoizer",
"memo_path_reuse", 64, [k32, pts, path0, path1, prep] {
return profile::with_prg([&] {
std::uint64_t h = 0;
for (auto x : *pts)
{
h ^= static_cast<std::uint64_t>(
(*dpf::eval_point(std::get<0>(*k32), x, *path0)).raw());
h ^= static_cast<std::uint64_t>(
(*dpf::eval_point(std::get<1>(*k32), x, *path1)).raw());
}
auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2);
return profile::with_costs(s, prep,
2 * sizeof(node_t) * 32, 64 * sizeof(std::uint64_t) * 2);
});
}));
works.push_back(profile::make_work("eval", "memoizer",
"memo_path_pointwise", 64, [k32, pts, prep] {
return profile::with_prg([&] {
std::uint64_t h = 0;
for (auto x : *pts)
{
h ^= static_cast<std::uint64_t>(
(*dpf::eval_point(std::get<0>(*k32), x)).raw());
h ^= static_cast<std::uint64_t>(
(*dpf::eval_point(std::get<1>(*k32), x)).raw());
}
auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2);
return profile::with_costs(s, prep, 0,
64 * sizeof(std::uint64_t) * 2);
});
}));
using recipe0_t = std::decay_t<decltype(dpf::make_sequence_recipe(
std::get<0>(*k32), pts->begin(), pts->end()))>;
using recipe1_t = std::decay_t<decltype(dpf::make_sequence_recipe(
std::get<1>(*k32), pts->begin(), pts->end()))>;
auto rec0 = std::make_shared<recipe0_t>(dpf::make_sequence_recipe(
std::get<0>(*k32), pts->begin(), pts->end()));
auto rec1 = std::make_shared<recipe1_t>(dpf::make_sequence_recipe(
std::get<1>(*k32), pts->begin(), pts->end()));
auto smemo0 = std::make_shared<std::decay_t<decltype(
dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0))>>(
dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0));
auto smemo1 = std::make_shared<std::decay_t<decltype(
dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1))>>(
dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1));
works.push_back(profile::make_work("eval", "memoizer",
"memo_recipe_reuse", 64, [k32, rec0, rec1, smemo0, smemo1, prep] {
return profile::with_prg([&] {
auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0);
auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1);
auto s = touch_buf(a.first) + touch_buf(b.first);
auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0);
auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1);
s = s + touch_buf(a2.first) + touch_buf(b2.first);
return profile::with_costs(s, prep, s.out_bytes,
64 * sizeof(std::uint64_t) * 4);
});
}));
works.push_back(profile::make_work("eval", "memoizer",
"memo_recipe_fresh", 64, [k32, rec0, rec1, prep] {
return profile::with_prg([&] {
auto m0 = dpf::make_inplace_reversing_sequence_memoizer(
std::get<0>(*k32), *rec0);
auto m1 = dpf::make_inplace_reversing_sequence_memoizer(
std::get<1>(*k32), *rec1);
auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0);
auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1);
auto s = touch_buf(a.first) + touch_buf(b.first);
auto m0b = dpf::make_inplace_reversing_sequence_memoizer(
std::get<0>(*k32), *rec0);
auto m1b = dpf::make_inplace_reversing_sequence_memoizer(
std::get<1>(*k32), *rec1);
auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0b);
auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1b);
s = s + touch_buf(a2.first) + touch_buf(b2.first);
return profile::with_costs(s, prep, s.out_bytes,
64 * sizeof(std::uint64_t) * 4);
});
}));
}
// --- bits: built-in iterators vs hand-rolled scans (u8 domain) ---
{
using input_type = std::uint8_t;
using output_type = dpf::bit;
const input_type alpha = 40;
auto bit_keys = std::make_shared<std::decay_t<decltype(
dpf::make_dpf(alpha, output_type::one))>>(
dpf::make_dpf(alpha, output_type::one));
using dpf_type = std::decay_t<decltype(std::get<0>(*bit_keys))>;
auto memo0 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_full_memoizer<dpf_type>())>>(
dpf::make_basic_full_memoizer<dpf_type>());
auto memo1 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_full_memoizer<dpf_type>())>>(
dpf::make_basic_full_memoizer<dpf_type>());
auto full0 = dpf::eval_full(std::get<0>(*bit_keys), *memo0);
auto full1 = dpf::eval_full(std::get<1>(*bit_keys), *memo1);
auto buf0 = std::make_shared<std::decay_t<decltype(full0.first)>>(std::move(full0.first));
auto buf1 = std::make_shared<std::decay_t<decltype(full1.first)>>(std::move(full1.first));
auto iter0 = std::make_shared<std::decay_t<decltype(full0.second)>>(std::move(full0.second));
auto iter1 = std::make_shared<std::decay_t<decltype(full1.second)>>(std::move(full1.second));
const auto leaf_nodes = static_cast<std::uint64_t>(1) << dpf_type::depth;
const auto bit_items = static_cast<std::uint64_t>(buf0->size());
const auto prep = sizeof(std::get<0>(*bit_keys)) + sizeof(std::get<1>(*bit_keys));
works.push_back(profile::make_work("eval", "bits",
"bits_advice_builtin", leaf_nodes, [memo0, memo1, prep, leaf_nodes] {
std::uint64_t h = 0;
std::uint64_t n = 0;
for (auto b : dpf::advice_bits_of(*memo0))
{
h = (h << 1) ^ (b ? 1u : 0u);
++n;
}
for (auto b : dpf::advice_bits_of(*memo1))
h ^= (b ? 1u : 0u);
auto s = touch_word(h ^ n, n);
return profile::with_costs(s, prep, 0, leaf_nodes);
}));
works.push_back(profile::make_work("eval", "bits",
"bits_advice_handroll", leaf_nodes, [memo0, memo1, prep, leaf_nodes] {
std::uint64_t h = 0;
std::uint64_t n = 0;
const auto * p0 = memo0->begin();
const auto * e0 = memo0->end();
for (; p0 != e0; ++p0)
{
const auto * bytes = reinterpret_cast<const char *>(p0);
h = (h << 1) ^ (bytes[0] & 1u);
++n;
}
const auto * p1 = memo1->begin();
const auto * e1 = memo1->end();
for (; p1 != e1; ++p1)
{
const auto * bytes = reinterpret_cast<const char *>(p1);
h ^= (bytes[0] & 1u);
}
auto s = touch_word(h ^ n, n);
return profile::with_costs(s, prep, 0, leaf_nodes);
}));
works.push_back(profile::make_work("eval", "bits",
"bits_setbit_builtin", bit_items, [iter0, iter1, prep] {
std::uint64_t h = 0;
std::uint64_t n = 0;
for (auto i : dpf::indices_set_in(*iter0))
{
h ^= static_cast<std::uint64_t>(i) + 0x9e3779b97f4a7c15ull;
++n;
}
for (auto i : dpf::indices_set_in(*iter1))
h ^= static_cast<std::uint64_t>(i);
auto s = touch_word(h ^ n, n * sizeof(std::size_t));
return profile::with_costs(s, prep, 0, n * sizeof(std::size_t));
}));
works.push_back(profile::make_work("eval", "bits",
"bits_setbit_handroll", bit_items, [buf0, buf1, prep] {
std::uint64_t h = 0;
std::uint64_t n = 0;
const auto scan = [&](const auto & buf) {
for (std::size_t i = 0; i < buf.size(); ++i)
{
if (static_cast<bool>(buf[i]))
{
h ^= i + 0x9e3779b97f4a7c15ull;
++n;
}
}
};
scan(*buf0);
scan(*buf1);
auto s = touch_word(h ^ n, n * sizeof(std::size_t));
return profile::with_costs(s, prep, 0, n * sizeof(std::size_t));
}));
works.push_back(profile::make_work("eval", "bits",
"bits_parallel_builtin", bit_items, [buf0, buf1, prep, bit_items] {
std::uint64_t h = 0;
std::uint64_t n = 0;
for (auto word : dpf::batch_of(*buf0, *buf1))
{
h ^= static_cast<std::uint64_t>(word[0])
^ static_cast<std::uint64_t>(word[1]);
++n;
}
auto s = touch_word(h ^ n, n * sizeof(std::uint64_t));
return profile::with_costs(s, prep, 0, bit_items);
}));
works.push_back(profile::make_work("eval", "bits",
"bits_parallel_handroll", bit_items, [buf0, buf1, prep, bit_items] {
std::uint64_t h = 0;
const std::size_t n = std::min(buf0->size(), buf1->size());
for (std::size_t i = 0; i < n; ++i)
h ^= (static_cast<bool>((*buf0)[i]) ? 1ull : 0ull)
^ (static_cast<bool>((*buf1)[i]) ? 2ull : 0ull);
auto s = touch_word(h ^ n, n);
return profile::with_costs(s, prep, 0, bit_items);
}));
}
// --- inner-product: built-in vs hand-rolled interval + dot ---
{
using u32 = std::uint32_t;
const u32 from = 1000000u;
const u32 to = 1000255u;
const auto items = static_cast<std::uint64_t>(to - from + 1);
auto weights = std::make_shared<std::vector<std::uint64_t>>(items);
for (std::size_t i = 0; i < items; ++i)
(*weights)[i] = 0x9e3779b97f4a7c15ull * (i + 1);
const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32));
auto ip_memo0 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to))>>(
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to));
auto ip_memo1 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to))>>(
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to));
works.push_back(profile::make_work("eval", "inner-product",
"ip_interval_builtin", items,
[k32, weights, from, to, prep, items, ip_memo0, ip_memo1] {
return profile::with_prg([&] {
auto a = dpf::eval_inner_product(std::get<0>(*k32), from, to,
*weights, *ip_memo0);
auto b = dpf::eval_inner_product(std::get<1>(*k32), from, to,
*weights, *ip_memo1);
std::uint64_t ha = 0;
std::uint64_t hb = 0;
std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a));
std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b));
auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b));
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
});
}));
works.push_back(profile::make_work("eval", "inner-product",
"ip_interval_handroll", items, [k32, weights, from, to, prep, items] {
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(*k32), from, to);
auto b = dpf::eval_interval(std::get<1>(*k32), from, to);
std::uint64_t dot0 = 0;
std::uint64_t dot1 = 0;
const std::size_t n = std::min(items, a.first.size());
for (std::size_t i = 0; i < n; ++i)
{
dot0 += static_cast<std::uint64_t>(a.first[i].raw()) * (*weights)[i];
dot1 += static_cast<std::uint64_t>(b.first[i].raw()) * (*weights)[i];
}
auto s = touch_word(dot0 ^ dot1, a.first.size() * sizeof(a.first[0])
+ b.first.size() * sizeof(b.first[0]));
return profile::with_costs(s, prep, s.out_bytes,
items * sizeof(std::uint64_t) * 2);
});
}));
works.push_back(profile::make_work("eval", "inner-product",
"ip_interval_paired", items,
[k32, weights, from, to, prep, items] {
return profile::with_prg([&] {
auto a = dpf::eval_inner_product(dpf::paired,
std::get<0>(*k32), from, to, *weights);
auto b = dpf::eval_inner_product(dpf::paired,
std::get<1>(*k32), from, to, *weights);
std::uint64_t ha = 0;
std::uint64_t hb = 0;
std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a));
std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b));
auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b));
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
});
}));
works.push_back(profile::make_work("eval", "inner-product",
"ip_interval_columns", items,
[k32, weights, from, to, prep, items] {
return profile::with_prg([&] {
auto a = dpf::eval_inner_product(dpf::columns,
std::get<0>(*k32), from, to, std::tie(*weights));
auto b = dpf::eval_inner_product(dpf::columns,
std::get<1>(*k32), from, to, std::tie(*weights));
std::uint64_t ha = 0;
std::uint64_t hb = 0;
const auto & a0 = std::get<0>(a);
const auto & b0 = std::get<0>(b);
std::memcpy(&ha, &a0,
sizeof(ha) < sizeof(a0) ? sizeof(ha) : sizeof(a0));
std::memcpy(&hb, &b0,
sizeof(hb) < sizeof(b0) ? sizeof(hb) : sizeof(b0));
auto s = touch_word(ha ^ hb, sizeof(a0) + sizeof(b0));
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
});
}));
}
// --- helpers: point vs interval vs full (representative widths) ---
{
using u16 = std::uint16_t;
const auto prep16 = sizeof(std::get<0>(*k16)) + sizeof(std::get<1>(*k16));
works.push_back(profile::make_work("eval", "helpers",
"helper_point_u16", 1, [k16, prep16] {
return profile::with_prg([&] {
auto a = *dpf::eval_point(std::get<0>(*k16), u16{1512});
auto b = *dpf::eval_point(std::get<1>(*k16), u16{1512});
auto s = touch_word(
static_cast<std::uint64_t>(a.raw()) ^ static_cast<std::uint64_t>(b.raw()),
sizeof(a) + sizeof(b));
return profile::with_costs(s, prep16, 0, sizeof(a) + sizeof(b));
});
}));
works.push_back(profile::make_work("eval", "helpers",
"helper_interval_u16_L256", 256, [k16, prep16] {
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(*k16), u16{1000}, u16{1255});
auto b = dpf::eval_interval(std::get<1>(*k16), u16{1000}, u16{1255});
auto s = touch_buf(a.first) + touch_buf(b.first);
return profile::with_costs(s, prep16, s.out_bytes,
256 * sizeof(std::uint64_t) * 2);
});
}));
works.push_back(profile::make_work("eval", "helpers",
"helper_full_u8", 256, [k8] {
const auto prep = sizeof(std::get<0>(*k8)) + sizeof(std::get<1>(*k8));
return profile::with_prg([&] {
auto a = dpf::eval_full(std::get<0>(*k8));
auto b = dpf::eval_full(std::get<1>(*k8));
auto s = touch_buf(a.first) + touch_buf(b.first);
return profile::with_costs(s, prep, s.out_bytes, s.out_bytes);
});
}));
}
// --- defer: pre-assign full-domain expand vs rotate-after-assign ---
{
using u8 = std::uint8_t;
using out_t = std::uint64_t;
const u8 alpha = 40;
const out_t beta = 0x9e3779b97f4a7c15ull;
works.push_back(profile::make_work("eval", "defer",
"defer_full_u8_expand", 256, [beta] {
return profile::with_prg([&] {
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys));
auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys));
auto d0 = dpf::defer_eval_full(std::get<0>(keys), b0);
auto d1 = dpf::defer_eval_full(std::get<1>(keys), b1);
(void)d0;
(void)d1;
return touch_buf(b0) + touch_buf(b1);
});
}));
works.push_back(profile::make_work("eval", "defer",
"defer_interval_u8_L64_expand", 64, [beta] {
return profile::with_prg([&] {
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys));
auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys));
auto d0 = dpf::defer_eval_interval(std::get<0>(keys),
u8{16}, u8{79}, b0);
auto d1 = dpf::defer_eval_interval(std::get<1>(keys),
u8{16}, u8{79}, b1);
(void)d0;
(void)d1;
return touch_buf(b0) + touch_buf(b1);
});
}));
// Expand once against stable key storage, assign, then time .get().
{
using keys_t = std::decay_t<decltype(
dpf::make_dpf(dpf::wildcard_value<u8>{}, beta))>;
auto keys = std::make_shared<keys_t>(
dpf::make_dpf(dpf::wildcard_value<u8>{}, beta));
using buf0_t = std::decay_t<decltype(
dpf::make_output_buffer_for_full(std::get<0>(*keys)))>;
using buf1_t = std::decay_t<decltype(
dpf::make_output_buffer_for_full(std::get<1>(*keys)))>;
auto b0 = std::make_shared<buf0_t>(
dpf::make_output_buffer_for_full(std::get<0>(*keys)));
auto b1 = std::make_shared<buf1_t>(
dpf::make_output_buffer_for_full(std::get<1>(*keys)));
using def0_t = std::decay_t<decltype(
dpf::defer_eval_full(std::get<0>(*keys), *b0))>;
using def1_t = std::decay_t<decltype(
dpf::defer_eval_full(std::get<1>(*keys), *b1))>;
auto def0 = std::make_shared<def0_t>(
dpf::defer_eval_full(std::get<0>(*keys), *b0));
auto def1 = std::make_shared<def1_t>(
dpf::defer_eval_full(std::get<1>(*keys), *b1));
{
auto & k0 = std::get<0>(*keys);
auto & k1 = std::get<1>(*keys);
const u8 a0 = 0x12;
const u8 a1 = static_cast<u8>(alpha - a0);
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
k0.offset_x.reconstruct(sh1);
k1.offset_x.reconstruct(sh0);
}
works.push_back(profile::make_work("eval", "defer",
"defer_full_u8_get", 256, [keys, def0, def1, b0, b1] {
(void)keys;
auto v0 = def0->get();
auto v1 = def1->get();
std::uint64_t sink = 0;
for (auto it = std::begin(v0); it != std::end(v0); ++it)
sink ^= static_cast<std::uint64_t>((*it).raw());
for (auto it = std::begin(v1); it != std::end(v1); ++it)
sink ^= static_cast<std::uint64_t>((*it).raw());
return touch_word(sink, b0->size() * sizeof((*b0)[0])
+ b1->size() * sizeof((*b1)[0]));
}));
}
works.push_back(profile::make_work("eval", "defer",
"eager_full_u8_after_assign", 256, [beta, alpha] {
return profile::with_prg([&] {
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
auto & k0 = std::get<0>(keys);
auto & k1 = std::get<1>(keys);
const u8 a0 = 0x12;
const u8 a1 = static_cast<u8>(alpha - a0);
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
k0.offset_x.reconstruct(sh1);
k1.offset_x.reconstruct(sh0);
auto a = dpf::eval_full(k0);
auto b = dpf::eval_full(k1);
return touch_buf(a.first) + touch_buf(b.first);
});
}));
}
return profile::run_works(opt, works);
}