908 lines
42 KiB
C++
908 lines
42 KiB
C++
|
|
/// @file test/profile/eval_profile.cpp
|
||
|
|
/// @brief Workload for sequence and interval evaluation.
|
||
|
|
///
|
||
|
|
/// Both party keys are evaluated inside each sample. Key generation is its
|
||
|
|
/// own case, not part of the eval samples. `out_bytes` is the output buffer
|
||
|
|
/// volume of one sample (both parties). `per_item_ns` divides by the point
|
||
|
|
/// count, not by the number of parties.
|
||
|
|
///
|
||
|
|
/// profile_eval --list
|
||
|
|
/// profile_eval --case interval_u32_L4096 --repeat 50 --warmup 3
|
||
|
|
///
|
||
|
|
/// Profile-guided build, separate from COVERAGE:
|
||
|
|
/// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release
|
||
|
|
/// cmake --build build-pgo --target profile_eval
|
||
|
|
/// build-pgo/bin/profile_eval --repeat 40 --warmup 2
|
||
|
|
/// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release
|
||
|
|
/// cmake --build build-pgo --target profile_eval
|
||
|
|
|
||
|
|
#include "harness.hpp"
|
||
|
|
|
||
|
|
#include "dpf.hpp"
|
||
|
|
|
||
|
|
#include <cstdint>
|
||
|
|
#include <initializer_list>
|
||
|
|
#include <memory>
|
||
|
|
#include <string>
|
||
|
|
#include <utility>
|
||
|
|
#include <vector>
|
||
|
|
#include <algorithm>
|
||
|
|
#include <cstring>
|
||
|
|
|
||
|
|
namespace
|
||
|
|
{
|
||
|
|
|
||
|
|
using profile::sample;
|
||
|
|
using profile::touch_buf;
|
||
|
|
using profile::touch_word;
|
||
|
|
using profile::work;
|
||
|
|
|
||
|
|
template <typename Input>
|
||
|
|
std::shared_ptr<std::vector<Input>> clustered(Input start, std::size_t n)
|
||
|
|
{
|
||
|
|
auto pts = std::make_shared<std::vector<Input>>(n);
|
||
|
|
for (std::size_t i = 0; i < n; ++i)
|
||
|
|
(*pts)[i] = static_cast<Input>(start + static_cast<Input>(i));
|
||
|
|
return pts;
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Input>
|
||
|
|
std::shared_ptr<std::vector<Input>> strided(Input start, Input step, std::size_t n)
|
||
|
|
{
|
||
|
|
auto pts = std::make_shared<std::vector<Input>>(n);
|
||
|
|
for (std::size_t i = 0; i < n; ++i)
|
||
|
|
(*pts)[i] = static_cast<Input>(start + step * static_cast<Input>(i));
|
||
|
|
return pts;
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys>
|
||
|
|
sample hash_roots(const Keys & keys)
|
||
|
|
{
|
||
|
|
std::uint64_t w0 = 0;
|
||
|
|
std::uint64_t w1 = 0;
|
||
|
|
const auto r0 = std::get<0>(keys).root();
|
||
|
|
const auto r1 = std::get<1>(keys).root();
|
||
|
|
const std::size_t n0 = sizeof(r0) < sizeof(w0) ? sizeof(r0) : sizeof(w0);
|
||
|
|
const std::size_t n1 = sizeof(r1) < sizeof(w1) ? sizeof(r1) : sizeof(w1);
|
||
|
|
std::memcpy(&w0, &r0, n0);
|
||
|
|
std::memcpy(&w1, &r1, n1);
|
||
|
|
return touch_word(w0 ^ w1, sizeof(r0) + sizeof(r1));
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys, typename Input>
|
||
|
|
sample interval_packed(const Keys & keys, Input from, Input to, unsigned lane_bits)
|
||
|
|
{
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_interval(std::get<0>(keys), from, to);
|
||
|
|
auto b = dpf::eval_interval(std::get<1>(keys), from, to);
|
||
|
|
auto fold = [lane_bits](const auto & buf) {
|
||
|
|
std::uint64_t w = 0;
|
||
|
|
const std::size_t bytes = (buf.size() * lane_bits + 7u) / 8u;
|
||
|
|
if (bytes != 0 && buf.data() != nullptr)
|
||
|
|
{
|
||
|
|
const std::size_t n = bytes < sizeof(w) ? bytes : sizeof(w);
|
||
|
|
std::memcpy(&w, buf.data(), n);
|
||
|
|
}
|
||
|
|
return touch_word(w, bytes);
|
||
|
|
};
|
||
|
|
return fold(a.first) + fold(b.first);
|
||
|
|
});
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys, typename Input>
|
||
|
|
sample interval_both(const Keys & keys, Input from, Input to)
|
||
|
|
{
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_interval(std::get<0>(keys), from, to);
|
||
|
|
auto b = dpf::eval_interval(std::get<1>(keys), from, to);
|
||
|
|
return touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
});
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys, typename Input>
|
||
|
|
sample interval_reuse(const Keys & keys, Input from, Input to,
|
||
|
|
decltype(dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to)) & buf0,
|
||
|
|
decltype(dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to)) & buf1)
|
||
|
|
{
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0);
|
||
|
|
auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1);
|
||
|
|
(void)i0;
|
||
|
|
(void)i1;
|
||
|
|
return touch_buf(buf0) + touch_buf(buf1);
|
||
|
|
});
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys, typename Input>
|
||
|
|
sample interval_prove(const Keys & keys, Input from, Input to)
|
||
|
|
{
|
||
|
|
auto buf0 = dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to);
|
||
|
|
auto buf1 = dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to);
|
||
|
|
dpf::proof_token p0{};
|
||
|
|
dpf::proof_token p1{};
|
||
|
|
auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0, dpf::prove(p0));
|
||
|
|
auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1, dpf::prove(p1));
|
||
|
|
(void)i0;
|
||
|
|
(void)i1;
|
||
|
|
std::uint64_t extra = 0;
|
||
|
|
std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0));
|
||
|
|
return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0);
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys, typename Pts>
|
||
|
|
sample sequence_both(const Keys & keys, const Pts & pts, bool output_only)
|
||
|
|
{
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
sample s;
|
||
|
|
if (output_only)
|
||
|
|
{
|
||
|
|
auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(),
|
||
|
|
dpf::return_output_only_tag_{});
|
||
|
|
auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(),
|
||
|
|
dpf::return_output_only_tag_{});
|
||
|
|
s = touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
}
|
||
|
|
else
|
||
|
|
{
|
||
|
|
auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end());
|
||
|
|
auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end());
|
||
|
|
s = touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
}
|
||
|
|
return s;
|
||
|
|
});
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys, typename Pts>
|
||
|
|
sample sequence_breadth(const Keys & keys, const Pts & pts)
|
||
|
|
{
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_sequence_breadth_first(std::get<0>(keys), pts.begin(), pts.end());
|
||
|
|
auto b = dpf::eval_sequence_breadth_first(std::get<1>(keys), pts.begin(), pts.end());
|
||
|
|
return touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
});
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys, typename Recipe0, typename Recipe1>
|
||
|
|
sample sequence_recipe(const Keys & keys, const Recipe0 & r0, const Recipe1 & r1)
|
||
|
|
{
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_sequence(std::get<0>(keys), r0);
|
||
|
|
auto b = dpf::eval_sequence(std::get<1>(keys), r1);
|
||
|
|
return touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
});
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Keys, typename Pts>
|
||
|
|
sample sequence_prove(const Keys & keys, const Pts & pts)
|
||
|
|
{
|
||
|
|
auto buf0 = dpf::make_output_buffer_for_subsequence(std::get<0>(keys),
|
||
|
|
pts.begin(), pts.end(), dpf::return_output_only_tag_{});
|
||
|
|
auto buf1 = dpf::make_output_buffer_for_subsequence(std::get<1>(keys),
|
||
|
|
pts.begin(), pts.end(), dpf::return_output_only_tag_{});
|
||
|
|
dpf::proof_token p0{};
|
||
|
|
dpf::proof_token p1{};
|
||
|
|
auto i0 = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(), buf0,
|
||
|
|
dpf::prove(p0), dpf::return_output_only_tag_{});
|
||
|
|
auto i1 = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(), buf1,
|
||
|
|
dpf::prove(p1), dpf::return_output_only_tag_{});
|
||
|
|
(void)i0;
|
||
|
|
(void)i1;
|
||
|
|
std::uint64_t extra = 0;
|
||
|
|
std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0));
|
||
|
|
return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0);
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Input, typename... Extra>
|
||
|
|
auto make_keys(Input alpha, Extra && ...extra)
|
||
|
|
{
|
||
|
|
auto keys = dpf::make_dpf(alpha, std::uint64_t{0x9e3779b97f4a7c15ull},
|
||
|
|
std::forward<Extra>(extra)...);
|
||
|
|
using keys_t = std::decay_t<decltype(keys)>;
|
||
|
|
return std::make_shared<keys_t>(std::move(keys));
|
||
|
|
}
|
||
|
|
|
||
|
|
template <typename Input, typename Output, typename... Extra>
|
||
|
|
auto make_out_keys(Input alpha, Output beta, Extra && ...extra)
|
||
|
|
{
|
||
|
|
auto keys = dpf::make_dpf(alpha, beta, std::forward<Extra>(extra)...);
|
||
|
|
using keys_t = std::decay_t<decltype(keys)>;
|
||
|
|
return std::make_shared<keys_t>(std::move(keys));
|
||
|
|
}
|
||
|
|
|
||
|
|
const char kUsage[] =
|
||
|
|
"profile_eval [--list] [--family F] [--slice S] [--case NAME] [--tier std]\n"
|
||
|
|
" [--repeat N=20] [--warmup W=2]\n"
|
||
|
|
"Slices: keygen, interval, interval-shape, sequence, sequence-algo,\n"
|
||
|
|
" memoizer, bits, gf, inner-product, helpers, defer.\n"
|
||
|
|
"Times sequence and interval evaluation on both party keys.\n";
|
||
|
|
|
||
|
|
template <typename Input, typename KeysPtr>
|
||
|
|
void add_interval_at(std::vector<work> & works, const char * slice,
|
||
|
|
const std::string & name, Input from, Input to, const KeysPtr & keys)
|
||
|
|
{
|
||
|
|
const auto items = static_cast<std::uint64_t>(to) - static_cast<std::uint64_t>(from) + 1;
|
||
|
|
works.push_back(profile::make_work("eval", slice, name, items, [keys, from, to] {
|
||
|
|
return interval_both(*keys, from, to);
|
||
|
|
}));
|
||
|
|
}
|
||
|
|
|
||
|
|
} // namespace
|
||
|
|
|
||
|
|
int main(int argc, char ** argv)
|
||
|
|
{
|
||
|
|
const auto opt = profile::parse_args(argc, argv, 20, 2, kUsage);
|
||
|
|
|
||
|
|
using u8 = std::uint8_t;
|
||
|
|
using u16 = std::uint16_t;
|
||
|
|
using u32 = std::uint32_t;
|
||
|
|
|
||
|
|
const auto k8 = make_keys(u8{40});
|
||
|
|
const auto k16 = make_keys(u16{1512});
|
||
|
|
const auto k16v = make_keys(u16{1512}, dpf::verifiable{});
|
||
|
|
const auto k32 = make_keys(u32{1002048u});
|
||
|
|
const auto k32v = make_keys(u32{1002048u}, dpf::verifiable{});
|
||
|
|
|
||
|
|
std::vector<work> works;
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "keygen", "keygen_u8", 1, [] {
|
||
|
|
return hash_roots(*make_keys(u8{40}));
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "keygen", "keygen_u16", 1, [] {
|
||
|
|
return hash_roots(*make_keys(u16{1512}));
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "keygen", "keygen_u32", 1, [] {
|
||
|
|
return hash_roots(*make_keys(u32{0x01000000u}));
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "keygen", "keygen_u32_verifiable", 1, [] {
|
||
|
|
return hash_roots(*make_keys(u32{0x01000000u}, dpf::verifiable{}));
|
||
|
|
}));
|
||
|
|
|
||
|
|
const auto add_gf = [&](const char * name, auto beta) {
|
||
|
|
using out = std::decay_t<decltype(beta)>;
|
||
|
|
works.push_back(profile::make_work("eval", "gf",
|
||
|
|
std::string("keygen_") + name, 1, [beta] {
|
||
|
|
return hash_roots(*make_out_keys(u8{40}, beta));
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "gf",
|
||
|
|
std::string("keygen_") + name + "_verifiable", 1, [beta] {
|
||
|
|
return hash_roots(*make_out_keys(u8{40}, beta, dpf::verifiable{}));
|
||
|
|
}));
|
||
|
|
const auto keys = make_out_keys(u8{40}, beta);
|
||
|
|
if constexpr (dpf::utils::is_packed_subbyte_v<out>)
|
||
|
|
{
|
||
|
|
constexpr unsigned bits = dpf::utils::packed_lane_bits_v<out>;
|
||
|
|
works.push_back(profile::make_work("eval", "gf",
|
||
|
|
std::string("interval_") + name + "_u8_L256", 256,
|
||
|
|
[keys, bits] {
|
||
|
|
return interval_packed(*keys, u8{0}, u8{255}, bits);
|
||
|
|
}));
|
||
|
|
}
|
||
|
|
else
|
||
|
|
{
|
||
|
|
add_interval_at(works, "gf", std::string("interval_") + name + "_u8_L256",
|
||
|
|
u8{0}, u8{255}, keys);
|
||
|
|
}
|
||
|
|
};
|
||
|
|
add_gf("gf2", dpf::gf2{1});
|
||
|
|
add_gf("gf22", dpf::gf22{3});
|
||
|
|
add_gf("gf24", dpf::gf24{0xa});
|
||
|
|
add_gf("gf28", dpf::gf28{0x1b});
|
||
|
|
add_gf("gf216", dpf::gf216{0x2d});
|
||
|
|
add_gf("gf232", dpf::gf232{0x90200001u});
|
||
|
|
add_gf("gf264", dpf::gf264{0x11});
|
||
|
|
|
||
|
|
const auto add_lengths = [&](auto from0, auto keys, const char * width,
|
||
|
|
std::initializer_list<std::uint64_t> lengths) {
|
||
|
|
using input = decltype(from0);
|
||
|
|
for (const std::uint64_t n : lengths)
|
||
|
|
{
|
||
|
|
const input from = from0;
|
||
|
|
const input to = static_cast<input>(from + static_cast<input>(n - 1));
|
||
|
|
add_interval_at(works, "interval",
|
||
|
|
std::string("interval_") + width + "_L" + std::to_string(n),
|
||
|
|
from, to, keys);
|
||
|
|
}
|
||
|
|
};
|
||
|
|
add_lengths(u8{0}, k8, "u8", {1, 16, 64, 256});
|
||
|
|
add_lengths(u16{1000}, k16, "u16", {1, 16, 256, 1024, 4096});
|
||
|
|
add_lengths(u32{1000000}, k32, "u32", {1, 16, 64, 256, 1024, 4096, 16384});
|
||
|
|
|
||
|
|
add_interval_at(works, "interval-shape", "interval_u32_L256_unaligned",
|
||
|
|
u32{1000003}, u32{1000258}, k32);
|
||
|
|
add_interval_at(works, "interval-shape", "interval_u32_L4096_unaligned",
|
||
|
|
u32{1000003}, u32{1004098}, k32);
|
||
|
|
|
||
|
|
const auto add_reuse = [&](u32 from, u32 to, const char * name) {
|
||
|
|
using buf0_t = std::decay_t<decltype(dpf::make_output_buffer_for_interval(
|
||
|
|
std::get<0>(*k32), from, to))>;
|
||
|
|
using buf1_t = std::decay_t<decltype(dpf::make_output_buffer_for_interval(
|
||
|
|
std::get<1>(*k32), from, to))>;
|
||
|
|
auto buf0 = std::make_shared<buf0_t>(
|
||
|
|
dpf::make_output_buffer_for_interval(std::get<0>(*k32), from, to));
|
||
|
|
auto buf1 = std::make_shared<buf1_t>(
|
||
|
|
dpf::make_output_buffer_for_interval(std::get<1>(*k32), from, to));
|
||
|
|
const auto items = static_cast<std::uint64_t>(to - from + 1);
|
||
|
|
works.push_back(profile::make_work("eval", "interval-shape", name, items,
|
||
|
|
[k32, buf0, buf1, from, to] {
|
||
|
|
return interval_reuse(*k32, from, to, *buf0, *buf1);
|
||
|
|
}));
|
||
|
|
};
|
||
|
|
add_reuse(1000000, 1000255, "interval_u32_L256_reuse");
|
||
|
|
add_reuse(1000000, 1004095, "interval_u32_L4096_reuse");
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "interval-shape",
|
||
|
|
"interval_u16_L1024_verifiable", 1024, [k16v] {
|
||
|
|
return interval_prove(*k16v, u16{1000}, u16{2023});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "interval-shape",
|
||
|
|
"interval_u32_L256_verifiable", 256, [k32v] {
|
||
|
|
return interval_prove(*k32v, u32{1000000}, u32{1000255});
|
||
|
|
}));
|
||
|
|
|
||
|
|
const auto add_seq = [&](auto keys, const char * width, const char * shape,
|
||
|
|
auto pts) {
|
||
|
|
const auto n = static_cast<std::uint64_t>(pts->size());
|
||
|
|
const std::string name = std::string("sequence_") + width + "_" + shape
|
||
|
|
+ "_" + std::to_string(n);
|
||
|
|
works.push_back(profile::make_work("eval", "sequence", name, n,
|
||
|
|
[keys, pts] {
|
||
|
|
return sequence_both(*keys, *pts, false);
|
||
|
|
}));
|
||
|
|
};
|
||
|
|
const std::size_t seq32[] = {8, 32, 64, 128, 256, 512, 1024, 2048};
|
||
|
|
for (const std::size_t n : seq32)
|
||
|
|
{
|
||
|
|
add_seq(k32, "u32", "cluster", clustered<u32>(1u << 20, n));
|
||
|
|
const u32 step = n <= 64 ? u32{1u << 20} : u32{1u << 12};
|
||
|
|
add_seq(k32, "u32", "stride", strided<u32>(16u, step, n));
|
||
|
|
}
|
||
|
|
const std::size_t seq16[] = {8, 64, 256, 1024};
|
||
|
|
for (const std::size_t n : seq16)
|
||
|
|
add_seq(k16, "u16", "cluster", clustered<u16>(1000, n));
|
||
|
|
const std::size_t seq8[] = {8, 32, 64};
|
||
|
|
for (const std::size_t n : seq8)
|
||
|
|
{
|
||
|
|
add_seq(k8, "u8", "cluster", clustered<u8>(0, n));
|
||
|
|
add_seq(k8, "u8", "stride", strided<u8>(0, 3, n));
|
||
|
|
}
|
||
|
|
|
||
|
|
const auto c256 = clustered<u32>(1u << 20, 256);
|
||
|
|
const auto s64 = strided<u32>(16u, 1u << 20, 64);
|
||
|
|
const auto c16s = clustered<u16>(1000, 64);
|
||
|
|
const auto c32s = clustered<u32>(1u << 20, 64);
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "sequence-algo",
|
||
|
|
"sequence_u32_cluster_256_output_only", 256, [k32, c256] {
|
||
|
|
return sequence_both(*k32, *c256, true);
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "sequence-algo",
|
||
|
|
"sequence_u32_cluster_256_breadth", 256, [k32, c256] {
|
||
|
|
return sequence_breadth(*k32, *c256);
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "sequence-algo",
|
||
|
|
"sequence_u32_stride_64_breadth", 64, [k32, s64] {
|
||
|
|
return sequence_breadth(*k32, *s64);
|
||
|
|
}));
|
||
|
|
|
||
|
|
using recipe0_t = std::decay_t<decltype(dpf::make_sequence_recipe(
|
||
|
|
std::get<0>(*k32), c256->begin(), c256->end()))>;
|
||
|
|
using recipe1_t = std::decay_t<decltype(dpf::make_sequence_recipe(
|
||
|
|
std::get<1>(*k32), c256->begin(), c256->end()))>;
|
||
|
|
auto recipe0 = std::make_shared<recipe0_t>(dpf::make_sequence_recipe(
|
||
|
|
std::get<0>(*k32), c256->begin(), c256->end()));
|
||
|
|
auto recipe1 = std::make_shared<recipe1_t>(dpf::make_sequence_recipe(
|
||
|
|
std::get<1>(*k32), c256->begin(), c256->end()));
|
||
|
|
works.push_back(profile::make_work("eval", "sequence-algo",
|
||
|
|
"sequence_u32_cluster_256_recipe", 256, [k32, recipe0, recipe1] {
|
||
|
|
return sequence_recipe(*k32, *recipe0, *recipe1);
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "sequence-algo",
|
||
|
|
"sequence_u16_cluster_64_verifiable", 64, [k16v, c16s] {
|
||
|
|
return sequence_prove(*k16v, *c16s);
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "sequence-algo",
|
||
|
|
"sequence_u32_cluster_64_verifiable", 64, [k32v, c32s] {
|
||
|
|
return sequence_prove(*k32v, *c32s);
|
||
|
|
}));
|
||
|
|
|
||
|
|
// --- memoizer: built-in reuse vs hand-rolled pointwise / fresh memo ---
|
||
|
|
{
|
||
|
|
using u32 = std::uint32_t;
|
||
|
|
const u32 from = 1000000u;
|
||
|
|
const u32 to = 1000255u; // L256
|
||
|
|
const auto items = static_cast<std::uint64_t>(to - from + 1);
|
||
|
|
const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32));
|
||
|
|
using dpf_t = std::decay_t<decltype(std::get<0>(*k32))>;
|
||
|
|
using node_t = typename dpf_t::interior_node;
|
||
|
|
const auto out_elem = sizeof(std::uint64_t);
|
||
|
|
const auto logical = items * out_elem * 2;
|
||
|
|
// Match `basic_interval_memoizer` capacity (leaf nodes, not output slots).
|
||
|
|
const auto leaf_nodes =
|
||
|
|
dpf::utils::get_leafnodes_in_output_interval<dpf_t>(from, to);
|
||
|
|
const auto slots = leaf_nodes == 0 ? std::size_t{1} : leaf_nodes;
|
||
|
|
const auto pivot = std::max((slots >> 1) + (slots & 1) - 1,
|
||
|
|
(slots + 6) >> 2);
|
||
|
|
const auto memo_nodes = pivot + ((slots + 2) >> 1);
|
||
|
|
const auto memo_bytes = 2ull * memo_nodes * sizeof(node_t);
|
||
|
|
|
||
|
|
auto memo0 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to))>>(
|
||
|
|
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to));
|
||
|
|
auto memo1 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to))>>(
|
||
|
|
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to));
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "memoizer",
|
||
|
|
"memo_interval_reuse", items, [k32, memo0, memo1, from, to, prep, memo_bytes, logical] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0);
|
||
|
|
auto b = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1);
|
||
|
|
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
// Warm reuse: second pass on the same memoizers.
|
||
|
|
auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0);
|
||
|
|
auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1);
|
||
|
|
s = s + touch_buf(a2.first) + touch_buf(b2.first);
|
||
|
|
return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "memoizer",
|
||
|
|
"memo_interval_fresh", items, [k32, from, to, prep, logical, memo_bytes] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto m0 = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to);
|
||
|
|
auto m1 = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to);
|
||
|
|
auto a = dpf::eval_interval(std::get<0>(*k32), from, to, m0);
|
||
|
|
auto b = dpf::eval_interval(std::get<1>(*k32), from, to, m1);
|
||
|
|
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
auto m0b = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to);
|
||
|
|
auto m1b = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to);
|
||
|
|
auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, m0b);
|
||
|
|
auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, m1b);
|
||
|
|
s = s + touch_buf(a2.first) + touch_buf(b2.first);
|
||
|
|
return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
|
||
|
|
const auto pts = clustered<u32>(1u << 20, 64);
|
||
|
|
auto path0 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_basic_path_memoizer(std::get<0>(*k32)))>>(
|
||
|
|
dpf::make_basic_path_memoizer(std::get<0>(*k32)));
|
||
|
|
auto path1 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_basic_path_memoizer(std::get<1>(*k32)))>>(
|
||
|
|
dpf::make_basic_path_memoizer(std::get<1>(*k32)));
|
||
|
|
works.push_back(profile::make_work("eval", "memoizer",
|
||
|
|
"memo_path_reuse", 64, [k32, pts, path0, path1, prep] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
std::uint64_t h = 0;
|
||
|
|
for (auto x : *pts)
|
||
|
|
{
|
||
|
|
h ^= static_cast<std::uint64_t>(
|
||
|
|
(*dpf::eval_point(std::get<0>(*k32), x, *path0)).raw());
|
||
|
|
h ^= static_cast<std::uint64_t>(
|
||
|
|
(*dpf::eval_point(std::get<1>(*k32), x, *path1)).raw());
|
||
|
|
}
|
||
|
|
auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2);
|
||
|
|
return profile::with_costs(s, prep,
|
||
|
|
2 * sizeof(node_t) * 32, 64 * sizeof(std::uint64_t) * 2);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "memoizer",
|
||
|
|
"memo_path_pointwise", 64, [k32, pts, prep] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
std::uint64_t h = 0;
|
||
|
|
for (auto x : *pts)
|
||
|
|
{
|
||
|
|
h ^= static_cast<std::uint64_t>(
|
||
|
|
(*dpf::eval_point(std::get<0>(*k32), x)).raw());
|
||
|
|
h ^= static_cast<std::uint64_t>(
|
||
|
|
(*dpf::eval_point(std::get<1>(*k32), x)).raw());
|
||
|
|
}
|
||
|
|
auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2);
|
||
|
|
return profile::with_costs(s, prep, 0,
|
||
|
|
64 * sizeof(std::uint64_t) * 2);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
|
||
|
|
using recipe0_t = std::decay_t<decltype(dpf::make_sequence_recipe(
|
||
|
|
std::get<0>(*k32), pts->begin(), pts->end()))>;
|
||
|
|
using recipe1_t = std::decay_t<decltype(dpf::make_sequence_recipe(
|
||
|
|
std::get<1>(*k32), pts->begin(), pts->end()))>;
|
||
|
|
auto rec0 = std::make_shared<recipe0_t>(dpf::make_sequence_recipe(
|
||
|
|
std::get<0>(*k32), pts->begin(), pts->end()));
|
||
|
|
auto rec1 = std::make_shared<recipe1_t>(dpf::make_sequence_recipe(
|
||
|
|
std::get<1>(*k32), pts->begin(), pts->end()));
|
||
|
|
auto smemo0 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0))>>(
|
||
|
|
dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0));
|
||
|
|
auto smemo1 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1))>>(
|
||
|
|
dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1));
|
||
|
|
works.push_back(profile::make_work("eval", "memoizer",
|
||
|
|
"memo_recipe_reuse", 64, [k32, rec0, rec1, smemo0, smemo1, prep] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0);
|
||
|
|
auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1);
|
||
|
|
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0);
|
||
|
|
auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1);
|
||
|
|
s = s + touch_buf(a2.first) + touch_buf(b2.first);
|
||
|
|
return profile::with_costs(s, prep, s.out_bytes,
|
||
|
|
64 * sizeof(std::uint64_t) * 4);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "memoizer",
|
||
|
|
"memo_recipe_fresh", 64, [k32, rec0, rec1, prep] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto m0 = dpf::make_inplace_reversing_sequence_memoizer(
|
||
|
|
std::get<0>(*k32), *rec0);
|
||
|
|
auto m1 = dpf::make_inplace_reversing_sequence_memoizer(
|
||
|
|
std::get<1>(*k32), *rec1);
|
||
|
|
auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0);
|
||
|
|
auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1);
|
||
|
|
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
auto m0b = dpf::make_inplace_reversing_sequence_memoizer(
|
||
|
|
std::get<0>(*k32), *rec0);
|
||
|
|
auto m1b = dpf::make_inplace_reversing_sequence_memoizer(
|
||
|
|
std::get<1>(*k32), *rec1);
|
||
|
|
auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0b);
|
||
|
|
auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1b);
|
||
|
|
s = s + touch_buf(a2.first) + touch_buf(b2.first);
|
||
|
|
return profile::with_costs(s, prep, s.out_bytes,
|
||
|
|
64 * sizeof(std::uint64_t) * 4);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
}
|
||
|
|
|
||
|
|
// --- bits: built-in iterators vs hand-rolled scans (u8 domain) ---
|
||
|
|
{
|
||
|
|
using input_type = std::uint8_t;
|
||
|
|
using output_type = dpf::bit;
|
||
|
|
const input_type alpha = 40;
|
||
|
|
auto bit_keys = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_dpf(alpha, output_type::one))>>(
|
||
|
|
dpf::make_dpf(alpha, output_type::one));
|
||
|
|
using dpf_type = std::decay_t<decltype(std::get<0>(*bit_keys))>;
|
||
|
|
auto memo0 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_basic_full_memoizer<dpf_type>())>>(
|
||
|
|
dpf::make_basic_full_memoizer<dpf_type>());
|
||
|
|
auto memo1 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_basic_full_memoizer<dpf_type>())>>(
|
||
|
|
dpf::make_basic_full_memoizer<dpf_type>());
|
||
|
|
auto full0 = dpf::eval_full(std::get<0>(*bit_keys), *memo0);
|
||
|
|
auto full1 = dpf::eval_full(std::get<1>(*bit_keys), *memo1);
|
||
|
|
auto buf0 = std::make_shared<std::decay_t<decltype(full0.first)>>(std::move(full0.first));
|
||
|
|
auto buf1 = std::make_shared<std::decay_t<decltype(full1.first)>>(std::move(full1.first));
|
||
|
|
auto iter0 = std::make_shared<std::decay_t<decltype(full0.second)>>(std::move(full0.second));
|
||
|
|
auto iter1 = std::make_shared<std::decay_t<decltype(full1.second)>>(std::move(full1.second));
|
||
|
|
const auto leaf_nodes = static_cast<std::uint64_t>(1) << dpf_type::depth;
|
||
|
|
const auto bit_items = static_cast<std::uint64_t>(buf0->size());
|
||
|
|
const auto prep = sizeof(std::get<0>(*bit_keys)) + sizeof(std::get<1>(*bit_keys));
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "bits",
|
||
|
|
"bits_advice_builtin", leaf_nodes, [memo0, memo1, prep, leaf_nodes] {
|
||
|
|
std::uint64_t h = 0;
|
||
|
|
std::uint64_t n = 0;
|
||
|
|
for (auto b : dpf::advice_bits_of(*memo0))
|
||
|
|
{
|
||
|
|
h = (h << 1) ^ (b ? 1u : 0u);
|
||
|
|
++n;
|
||
|
|
}
|
||
|
|
for (auto b : dpf::advice_bits_of(*memo1))
|
||
|
|
h ^= (b ? 1u : 0u);
|
||
|
|
auto s = touch_word(h ^ n, n);
|
||
|
|
return profile::with_costs(s, prep, 0, leaf_nodes);
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "bits",
|
||
|
|
"bits_advice_handroll", leaf_nodes, [memo0, memo1, prep, leaf_nodes] {
|
||
|
|
std::uint64_t h = 0;
|
||
|
|
std::uint64_t n = 0;
|
||
|
|
const auto * p0 = memo0->begin();
|
||
|
|
const auto * e0 = memo0->end();
|
||
|
|
for (; p0 != e0; ++p0)
|
||
|
|
{
|
||
|
|
const auto * bytes = reinterpret_cast<const char *>(p0);
|
||
|
|
h = (h << 1) ^ (bytes[0] & 1u);
|
||
|
|
++n;
|
||
|
|
}
|
||
|
|
const auto * p1 = memo1->begin();
|
||
|
|
const auto * e1 = memo1->end();
|
||
|
|
for (; p1 != e1; ++p1)
|
||
|
|
{
|
||
|
|
const auto * bytes = reinterpret_cast<const char *>(p1);
|
||
|
|
h ^= (bytes[0] & 1u);
|
||
|
|
}
|
||
|
|
auto s = touch_word(h ^ n, n);
|
||
|
|
return profile::with_costs(s, prep, 0, leaf_nodes);
|
||
|
|
}));
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "bits",
|
||
|
|
"bits_setbit_builtin", bit_items, [iter0, iter1, prep] {
|
||
|
|
std::uint64_t h = 0;
|
||
|
|
std::uint64_t n = 0;
|
||
|
|
for (auto i : dpf::indices_set_in(*iter0))
|
||
|
|
{
|
||
|
|
h ^= static_cast<std::uint64_t>(i) + 0x9e3779b97f4a7c15ull;
|
||
|
|
++n;
|
||
|
|
}
|
||
|
|
for (auto i : dpf::indices_set_in(*iter1))
|
||
|
|
h ^= static_cast<std::uint64_t>(i);
|
||
|
|
auto s = touch_word(h ^ n, n * sizeof(std::size_t));
|
||
|
|
return profile::with_costs(s, prep, 0, n * sizeof(std::size_t));
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "bits",
|
||
|
|
"bits_setbit_handroll", bit_items, [buf0, buf1, prep] {
|
||
|
|
std::uint64_t h = 0;
|
||
|
|
std::uint64_t n = 0;
|
||
|
|
const auto scan = [&](const auto & buf) {
|
||
|
|
for (std::size_t i = 0; i < buf.size(); ++i)
|
||
|
|
{
|
||
|
|
if (static_cast<bool>(buf[i]))
|
||
|
|
{
|
||
|
|
h ^= i + 0x9e3779b97f4a7c15ull;
|
||
|
|
++n;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
};
|
||
|
|
scan(*buf0);
|
||
|
|
scan(*buf1);
|
||
|
|
auto s = touch_word(h ^ n, n * sizeof(std::size_t));
|
||
|
|
return profile::with_costs(s, prep, 0, n * sizeof(std::size_t));
|
||
|
|
}));
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "bits",
|
||
|
|
"bits_parallel_builtin", bit_items, [buf0, buf1, prep, bit_items] {
|
||
|
|
std::uint64_t h = 0;
|
||
|
|
std::uint64_t n = 0;
|
||
|
|
for (auto word : dpf::batch_of(*buf0, *buf1))
|
||
|
|
{
|
||
|
|
h ^= static_cast<std::uint64_t>(word[0])
|
||
|
|
^ static_cast<std::uint64_t>(word[1]);
|
||
|
|
++n;
|
||
|
|
}
|
||
|
|
auto s = touch_word(h ^ n, n * sizeof(std::uint64_t));
|
||
|
|
return profile::with_costs(s, prep, 0, bit_items);
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "bits",
|
||
|
|
"bits_parallel_handroll", bit_items, [buf0, buf1, prep, bit_items] {
|
||
|
|
std::uint64_t h = 0;
|
||
|
|
const std::size_t n = std::min(buf0->size(), buf1->size());
|
||
|
|
for (std::size_t i = 0; i < n; ++i)
|
||
|
|
h ^= (static_cast<bool>((*buf0)[i]) ? 1ull : 0ull)
|
||
|
|
^ (static_cast<bool>((*buf1)[i]) ? 2ull : 0ull);
|
||
|
|
auto s = touch_word(h ^ n, n);
|
||
|
|
return profile::with_costs(s, prep, 0, bit_items);
|
||
|
|
}));
|
||
|
|
}
|
||
|
|
|
||
|
|
// --- inner-product: built-in vs hand-rolled interval + dot ---
|
||
|
|
{
|
||
|
|
using u32 = std::uint32_t;
|
||
|
|
const u32 from = 1000000u;
|
||
|
|
const u32 to = 1000255u;
|
||
|
|
const auto items = static_cast<std::uint64_t>(to - from + 1);
|
||
|
|
auto weights = std::make_shared<std::vector<std::uint64_t>>(items);
|
||
|
|
for (std::size_t i = 0; i < items; ++i)
|
||
|
|
(*weights)[i] = 0x9e3779b97f4a7c15ull * (i + 1);
|
||
|
|
const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32));
|
||
|
|
auto ip_memo0 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to))>>(
|
||
|
|
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to));
|
||
|
|
auto ip_memo1 = std::make_shared<std::decay_t<decltype(
|
||
|
|
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to))>>(
|
||
|
|
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to));
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "inner-product",
|
||
|
|
"ip_interval_builtin", items,
|
||
|
|
[k32, weights, from, to, prep, items, ip_memo0, ip_memo1] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_inner_product(std::get<0>(*k32), from, to,
|
||
|
|
*weights, *ip_memo0);
|
||
|
|
auto b = dpf::eval_inner_product(std::get<1>(*k32), from, to,
|
||
|
|
*weights, *ip_memo1);
|
||
|
|
std::uint64_t ha = 0;
|
||
|
|
std::uint64_t hb = 0;
|
||
|
|
std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a));
|
||
|
|
std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b));
|
||
|
|
auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b));
|
||
|
|
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "inner-product",
|
||
|
|
"ip_interval_handroll", items, [k32, weights, from, to, prep, items] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_interval(std::get<0>(*k32), from, to);
|
||
|
|
auto b = dpf::eval_interval(std::get<1>(*k32), from, to);
|
||
|
|
std::uint64_t dot0 = 0;
|
||
|
|
std::uint64_t dot1 = 0;
|
||
|
|
const std::size_t n = std::min(items, a.first.size());
|
||
|
|
for (std::size_t i = 0; i < n; ++i)
|
||
|
|
{
|
||
|
|
dot0 += static_cast<std::uint64_t>(a.first[i].raw()) * (*weights)[i];
|
||
|
|
dot1 += static_cast<std::uint64_t>(b.first[i].raw()) * (*weights)[i];
|
||
|
|
}
|
||
|
|
auto s = touch_word(dot0 ^ dot1, a.first.size() * sizeof(a.first[0])
|
||
|
|
+ b.first.size() * sizeof(b.first[0]));
|
||
|
|
return profile::with_costs(s, prep, s.out_bytes,
|
||
|
|
items * sizeof(std::uint64_t) * 2);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "inner-product",
|
||
|
|
"ip_interval_paired", items,
|
||
|
|
[k32, weights, from, to, prep, items] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_inner_product(dpf::paired,
|
||
|
|
std::get<0>(*k32), from, to, *weights);
|
||
|
|
auto b = dpf::eval_inner_product(dpf::paired,
|
||
|
|
std::get<1>(*k32), from, to, *weights);
|
||
|
|
std::uint64_t ha = 0;
|
||
|
|
std::uint64_t hb = 0;
|
||
|
|
std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a));
|
||
|
|
std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b));
|
||
|
|
auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b));
|
||
|
|
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "inner-product",
|
||
|
|
"ip_interval_columns", items,
|
||
|
|
[k32, weights, from, to, prep, items] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_inner_product(dpf::columns,
|
||
|
|
std::get<0>(*k32), from, to, std::tie(*weights));
|
||
|
|
auto b = dpf::eval_inner_product(dpf::columns,
|
||
|
|
std::get<1>(*k32), from, to, std::tie(*weights));
|
||
|
|
std::uint64_t ha = 0;
|
||
|
|
std::uint64_t hb = 0;
|
||
|
|
const auto & a0 = std::get<0>(a);
|
||
|
|
const auto & b0 = std::get<0>(b);
|
||
|
|
std::memcpy(&ha, &a0,
|
||
|
|
sizeof(ha) < sizeof(a0) ? sizeof(ha) : sizeof(a0));
|
||
|
|
std::memcpy(&hb, &b0,
|
||
|
|
sizeof(hb) < sizeof(b0) ? sizeof(hb) : sizeof(b0));
|
||
|
|
auto s = touch_word(ha ^ hb, sizeof(a0) + sizeof(b0));
|
||
|
|
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
}
|
||
|
|
|
||
|
|
// --- helpers: point vs interval vs full (representative widths) ---
|
||
|
|
{
|
||
|
|
using u16 = std::uint16_t;
|
||
|
|
const auto prep16 = sizeof(std::get<0>(*k16)) + sizeof(std::get<1>(*k16));
|
||
|
|
works.push_back(profile::make_work("eval", "helpers",
|
||
|
|
"helper_point_u16", 1, [k16, prep16] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = *dpf::eval_point(std::get<0>(*k16), u16{1512});
|
||
|
|
auto b = *dpf::eval_point(std::get<1>(*k16), u16{1512});
|
||
|
|
auto s = touch_word(
|
||
|
|
static_cast<std::uint64_t>(a.raw()) ^ static_cast<std::uint64_t>(b.raw()),
|
||
|
|
sizeof(a) + sizeof(b));
|
||
|
|
return profile::with_costs(s, prep16, 0, sizeof(a) + sizeof(b));
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "helpers",
|
||
|
|
"helper_interval_u16_L256", 256, [k16, prep16] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_interval(std::get<0>(*k16), u16{1000}, u16{1255});
|
||
|
|
auto b = dpf::eval_interval(std::get<1>(*k16), u16{1000}, u16{1255});
|
||
|
|
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
return profile::with_costs(s, prep16, s.out_bytes,
|
||
|
|
256 * sizeof(std::uint64_t) * 2);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
works.push_back(profile::make_work("eval", "helpers",
|
||
|
|
"helper_full_u8", 256, [k8] {
|
||
|
|
const auto prep = sizeof(std::get<0>(*k8)) + sizeof(std::get<1>(*k8));
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto a = dpf::eval_full(std::get<0>(*k8));
|
||
|
|
auto b = dpf::eval_full(std::get<1>(*k8));
|
||
|
|
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
return profile::with_costs(s, prep, s.out_bytes, s.out_bytes);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
}
|
||
|
|
|
||
|
|
// --- defer: pre-assign full-domain expand vs rotate-after-assign ---
|
||
|
|
{
|
||
|
|
using u8 = std::uint8_t;
|
||
|
|
using out_t = std::uint64_t;
|
||
|
|
const u8 alpha = 40;
|
||
|
|
const out_t beta = 0x9e3779b97f4a7c15ull;
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "defer",
|
||
|
|
"defer_full_u8_expand", 256, [beta] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
|
||
|
|
auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys));
|
||
|
|
auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys));
|
||
|
|
auto d0 = dpf::defer_eval_full(std::get<0>(keys), b0);
|
||
|
|
auto d1 = dpf::defer_eval_full(std::get<1>(keys), b1);
|
||
|
|
(void)d0;
|
||
|
|
(void)d1;
|
||
|
|
return touch_buf(b0) + touch_buf(b1);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "defer",
|
||
|
|
"defer_interval_u8_L64_expand", 64, [beta] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
|
||
|
|
auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys));
|
||
|
|
auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys));
|
||
|
|
auto d0 = dpf::defer_eval_interval(std::get<0>(keys),
|
||
|
|
u8{16}, u8{79}, b0);
|
||
|
|
auto d1 = dpf::defer_eval_interval(std::get<1>(keys),
|
||
|
|
u8{16}, u8{79}, b1);
|
||
|
|
(void)d0;
|
||
|
|
(void)d1;
|
||
|
|
return touch_buf(b0) + touch_buf(b1);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
|
||
|
|
// Expand once against stable key storage, assign, then time .get().
|
||
|
|
{
|
||
|
|
using keys_t = std::decay_t<decltype(
|
||
|
|
dpf::make_dpf(dpf::wildcard_value<u8>{}, beta))>;
|
||
|
|
auto keys = std::make_shared<keys_t>(
|
||
|
|
dpf::make_dpf(dpf::wildcard_value<u8>{}, beta));
|
||
|
|
using buf0_t = std::decay_t<decltype(
|
||
|
|
dpf::make_output_buffer_for_full(std::get<0>(*keys)))>;
|
||
|
|
using buf1_t = std::decay_t<decltype(
|
||
|
|
dpf::make_output_buffer_for_full(std::get<1>(*keys)))>;
|
||
|
|
auto b0 = std::make_shared<buf0_t>(
|
||
|
|
dpf::make_output_buffer_for_full(std::get<0>(*keys)));
|
||
|
|
auto b1 = std::make_shared<buf1_t>(
|
||
|
|
dpf::make_output_buffer_for_full(std::get<1>(*keys)));
|
||
|
|
using def0_t = std::decay_t<decltype(
|
||
|
|
dpf::defer_eval_full(std::get<0>(*keys), *b0))>;
|
||
|
|
using def1_t = std::decay_t<decltype(
|
||
|
|
dpf::defer_eval_full(std::get<1>(*keys), *b1))>;
|
||
|
|
auto def0 = std::make_shared<def0_t>(
|
||
|
|
dpf::defer_eval_full(std::get<0>(*keys), *b0));
|
||
|
|
auto def1 = std::make_shared<def1_t>(
|
||
|
|
dpf::defer_eval_full(std::get<1>(*keys), *b1));
|
||
|
|
{
|
||
|
|
auto & k0 = std::get<0>(*keys);
|
||
|
|
auto & k1 = std::get<1>(*keys);
|
||
|
|
const u8 a0 = 0x12;
|
||
|
|
const u8 a1 = static_cast<u8>(alpha - a0);
|
||
|
|
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
|
||
|
|
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
|
||
|
|
k0.offset_x.reconstruct(sh1);
|
||
|
|
k1.offset_x.reconstruct(sh0);
|
||
|
|
}
|
||
|
|
works.push_back(profile::make_work("eval", "defer",
|
||
|
|
"defer_full_u8_get", 256, [keys, def0, def1, b0, b1] {
|
||
|
|
(void)keys;
|
||
|
|
auto v0 = def0->get();
|
||
|
|
auto v1 = def1->get();
|
||
|
|
std::uint64_t sink = 0;
|
||
|
|
for (auto it = std::begin(v0); it != std::end(v0); ++it)
|
||
|
|
sink ^= static_cast<std::uint64_t>((*it).raw());
|
||
|
|
for (auto it = std::begin(v1); it != std::end(v1); ++it)
|
||
|
|
sink ^= static_cast<std::uint64_t>((*it).raw());
|
||
|
|
return touch_word(sink, b0->size() * sizeof((*b0)[0])
|
||
|
|
+ b1->size() * sizeof((*b1)[0]));
|
||
|
|
}));
|
||
|
|
}
|
||
|
|
|
||
|
|
works.push_back(profile::make_work("eval", "defer",
|
||
|
|
"eager_full_u8_after_assign", 256, [beta, alpha] {
|
||
|
|
return profile::with_prg([&] {
|
||
|
|
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
|
||
|
|
auto & k0 = std::get<0>(keys);
|
||
|
|
auto & k1 = std::get<1>(keys);
|
||
|
|
const u8 a0 = 0x12;
|
||
|
|
const u8 a1 = static_cast<u8>(alpha - a0);
|
||
|
|
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
|
||
|
|
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
|
||
|
|
k0.offset_x.reconstruct(sh1);
|
||
|
|
k1.offset_x.reconstruct(sh0);
|
||
|
|
auto a = dpf::eval_full(k0);
|
||
|
|
auto b = dpf::eval_full(k1);
|
||
|
|
return touch_buf(a.first) + touch_buf(b.first);
|
||
|
|
});
|
||
|
|
}));
|
||
|
|
}
|
||
|
|
|
||
|
|
return profile::run_works(opt, works);
|
||
|
|
}
|