/// @file test/profile/eval_profile.cpp /// @brief Workload for sequence and interval evaluation. /// /// Both party keys are evaluated inside each sample. Key generation is its /// own case, not part of the eval samples. `out_bytes` is the output buffer /// volume of one sample (both parties). `per_item_ns` divides by the point /// count, not by the number of parties. /// /// profile_eval --list /// profile_eval --case interval_u32_L4096 --repeat 50 --warmup 3 /// /// Profile-guided build, separate from COVERAGE: /// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release /// cmake --build build-pgo --target profile_eval /// build-pgo/bin/profile_eval --repeat 40 --warmup 2 /// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release /// cmake --build build-pgo --target profile_eval #include "harness.hpp" #include "dpf.hpp" #include #include #include #include #include #include #include #include namespace { using profile::sample; using profile::touch_buf; using profile::touch_word; using profile::work; template std::shared_ptr> clustered(Input start, std::size_t n) { auto pts = std::make_shared>(n); for (std::size_t i = 0; i < n; ++i) (*pts)[i] = static_cast(start + static_cast(i)); return pts; } template std::shared_ptr> strided(Input start, Input step, std::size_t n) { auto pts = std::make_shared>(n); for (std::size_t i = 0; i < n; ++i) (*pts)[i] = static_cast(start + step * static_cast(i)); return pts; } template sample hash_roots(const Keys & keys) { std::uint64_t w0 = 0; std::uint64_t w1 = 0; const auto r0 = std::get<0>(keys).root(); const auto r1 = std::get<1>(keys).root(); const std::size_t n0 = sizeof(r0) < sizeof(w0) ? sizeof(r0) : sizeof(w0); const std::size_t n1 = sizeof(r1) < sizeof(w1) ? sizeof(r1) : sizeof(w1); std::memcpy(&w0, &r0, n0); std::memcpy(&w1, &r1, n1); return touch_word(w0 ^ w1, sizeof(r0) + sizeof(r1)); } template sample interval_packed(const Keys & keys, Input from, Input to, unsigned lane_bits) { return profile::with_prg([&] { auto a = dpf::eval_interval(std::get<0>(keys), from, to); auto b = dpf::eval_interval(std::get<1>(keys), from, to); auto fold = [lane_bits](const auto & buf) { std::uint64_t w = 0; const std::size_t bytes = (buf.size() * lane_bits + 7u) / 8u; if (bytes != 0 && buf.data() != nullptr) { const std::size_t n = bytes < sizeof(w) ? bytes : sizeof(w); std::memcpy(&w, buf.data(), n); } return touch_word(w, bytes); }; return fold(a.first) + fold(b.first); }); } template sample interval_both(const Keys & keys, Input from, Input to) { return profile::with_prg([&] { auto a = dpf::eval_interval(std::get<0>(keys), from, to); auto b = dpf::eval_interval(std::get<1>(keys), from, to); return touch_buf(a.first) + touch_buf(b.first); }); } template sample interval_reuse(const Keys & keys, Input from, Input to, decltype(dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to)) & buf0, decltype(dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to)) & buf1) { return profile::with_prg([&] { auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0); auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1); (void)i0; (void)i1; return touch_buf(buf0) + touch_buf(buf1); }); } template sample interval_prove(const Keys & keys, Input from, Input to) { auto buf0 = dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to); auto buf1 = dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to); dpf::proof_token p0{}; dpf::proof_token p1{}; auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0, dpf::prove(p0)); auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1, dpf::prove(p1)); (void)i0; (void)i1; std::uint64_t extra = 0; std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0)); return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0); } template sample sequence_both(const Keys & keys, const Pts & pts, bool output_only) { return profile::with_prg([&] { sample s; if (output_only) { auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(), dpf::return_output_only_tag_{}); auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(), dpf::return_output_only_tag_{}); s = touch_buf(a.first) + touch_buf(b.first); } else { auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end()); auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end()); s = touch_buf(a.first) + touch_buf(b.first); } return s; }); } template sample sequence_breadth(const Keys & keys, const Pts & pts) { return profile::with_prg([&] { auto a = dpf::eval_sequence_breadth_first(std::get<0>(keys), pts.begin(), pts.end()); auto b = dpf::eval_sequence_breadth_first(std::get<1>(keys), pts.begin(), pts.end()); return touch_buf(a.first) + touch_buf(b.first); }); } template sample sequence_recipe(const Keys & keys, const Recipe0 & r0, const Recipe1 & r1) { return profile::with_prg([&] { auto a = dpf::eval_sequence(std::get<0>(keys), r0); auto b = dpf::eval_sequence(std::get<1>(keys), r1); return touch_buf(a.first) + touch_buf(b.first); }); } template sample sequence_prove(const Keys & keys, const Pts & pts) { auto buf0 = dpf::make_output_buffer_for_subsequence(std::get<0>(keys), pts.begin(), pts.end(), dpf::return_output_only_tag_{}); auto buf1 = dpf::make_output_buffer_for_subsequence(std::get<1>(keys), pts.begin(), pts.end(), dpf::return_output_only_tag_{}); dpf::proof_token p0{}; dpf::proof_token p1{}; auto i0 = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(), buf0, dpf::prove(p0), dpf::return_output_only_tag_{}); auto i1 = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(), buf1, dpf::prove(p1), dpf::return_output_only_tag_{}); (void)i0; (void)i1; std::uint64_t extra = 0; std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0)); return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0); } template auto make_keys(Input alpha, Extra && ...extra) { auto keys = dpf::make_dpf(alpha, std::uint64_t{0x9e3779b97f4a7c15ull}, std::forward(extra)...); using keys_t = std::decay_t; return std::make_shared(std::move(keys)); } template auto make_out_keys(Input alpha, Output beta, Extra && ...extra) { auto keys = dpf::make_dpf(alpha, beta, std::forward(extra)...); using keys_t = std::decay_t; return std::make_shared(std::move(keys)); } const char kUsage[] = "profile_eval [--list] [--family F] [--slice S] [--case NAME] [--tier std]\n" " [--repeat N=20] [--warmup W=2]\n" "Slices: keygen, interval, interval-shape, sequence, sequence-algo,\n" " memoizer, bits, gf, inner-product, helpers, defer.\n" "Times sequence and interval evaluation on both party keys.\n"; template void add_interval_at(std::vector & works, const char * slice, const std::string & name, Input from, Input to, const KeysPtr & keys) { const auto items = static_cast(to) - static_cast(from) + 1; works.push_back(profile::make_work("eval", slice, name, items, [keys, from, to] { return interval_both(*keys, from, to); })); } } // namespace int main(int argc, char ** argv) { const auto opt = profile::parse_args(argc, argv, 20, 2, kUsage); using u8 = std::uint8_t; using u16 = std::uint16_t; using u32 = std::uint32_t; const auto k8 = make_keys(u8{40}); const auto k16 = make_keys(u16{1512}); const auto k16v = make_keys(u16{1512}, dpf::verifiable{}); const auto k32 = make_keys(u32{1002048u}); const auto k32v = make_keys(u32{1002048u}, dpf::verifiable{}); std::vector works; works.push_back(profile::make_work("eval", "keygen", "keygen_u8", 1, [] { return hash_roots(*make_keys(u8{40})); })); works.push_back(profile::make_work("eval", "keygen", "keygen_u16", 1, [] { return hash_roots(*make_keys(u16{1512})); })); works.push_back(profile::make_work("eval", "keygen", "keygen_u32", 1, [] { return hash_roots(*make_keys(u32{0x01000000u})); })); works.push_back(profile::make_work("eval", "keygen", "keygen_u32_verifiable", 1, [] { return hash_roots(*make_keys(u32{0x01000000u}, dpf::verifiable{})); })); const auto add_gf = [&](const char * name, auto beta) { using out = std::decay_t; works.push_back(profile::make_work("eval", "gf", std::string("keygen_") + name, 1, [beta] { return hash_roots(*make_out_keys(u8{40}, beta)); })); works.push_back(profile::make_work("eval", "gf", std::string("keygen_") + name + "_verifiable", 1, [beta] { return hash_roots(*make_out_keys(u8{40}, beta, dpf::verifiable{})); })); const auto keys = make_out_keys(u8{40}, beta); if constexpr (dpf::utils::is_packed_subbyte_v) { constexpr unsigned bits = dpf::utils::packed_lane_bits_v; works.push_back(profile::make_work("eval", "gf", std::string("interval_") + name + "_u8_L256", 256, [keys, bits] { return interval_packed(*keys, u8{0}, u8{255}, bits); })); } else { add_interval_at(works, "gf", std::string("interval_") + name + "_u8_L256", u8{0}, u8{255}, keys); } }; add_gf("gf2", dpf::gf2{1}); add_gf("gf22", dpf::gf22{3}); add_gf("gf24", dpf::gf24{0xa}); add_gf("gf28", dpf::gf28{0x1b}); add_gf("gf216", dpf::gf216{0x2d}); add_gf("gf232", dpf::gf232{0x90200001u}); add_gf("gf264", dpf::gf264{0x11}); const auto add_lengths = [&](auto from0, auto keys, const char * width, std::initializer_list lengths) { using input = decltype(from0); for (const std::uint64_t n : lengths) { const input from = from0; const input to = static_cast(from + static_cast(n - 1)); add_interval_at(works, "interval", std::string("interval_") + width + "_L" + std::to_string(n), from, to, keys); } }; add_lengths(u8{0}, k8, "u8", {1, 16, 64, 256}); add_lengths(u16{1000}, k16, "u16", {1, 16, 256, 1024, 4096}); add_lengths(u32{1000000}, k32, "u32", {1, 16, 64, 256, 1024, 4096, 16384}); add_interval_at(works, "interval-shape", "interval_u32_L256_unaligned", u32{1000003}, u32{1000258}, k32); add_interval_at(works, "interval-shape", "interval_u32_L4096_unaligned", u32{1000003}, u32{1004098}, k32); const auto add_reuse = [&](u32 from, u32 to, const char * name) { using buf0_t = std::decay_t(*k32), from, to))>; using buf1_t = std::decay_t(*k32), from, to))>; auto buf0 = std::make_shared( dpf::make_output_buffer_for_interval(std::get<0>(*k32), from, to)); auto buf1 = std::make_shared( dpf::make_output_buffer_for_interval(std::get<1>(*k32), from, to)); const auto items = static_cast(to - from + 1); works.push_back(profile::make_work("eval", "interval-shape", name, items, [k32, buf0, buf1, from, to] { return interval_reuse(*k32, from, to, *buf0, *buf1); })); }; add_reuse(1000000, 1000255, "interval_u32_L256_reuse"); add_reuse(1000000, 1004095, "interval_u32_L4096_reuse"); works.push_back(profile::make_work("eval", "interval-shape", "interval_u16_L1024_verifiable", 1024, [k16v] { return interval_prove(*k16v, u16{1000}, u16{2023}); })); works.push_back(profile::make_work("eval", "interval-shape", "interval_u32_L256_verifiable", 256, [k32v] { return interval_prove(*k32v, u32{1000000}, u32{1000255}); })); const auto add_seq = [&](auto keys, const char * width, const char * shape, auto pts) { const auto n = static_cast(pts->size()); const std::string name = std::string("sequence_") + width + "_" + shape + "_" + std::to_string(n); works.push_back(profile::make_work("eval", "sequence", name, n, [keys, pts] { return sequence_both(*keys, *pts, false); })); }; const std::size_t seq32[] = {8, 32, 64, 128, 256, 512, 1024, 2048}; for (const std::size_t n : seq32) { add_seq(k32, "u32", "cluster", clustered(1u << 20, n)); const u32 step = n <= 64 ? u32{1u << 20} : u32{1u << 12}; add_seq(k32, "u32", "stride", strided(16u, step, n)); } const std::size_t seq16[] = {8, 64, 256, 1024}; for (const std::size_t n : seq16) add_seq(k16, "u16", "cluster", clustered(1000, n)); const std::size_t seq8[] = {8, 32, 64}; for (const std::size_t n : seq8) { add_seq(k8, "u8", "cluster", clustered(0, n)); add_seq(k8, "u8", "stride", strided(0, 3, n)); } const auto c256 = clustered(1u << 20, 256); const auto s64 = strided(16u, 1u << 20, 64); const auto c16s = clustered(1000, 64); const auto c32s = clustered(1u << 20, 64); works.push_back(profile::make_work("eval", "sequence-algo", "sequence_u32_cluster_256_output_only", 256, [k32, c256] { return sequence_both(*k32, *c256, true); })); works.push_back(profile::make_work("eval", "sequence-algo", "sequence_u32_cluster_256_breadth", 256, [k32, c256] { return sequence_breadth(*k32, *c256); })); works.push_back(profile::make_work("eval", "sequence-algo", "sequence_u32_stride_64_breadth", 64, [k32, s64] { return sequence_breadth(*k32, *s64); })); using recipe0_t = std::decay_t(*k32), c256->begin(), c256->end()))>; using recipe1_t = std::decay_t(*k32), c256->begin(), c256->end()))>; auto recipe0 = std::make_shared(dpf::make_sequence_recipe( std::get<0>(*k32), c256->begin(), c256->end())); auto recipe1 = std::make_shared(dpf::make_sequence_recipe( std::get<1>(*k32), c256->begin(), c256->end())); works.push_back(profile::make_work("eval", "sequence-algo", "sequence_u32_cluster_256_recipe", 256, [k32, recipe0, recipe1] { return sequence_recipe(*k32, *recipe0, *recipe1); })); works.push_back(profile::make_work("eval", "sequence-algo", "sequence_u16_cluster_64_verifiable", 64, [k16v, c16s] { return sequence_prove(*k16v, *c16s); })); works.push_back(profile::make_work("eval", "sequence-algo", "sequence_u32_cluster_64_verifiable", 64, [k32v, c32s] { return sequence_prove(*k32v, *c32s); })); // --- memoizer: built-in reuse vs hand-rolled pointwise / fresh memo --- { using u32 = std::uint32_t; const u32 from = 1000000u; const u32 to = 1000255u; // L256 const auto items = static_cast(to - from + 1); const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32)); using dpf_t = std::decay_t(*k32))>; using node_t = typename dpf_t::interior_node; const auto out_elem = sizeof(std::uint64_t); const auto logical = items * out_elem * 2; // Match `basic_interval_memoizer` capacity (leaf nodes, not output slots). const auto leaf_nodes = dpf::utils::get_leafnodes_in_output_interval(from, to); const auto slots = leaf_nodes == 0 ? std::size_t{1} : leaf_nodes; const auto pivot = std::max((slots >> 1) + (slots & 1) - 1, (slots + 6) >> 2); const auto memo_nodes = pivot + ((slots + 2) >> 1); const auto memo_bytes = 2ull * memo_nodes * sizeof(node_t); auto memo0 = std::make_shared(*k32), from, to))>>( dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to)); auto memo1 = std::make_shared(*k32), from, to))>>( dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to)); works.push_back(profile::make_work("eval", "memoizer", "memo_interval_reuse", items, [k32, memo0, memo1, from, to, prep, memo_bytes, logical] { return profile::with_prg([&] { auto a = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0); auto b = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1); auto s = touch_buf(a.first) + touch_buf(b.first); // Warm reuse: second pass on the same memoizers. auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0); auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1); s = s + touch_buf(a2.first) + touch_buf(b2.first); return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2); }); })); works.push_back(profile::make_work("eval", "memoizer", "memo_interval_fresh", items, [k32, from, to, prep, logical, memo_bytes] { return profile::with_prg([&] { auto m0 = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to); auto m1 = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to); auto a = dpf::eval_interval(std::get<0>(*k32), from, to, m0); auto b = dpf::eval_interval(std::get<1>(*k32), from, to, m1); auto s = touch_buf(a.first) + touch_buf(b.first); auto m0b = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to); auto m1b = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to); auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, m0b); auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, m1b); s = s + touch_buf(a2.first) + touch_buf(b2.first); return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2); }); })); const auto pts = clustered(1u << 20, 64); auto path0 = std::make_shared(*k32)))>>( dpf::make_basic_path_memoizer(std::get<0>(*k32))); auto path1 = std::make_shared(*k32)))>>( dpf::make_basic_path_memoizer(std::get<1>(*k32))); works.push_back(profile::make_work("eval", "memoizer", "memo_path_reuse", 64, [k32, pts, path0, path1, prep] { return profile::with_prg([&] { std::uint64_t h = 0; for (auto x : *pts) { h ^= static_cast( (*dpf::eval_point(std::get<0>(*k32), x, *path0)).raw()); h ^= static_cast( (*dpf::eval_point(std::get<1>(*k32), x, *path1)).raw()); } auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2); return profile::with_costs(s, prep, 2 * sizeof(node_t) * 32, 64 * sizeof(std::uint64_t) * 2); }); })); works.push_back(profile::make_work("eval", "memoizer", "memo_path_pointwise", 64, [k32, pts, prep] { return profile::with_prg([&] { std::uint64_t h = 0; for (auto x : *pts) { h ^= static_cast( (*dpf::eval_point(std::get<0>(*k32), x)).raw()); h ^= static_cast( (*dpf::eval_point(std::get<1>(*k32), x)).raw()); } auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2); return profile::with_costs(s, prep, 0, 64 * sizeof(std::uint64_t) * 2); }); })); using recipe0_t = std::decay_t(*k32), pts->begin(), pts->end()))>; using recipe1_t = std::decay_t(*k32), pts->begin(), pts->end()))>; auto rec0 = std::make_shared(dpf::make_sequence_recipe( std::get<0>(*k32), pts->begin(), pts->end())); auto rec1 = std::make_shared(dpf::make_sequence_recipe( std::get<1>(*k32), pts->begin(), pts->end())); auto smemo0 = std::make_shared(*k32), *rec0))>>( dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0)); auto smemo1 = std::make_shared(*k32), *rec1))>>( dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1)); works.push_back(profile::make_work("eval", "memoizer", "memo_recipe_reuse", 64, [k32, rec0, rec1, smemo0, smemo1, prep] { return profile::with_prg([&] { auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0); auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1); auto s = touch_buf(a.first) + touch_buf(b.first); auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0); auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1); s = s + touch_buf(a2.first) + touch_buf(b2.first); return profile::with_costs(s, prep, s.out_bytes, 64 * sizeof(std::uint64_t) * 4); }); })); works.push_back(profile::make_work("eval", "memoizer", "memo_recipe_fresh", 64, [k32, rec0, rec1, prep] { return profile::with_prg([&] { auto m0 = dpf::make_inplace_reversing_sequence_memoizer( std::get<0>(*k32), *rec0); auto m1 = dpf::make_inplace_reversing_sequence_memoizer( std::get<1>(*k32), *rec1); auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0); auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1); auto s = touch_buf(a.first) + touch_buf(b.first); auto m0b = dpf::make_inplace_reversing_sequence_memoizer( std::get<0>(*k32), *rec0); auto m1b = dpf::make_inplace_reversing_sequence_memoizer( std::get<1>(*k32), *rec1); auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0b); auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1b); s = s + touch_buf(a2.first) + touch_buf(b2.first); return profile::with_costs(s, prep, s.out_bytes, 64 * sizeof(std::uint64_t) * 4); }); })); } // --- bits: built-in iterators vs hand-rolled scans (u8 domain) --- { using input_type = std::uint8_t; using output_type = dpf::bit; const input_type alpha = 40; auto bit_keys = std::make_shared>( dpf::make_dpf(alpha, output_type::one)); using dpf_type = std::decay_t(*bit_keys))>; auto memo0 = std::make_shared())>>( dpf::make_basic_full_memoizer()); auto memo1 = std::make_shared())>>( dpf::make_basic_full_memoizer()); auto full0 = dpf::eval_full(std::get<0>(*bit_keys), *memo0); auto full1 = dpf::eval_full(std::get<1>(*bit_keys), *memo1); auto buf0 = std::make_shared>(std::move(full0.first)); auto buf1 = std::make_shared>(std::move(full1.first)); auto iter0 = std::make_shared>(std::move(full0.second)); auto iter1 = std::make_shared>(std::move(full1.second)); const auto leaf_nodes = static_cast(1) << dpf_type::depth; const auto bit_items = static_cast(buf0->size()); const auto prep = sizeof(std::get<0>(*bit_keys)) + sizeof(std::get<1>(*bit_keys)); works.push_back(profile::make_work("eval", "bits", "bits_advice_builtin", leaf_nodes, [memo0, memo1, prep, leaf_nodes] { std::uint64_t h = 0; std::uint64_t n = 0; for (auto b : dpf::advice_bits_of(*memo0)) { h = (h << 1) ^ (b ? 1u : 0u); ++n; } for (auto b : dpf::advice_bits_of(*memo1)) h ^= (b ? 1u : 0u); auto s = touch_word(h ^ n, n); return profile::with_costs(s, prep, 0, leaf_nodes); })); works.push_back(profile::make_work("eval", "bits", "bits_advice_handroll", leaf_nodes, [memo0, memo1, prep, leaf_nodes] { std::uint64_t h = 0; std::uint64_t n = 0; const auto * p0 = memo0->begin(); const auto * e0 = memo0->end(); for (; p0 != e0; ++p0) { const auto * bytes = reinterpret_cast(p0); h = (h << 1) ^ (bytes[0] & 1u); ++n; } const auto * p1 = memo1->begin(); const auto * e1 = memo1->end(); for (; p1 != e1; ++p1) { const auto * bytes = reinterpret_cast(p1); h ^= (bytes[0] & 1u); } auto s = touch_word(h ^ n, n); return profile::with_costs(s, prep, 0, leaf_nodes); })); works.push_back(profile::make_work("eval", "bits", "bits_setbit_builtin", bit_items, [iter0, iter1, prep] { std::uint64_t h = 0; std::uint64_t n = 0; for (auto i : dpf::indices_set_in(*iter0)) { h ^= static_cast(i) + 0x9e3779b97f4a7c15ull; ++n; } for (auto i : dpf::indices_set_in(*iter1)) h ^= static_cast(i); auto s = touch_word(h ^ n, n * sizeof(std::size_t)); return profile::with_costs(s, prep, 0, n * sizeof(std::size_t)); })); works.push_back(profile::make_work("eval", "bits", "bits_setbit_handroll", bit_items, [buf0, buf1, prep] { std::uint64_t h = 0; std::uint64_t n = 0; const auto scan = [&](const auto & buf) { for (std::size_t i = 0; i < buf.size(); ++i) { if (static_cast(buf[i])) { h ^= i + 0x9e3779b97f4a7c15ull; ++n; } } }; scan(*buf0); scan(*buf1); auto s = touch_word(h ^ n, n * sizeof(std::size_t)); return profile::with_costs(s, prep, 0, n * sizeof(std::size_t)); })); works.push_back(profile::make_work("eval", "bits", "bits_parallel_builtin", bit_items, [buf0, buf1, prep, bit_items] { std::uint64_t h = 0; std::uint64_t n = 0; for (auto word : dpf::batch_of(*buf0, *buf1)) { h ^= static_cast(word[0]) ^ static_cast(word[1]); ++n; } auto s = touch_word(h ^ n, n * sizeof(std::uint64_t)); return profile::with_costs(s, prep, 0, bit_items); })); works.push_back(profile::make_work("eval", "bits", "bits_parallel_handroll", bit_items, [buf0, buf1, prep, bit_items] { std::uint64_t h = 0; const std::size_t n = std::min(buf0->size(), buf1->size()); for (std::size_t i = 0; i < n; ++i) h ^= (static_cast((*buf0)[i]) ? 1ull : 0ull) ^ (static_cast((*buf1)[i]) ? 2ull : 0ull); auto s = touch_word(h ^ n, n); return profile::with_costs(s, prep, 0, bit_items); })); } // --- inner-product: built-in vs hand-rolled interval + dot --- { using u32 = std::uint32_t; const u32 from = 1000000u; const u32 to = 1000255u; const auto items = static_cast(to - from + 1); auto weights = std::make_shared>(items); for (std::size_t i = 0; i < items; ++i) (*weights)[i] = 0x9e3779b97f4a7c15ull * (i + 1); const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32)); auto ip_memo0 = std::make_shared(*k32), from, to))>>( dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to)); auto ip_memo1 = std::make_shared(*k32), from, to))>>( dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to)); works.push_back(profile::make_work("eval", "inner-product", "ip_interval_builtin", items, [k32, weights, from, to, prep, items, ip_memo0, ip_memo1] { return profile::with_prg([&] { auto a = dpf::eval_inner_product(std::get<0>(*k32), from, to, *weights, *ip_memo0); auto b = dpf::eval_inner_product(std::get<1>(*k32), from, to, *weights, *ip_memo1); std::uint64_t ha = 0; std::uint64_t hb = 0; std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a)); std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b)); auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b)); return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t)); }); })); works.push_back(profile::make_work("eval", "inner-product", "ip_interval_handroll", items, [k32, weights, from, to, prep, items] { return profile::with_prg([&] { auto a = dpf::eval_interval(std::get<0>(*k32), from, to); auto b = dpf::eval_interval(std::get<1>(*k32), from, to); std::uint64_t dot0 = 0; std::uint64_t dot1 = 0; const std::size_t n = std::min(items, a.first.size()); for (std::size_t i = 0; i < n; ++i) { dot0 += static_cast(a.first[i].raw()) * (*weights)[i]; dot1 += static_cast(b.first[i].raw()) * (*weights)[i]; } auto s = touch_word(dot0 ^ dot1, a.first.size() * sizeof(a.first[0]) + b.first.size() * sizeof(b.first[0])); return profile::with_costs(s, prep, s.out_bytes, items * sizeof(std::uint64_t) * 2); }); })); works.push_back(profile::make_work("eval", "inner-product", "ip_interval_paired", items, [k32, weights, from, to, prep, items] { return profile::with_prg([&] { auto a = dpf::eval_inner_product(dpf::paired, std::get<0>(*k32), from, to, *weights); auto b = dpf::eval_inner_product(dpf::paired, std::get<1>(*k32), from, to, *weights); std::uint64_t ha = 0; std::uint64_t hb = 0; std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a)); std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b)); auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b)); return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t)); }); })); works.push_back(profile::make_work("eval", "inner-product", "ip_interval_columns", items, [k32, weights, from, to, prep, items] { return profile::with_prg([&] { auto a = dpf::eval_inner_product(dpf::columns, std::get<0>(*k32), from, to, std::tie(*weights)); auto b = dpf::eval_inner_product(dpf::columns, std::get<1>(*k32), from, to, std::tie(*weights)); std::uint64_t ha = 0; std::uint64_t hb = 0; const auto & a0 = std::get<0>(a); const auto & b0 = std::get<0>(b); std::memcpy(&ha, &a0, sizeof(ha) < sizeof(a0) ? sizeof(ha) : sizeof(a0)); std::memcpy(&hb, &b0, sizeof(hb) < sizeof(b0) ? sizeof(hb) : sizeof(b0)); auto s = touch_word(ha ^ hb, sizeof(a0) + sizeof(b0)); return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t)); }); })); } // --- helpers: point vs interval vs full (representative widths) --- { using u16 = std::uint16_t; const auto prep16 = sizeof(std::get<0>(*k16)) + sizeof(std::get<1>(*k16)); works.push_back(profile::make_work("eval", "helpers", "helper_point_u16", 1, [k16, prep16] { return profile::with_prg([&] { auto a = *dpf::eval_point(std::get<0>(*k16), u16{1512}); auto b = *dpf::eval_point(std::get<1>(*k16), u16{1512}); auto s = touch_word( static_cast(a.raw()) ^ static_cast(b.raw()), sizeof(a) + sizeof(b)); return profile::with_costs(s, prep16, 0, sizeof(a) + sizeof(b)); }); })); works.push_back(profile::make_work("eval", "helpers", "helper_interval_u16_L256", 256, [k16, prep16] { return profile::with_prg([&] { auto a = dpf::eval_interval(std::get<0>(*k16), u16{1000}, u16{1255}); auto b = dpf::eval_interval(std::get<1>(*k16), u16{1000}, u16{1255}); auto s = touch_buf(a.first) + touch_buf(b.first); return profile::with_costs(s, prep16, s.out_bytes, 256 * sizeof(std::uint64_t) * 2); }); })); works.push_back(profile::make_work("eval", "helpers", "helper_full_u8", 256, [k8] { const auto prep = sizeof(std::get<0>(*k8)) + sizeof(std::get<1>(*k8)); return profile::with_prg([&] { auto a = dpf::eval_full(std::get<0>(*k8)); auto b = dpf::eval_full(std::get<1>(*k8)); auto s = touch_buf(a.first) + touch_buf(b.first); return profile::with_costs(s, prep, s.out_bytes, s.out_bytes); }); })); } // --- defer: pre-assign full-domain expand vs rotate-after-assign --- { using u8 = std::uint8_t; using out_t = std::uint64_t; const u8 alpha = 40; const out_t beta = 0x9e3779b97f4a7c15ull; works.push_back(profile::make_work("eval", "defer", "defer_full_u8_expand", 256, [beta] { return profile::with_prg([&] { auto keys = dpf::make_dpf(dpf::wildcard_value{}, beta); auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys)); auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys)); auto d0 = dpf::defer_eval_full(std::get<0>(keys), b0); auto d1 = dpf::defer_eval_full(std::get<1>(keys), b1); (void)d0; (void)d1; return touch_buf(b0) + touch_buf(b1); }); })); works.push_back(profile::make_work("eval", "defer", "defer_interval_u8_L64_expand", 64, [beta] { return profile::with_prg([&] { auto keys = dpf::make_dpf(dpf::wildcard_value{}, beta); auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys)); auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys)); auto d0 = dpf::defer_eval_interval(std::get<0>(keys), u8{16}, u8{79}, b0); auto d1 = dpf::defer_eval_interval(std::get<1>(keys), u8{16}, u8{79}, b1); (void)d0; (void)d1; return touch_buf(b0) + touch_buf(b1); }); })); // Expand once against stable key storage, assign, then time .get(). { using keys_t = std::decay_t{}, beta))>; auto keys = std::make_shared( dpf::make_dpf(dpf::wildcard_value{}, beta)); using buf0_t = std::decay_t(*keys)))>; using buf1_t = std::decay_t(*keys)))>; auto b0 = std::make_shared( dpf::make_output_buffer_for_full(std::get<0>(*keys))); auto b1 = std::make_shared( dpf::make_output_buffer_for_full(std::get<1>(*keys))); using def0_t = std::decay_t(*keys), *b0))>; using def1_t = std::decay_t(*keys), *b1))>; auto def0 = std::make_shared( dpf::defer_eval_full(std::get<0>(*keys), *b0)); auto def1 = std::make_shared( dpf::defer_eval_full(std::get<1>(*keys), *b1)); { auto & k0 = std::get<0>(*keys); auto & k1 = std::get<1>(*keys); const u8 a0 = 0x12; const u8 a1 = static_cast(alpha - a0); const auto sh0 = k0.offset_x.compute_and_get_share(a0); const auto sh1 = k1.offset_x.compute_and_get_share(a1); k0.offset_x.reconstruct(sh1); k1.offset_x.reconstruct(sh0); } works.push_back(profile::make_work("eval", "defer", "defer_full_u8_get", 256, [keys, def0, def1, b0, b1] { (void)keys; auto v0 = def0->get(); auto v1 = def1->get(); std::uint64_t sink = 0; for (auto it = std::begin(v0); it != std::end(v0); ++it) sink ^= static_cast((*it).raw()); for (auto it = std::begin(v1); it != std::end(v1); ++it) sink ^= static_cast((*it).raw()); return touch_word(sink, b0->size() * sizeof((*b0)[0]) + b1->size() * sizeof((*b1)[0])); })); } works.push_back(profile::make_work("eval", "defer", "eager_full_u8_after_assign", 256, [beta, alpha] { return profile::with_prg([&] { auto keys = dpf::make_dpf(dpf::wildcard_value{}, beta); auto & k0 = std::get<0>(keys); auto & k1 = std::get<1>(keys); const u8 a0 = 0x12; const u8 a1 = static_cast(alpha - a0); const auto sh0 = k0.offset_x.compute_and_get_share(a0); const auto sh1 = k1.offset_x.compute_and_get_share(a1); k0.offset_x.reconstruct(sh1); k1.offset_x.reconstruct(sh0); auto a = dpf::eval_full(k0); auto b = dpf::eval_full(k1); return touch_buf(a.first) + touch_buf(b.first); }); })); } return profile::run_works(opt, works); }