diff --git a/doc/Doxyfile b/doc/Doxyfile index 1f47775..b0efa8b 100644 --- a/doc/Doxyfile +++ b/doc/Doxyfile @@ -1037,7 +1037,7 @@ EXAMPLE_PATTERNS = * # irrespective of the value of the RECURSIVE tag. # The default value is: NO. -EXAMPLE_RECURSIVE = NO +EXAMPLE_RECURSIVE = YES # The IMAGE_PATH tag can be used to specify one or more files or directories # that contain images that are to be included in the documentation (see the diff --git a/doc/examples.dox b/doc/examples.dox index cb34c12..fa59ad8 100644 --- a/doc/examples.dox +++ b/doc/examples.dox @@ -17,10 +17,10 @@ /// @brief an example of `dpf::keyword` in use /// @example input_types/xor_wrapper.cpp xor_wrapper.cpp -/// @brief an example of `dpf::memoizers` in use +/// @brief an example of `dpf::xor_wrapper` as an input /// @example input_types/custom.cpp custom.cpp -/// @brief an example of `dpf::output_buffers` in use +/// @brief an example of a custom input type /// @} @@ -30,25 +30,25 @@ /// @{ /// @example output_types/integral_types.cpp integral_types.cpp -/// @brief an example of `dpf::eval_point` in use +/// @brief an example of an integer output /// @example output_types/extended_types.cpp extended_types.cpp -/// @brief an example of `dpf::eval_interval` in use +/// @brief an example of a 128-bit integer output /// @example output_types/bit.cpp bit.cpp -/// @brief an example of `dpf::eval_full` in use +/// @brief an example of `dpf::bit` as an output /// @example output_types/bitstring.cpp bitstring.cpp -/// @brief an example of `dpf::eval_sequence` in use +/// @brief an example of `dpf::bitstring` as an output /// @example output_types/wildcard.cpp wildcard.cpp -/// @brief an example of `dpf::eval_sequence` in use +/// @brief an example of `dpf::wildcard` as an output /// @example output_types/xor_wrapper.cpp xor_wrapper.cpp -/// @brief an example of `dpf::memoizers` in use +/// @brief an example of `dpf::xor_wrapper` as an output /// @example output_types/custom.cpp custom.cpp -/// @brief an example of `dpf::output_buffers` in use +/// @brief an example of a custom output type /// @} @@ -73,7 +73,10 @@ /// @brief an example of `dpf::memoizers` in use /// @example evaluation/output_buffers.cpp output_buffers.cpp -/// @brief an example of `dpf::output_buffers` in use +/// @brief an example of `dpf::output_buffer` in use + +/// @example evaluation/buffered_prg.cpp buffered_prg.cpp +/// @brief an example of `dpf::randomness::buffered_prg` and `lane_table` /// @} diff --git a/doc/pages/evaluation.md b/doc/pages/evaluation.md index d189998..328fe7a 100644 --- a/doc/pages/evaluation.md +++ b/doc/pages/evaluation.md @@ -1,34 +1,126 @@ - -Once a `DPF` generated, the *eval_* * functions are used to evaluate differents inputs. -The appropriate function depends on your specific needs. If you only need the DPF's output -for a single input value, use `eval_point`. For evaluating a continuous range of inputs, -`eval_interval` is suitable. To analyze the DPF's behavior across its entire domain, use `eval_full`. -The code likely offers different implementations of memoization and output buffers, allowing you to -optimize for memory usage or execution speed depending on your needs.\n -Using a PRG while making a `DPF` allows the user to check how much it cost to manipulate the `DPF`s. + -# Memoizers -The `memoizers` remembers the most used path while the DPF is being created. These are usefull functions -to improve the speed and the cost of execution. +`make_dpf(x, y)` returns one key per party. Evaluation of a key yields that +party's share. Leaf outputs are subtractive shares: open them with +`dpf::reconstruct`, which computes `share0 - share1`. Comparison outputs are +additive shares: `reconstruct` computes `share0 + share1`. A single-output +`eval_point` returns a small handle; `*handle` is the share. An unassigned +`dpf::wildcard` output throws `std::runtime_error`. + +`[from, to]` is inclusive. The points passed to `eval_sequence` are a +nondecreasing range; an unsorted range throws `std::runtime_error`. + +Memoizers hold interior nodes between calls. Output buffers hold the shares +a multi-point evaluation writes. Pass both as mutable named objects when a +later call should reuse them. The factories +`make_basic_path_memoizer`, `make_basic_interval_memoizer`, and +`make_*_sequence_memoizer` unwrap `party_key`, so a workspace built from +either party's type accepts both parties. Name that type with +`dpf::unwrap_party_key_t>`. + +# Memoizers {#memoizers} + +## Path memoizers {#path_memoizers} + +`eval_point` walks one root-to-leaf path. `make_basic_path_memoizer()` +keeps every node of the previous point, and the next point recomputes only +the suffix after the common prefix. `make_nonmemoizing_path_memoizer()` +keeps one node and starts from the root on every call. A one-off +`eval_point(key, x)` uses the nonmemoizing memoizer. + +Pass the memoizer as a mutable lvalue. The default argument is a new +temporary, so it has no previous point to resume from. Keep a separate +memoizer for each key you are in the middle of evaluating. A different root +restarts the path. **Code samples**\n -For instance, in the code below the utilization of `dpf::make_basic_path_memoizer` reduced by 10 the time of execution -compare to the code that is commented that doesn't use the `memoizers`.
- memoizers.cpp \include{cpp} evaluation/memoizers.cpp
+## Interval memoizers {#interval_memoizers} -# dpf::eval_point -This function evaluate a single input of a DPF. The XOR result of the `eval_point` for both shares will only be equal to 1 if it represents the correct input in both evaluations. As input arguments it uses the `share` and the input to evaluate (note: it can't be a wildcard, otherwise it will throw an error).\n +`eval_interval` and `eval_full` expand every leaf in a range. +`make_basic_interval_memoizer(from, to)` stores two levels of that +range. That is the workspace the convenience overloads allocate. +`make_full_tree_interval_memoizer(from, to)` keeps every level. +`make_basic_full_memoizer()` and `make_full_tree_full_memoizer()` +are the same workspaces sized for the whole domain. -**See also**\n -PIR +Size the memoizer for the widest interval you will pass to it. A wider +interval throws `std::length_error`. The same key and the same endpoints +leave the final interior level in place. A different key or a different +interval rebuilds into the same allocation. -**Pro tip**\n -Use the `dpf::pathmemoizer` for a faster execution.\n +Passing only the memoizer still allocates a fresh output buffer and returns +`std::pair(buffer, iterable)`. + +## Sequence memoizers {#sequence_memoizers} + +`make_sequence_recipe(begin, end)` compiles a sorted point list into a +traversal. The recipe depends on the input type, and one recipe serves every +key of that type. + +A sequence memoizer stores a reference to the recipe object it was built +from and checks later calls by address. Pass that same object, and keep the +recipe alive for as long as the memoizer is used. A copy of the recipe +throws `std::logic_error`. + +`make_double_space_sequence_memoizer(recipe)` keeps two levels. It is +what `eval_sequence(key, recipe, buffer)` allocates when you omit the +memoizer. `make_inplace_reversing_sequence_memoizer(recipe)` keeps one +level and reverses direction as it descends. +`make_full_tree_sequence_memoizer(recipe)` retains every level. A key +whose depth differs from the recipe throws `std::logic_error`. + +# Output buffers {#output_buffers} + +`output_buffer` is move-only storage with `size`, iterators, `data`, and +`operator[]`. Build it with the factory that matches the evaluation: + +- `make_output_buffer_for_interval(key, from, to)` +- `make_output_buffer_for_full(key)` +- `make_output_buffer_for_subsequence(key, begin, end, tag)` +- `make_output_buffer_for_recipe_subsequence(key, recipe, tag)` + +On a `party_key`, leaf slots are `subtractive_share`s and comparison slots +are `additive_share`s. `dpf::bit`, `dpf::twobit`, and `dpf::nyble` slots are +packed. Trivially default-constructible slot types are left uninitialized; +the evaluation overwrites every slot it is responsible for. + +`eval_interval` and recipe `eval_sequence` take the buffer as a non-const +reference, so the argument is a named object. The returned iterable refers +into that buffer. Read it while the buffer is alive, and only over the +points the iterable covers. The next evaluation overwrites those slots. + +For one output, the convenience overload returns the buffer itself as the +first element of the pair. For several output indices it returns a tuple of +buffers. + +`make_output_buffer(dpf::out, key, from, to)` and +`make_output_buffer(dpf::cmp, key, n)` size a buffer for one channel of a +multi-output or comparison key. The slot types follow the same party-share +rule. + +**Code samples**\n +
+ + - output_buffers.cpp \include{cpp} evaluation/output_buffers.cpp + +
+ +# dpf::eval_point {#eval_point} + +`eval_point(key, x)` evaluates output 0 at one input. +`eval_point(key, x)` selects another output. Two or more indices, +`eval_point<0, 1>(key, x)`, return a tuple of shares rather than handles. +`eval_point(key, x, path)` continues a path memoizer. + +`eval_point(dpf::out, key, x, path)` and `eval_point(dpf::cmp, key, x, path)` +are the same walk with an explicit channel. Comparison results are additive +shares. **Code samples**\n
@@ -37,12 +129,13 @@ Use the `dpf::pathmemoizer` for a faster execution.\n
-# dpf::eval_interval -This function evaluates a contiguous range of inputs. As input arguments it uses the `share` generated by `make_dpf`, -`from` and `to` for the range of inputs to evaluate.\n +# dpf::eval_interval {#eval_interval} -**See also**\n -PIR +`eval_interval(key, from, to)` evaluates every input from `from` through +`to`. The iterable yields one share per input, in that order. Optional +arguments are an output buffer and then an interval memoizer. An output +index pack, `eval_interval<0, 1>(key, from, to, buffers, memo)`, writes each +selected output. **Code samples**\n
@@ -51,8 +144,13 @@ PIR
-# dpf::eval_full -This function evaluate all the passible inputs it only uses as argument the `share` of the `DPF` to evaluate. +# dpf::eval_full {#eval_full} + +`eval_full(key)` is the closed interval from +`std::numeric_limits::min()` through `max()`. The buffer and +full-domain memoizer overloads match `eval_interval`. +`make_output_buffer_for_full(key)` and `make_basic_full_memoizer()` +size both for that domain. **Code samples**\n
@@ -61,14 +159,16 @@ This function evaluate all the passible inputs it only uses as argument the `sha
+# dpf::eval_sequence {#eval_sequence} -# dpf::eval_sequence -This function evaluate a subset of inputs that is not contiguous (useful for a `DPF` made with `keyword`), -it uses as arguments the `share` generated by `make_dpf`, `from` and `to` for the subset of inputs to evaluate.\n -For a better utilization, you can use the `make_sequence_recipe` it takes as input a sorted list and returns a `recipe`. -The cost of creating is a little bit worse than just calling `eval_sequence`, however once the `recipe` created -`eval_sequence` is faster and has a better cost. +`eval_sequence(key, begin, end, tag)` evaluates a sorted list. +`dpf::return_output_only_tag_` stores one share per listed point. +`dpf::return_entire_node_tag_` stores whole leaves; it is the default. +The iterable still yields one share per listed point, in list order. +`eval_sequence(key, recipe, buffer, memo, tag)` repeats that list. +`memo` is a sequence memoizer bound to `recipe`. Omit `memo` to allocate a +`double_space` workspace for that call. **Code samples**\n
@@ -77,4 +177,22 @@ The cost of creating is a little bit worse than just calling `eval_sequence`, ho
-# Output buffers \ No newline at end of file +# Buffered PRG {#buffered_prg} + +`dpf::randomness::buffered_prg` (alias +`dpf::randomness::aes_buffered_prg`) is a forward cursor with one +PRG stream per value type. `get()` and `fill(out, n)` consume the +cursor. `at(index)` reads an absolute index and leaves the cursor where +it is. `sampled()` is how far `get` and `fill` have advanced. +`per_stream_buffer_elems` is at least 1. + +`dpf::randomness::lane_table` is the seekable form for a runtime set of +roles. `value_at(role, index)` and `mask_at(role, index)` are independent +streams, and a repeated index returns the same element. + +**Code samples**\n +
+ + - buffered_prg.cpp \include{cpp} evaluation/buffered_prg.cpp + +
diff --git a/examples/evaluation/buffered_prg.cpp b/examples/evaluation/buffered_prg.cpp new file mode 100644 index 0000000..8475b99 --- /dev/null +++ b/examples/evaluation/buffered_prg.cpp @@ -0,0 +1,62 @@ +#include +#include +#include + +#include "dpf.hpp" + +/// `buffered_prg` is a forward cursor, one stream per value type. +/// `at(index)` reads by absolute index and does not move the cursor. +/// `lane_table` is the seekable form: value and mask streams per role. +int main() +{ + //! [buffered-prg] + dpf::randomness::aes_buffered_prg prg( + /*per stream*/ 64); + + std::uint64_t first = prg.get<0>(); + std::uint64_t second = prg.get<0>(); + // Absolute index 0 is `first` again. The cursor stays at 2. + std::uint64_t replay = prg.at<0>(0); + std::uint32_t other_stream = prg.get<1>(); + + std::uint64_t batch[4]; + prg.fill<0>(batch, 4); + //! [buffered-prg] + + if (std::memcmp(&replay, &first, sizeof(first)) != 0) + { + std::cerr << "buffered_prg at(0)\n"; + return 1; + } + if (prg.sampled<0>() != 6) + { + std::cerr << "buffered_prg cursor\n"; + return 1; + } + // Stream 1 has its own cursor. + if (prg.sampled<1>() != 1) + { + std::cerr << "buffered_prg stream 1\n"; + return 1; + } + (void)second; + (void)other_stream; + (void)batch; + + //! [lane-table] + dpf::randomness::lane_table lanes(/*window*/ 32); + // Order does not matter. Masks are a separate stream from values. + auto v_late = lanes.value_at(/*role*/ 3, /*index*/ 10); + auto v_early = lanes.value_at(3, 10); + auto mask = lanes.mask_at(3, 10); + //! [lane-table] + if (std::memcmp(&v_late, &v_early, sizeof(v_late)) != 0) + { + std::cerr << "lane_table replay\n"; + return 1; + } + (void)mask; + + std::cout << "ok\n"; + return 0; +} diff --git a/examples/evaluation/eval_full.cpp b/examples/evaluation/eval_full.cpp index d6e3999..f30a0d6 100644 --- a/examples/evaluation/eval_full.cpp +++ b/examples/evaluation/eval_full.cpp @@ -1,31 +1,44 @@ +#include #include +#include #include "dpf.hpp" -int main(int arc, char * argv[]) +/// Every input of the domain, from `numeric_limits::min()` through +/// `max()`. Same shape as `eval_interval`. +int main() { - uint16_t x = 42; // Input value - using prg = dpf::prg::counter_wrapper; // This is just to count the number of PRG invocations - auto before = prg::count(); // In order to show how much this program cost - auto [dpf0, dpf1] = dpf::make_dpf(x); - auto after = prg::count(); - std::cout << "dpf::make_dpf used " << (after-before) << "\n"; + const std::uint8_t alpha = 42; + const std::uint64_t beta = 7; + auto [k0, k1] = dpf::make_dpf(alpha, beta); - before = prg::count(); - auto [buf0, iter0] = dpf::eval_full(dpf0); - after = prg::count(); - std::cout << "dpf::eval_full(dpf0) used " << (after-before) << "\n"; - before = prg::count(); - auto [buf1, iter1] = dpf::eval_full(dpf1); - after = prg::count(); - std::cout << "dpf::eval_full(dpf1) used " << (after-before) << "\n"; - // Retrieve the original input by iterating over the two buffers - for (size_t i = 0; i < buf0.size(); ++i) { - bool item1 = buf0[i]; - bool item2 = buf1[i]; - if (item1 ^ item2) std::cout << "The original input is: " << i << std::endl; + //! [eval-full] + auto [buf0, iter0] = dpf::eval_full(k0); + auto [buf1, iter1] = dpf::eval_full(k1); + + auto it0 = std::begin(iter0); + auto it1 = std::begin(iter1); + for (int x = std::numeric_limits::min(); + x <= std::numeric_limits::max(); + ++x, ++it0, ++it1) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (static_cast(x) == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "eval_full\n"; + return 1; + } + } + //! [eval-full] + if (it0 != std::end(iter0) || it1 != std::end(iter1)) + { + std::cerr << "eval_full length\n"; + return 1; } - std::cout << "Total PRG invocation: " << prg::count() << "\n"; + std::cout << beta << "\n"; + (void)buf0; + (void)buf1; return 0; -} \ No newline at end of file +} diff --git a/examples/evaluation/eval_interval.cpp b/examples/evaluation/eval_interval.cpp index 901af7e..69df068 100644 --- a/examples/evaluation/eval_interval.cpp +++ b/examples/evaluation/eval_interval.cpp @@ -1,42 +1,43 @@ +#include #include + #include "dpf.hpp" -int main(int arc, char * argv[]) +/// Inclusive range. The returned iterable yields one share per input in +/// `[from, to]`, in that order. +int main() { - uint16_t x = 42, y; - using prg = dpf::prg::counter_wrapper; + const std::uint8_t alpha = 42; + const std::uint64_t beta = 7; + const std::uint8_t from = 40; + const std::uint8_t to = 50; + auto [k0, k1] = dpf::make_dpf(alpha, beta); - // Make the DPF - auto before = prg::count(); - auto [dpf0, dpf1] = dpf::make_dpf(x); - auto after = prg::count(); - std::cout << "dpf::make_dpf prg invocation: " << (after-before) << "\n"; + //! [eval-interval] + auto [buf0, iter0] = dpf::eval_interval(k0, from, to); + auto [buf1, iter1] = dpf::eval_interval(k1, from, to); - // Evaluate the DPF by interval - before = prg::count(); - int from = 0, to = 49; - auto [buf0, iter0] = dpf::eval_interval(dpf0, from, to); - auto [buf1, iter1] = dpf::eval_interval(dpf1, from, to); - after = prg::count(); - std::cout << "dpf::eval_interval prg invocation: " << (after-before) << "\n"; - - // Retrieve the original input by iterating over the two buffers - std::vector result; - for (size_t i = from; i < to+1; ++i) { - bool item1 = buf0[i]; - bool item2 = buf1[i]; - result.push_back(item1 ^ item2); - if (item1 ^ item2) y=i; + auto it0 = std::begin(iter0); + auto it1 = std::begin(iter1); + for (std::uint8_t x = from; x <= to; ++x, ++it0, ++it1) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "eval_interval\n"; + return 1; + } } - // Print out the XOR interval - for (const auto& item : result) { - std::cout << static_cast(item); + //! [eval-interval] + if (it0 != std::end(iter0) || it1 != std::end(iter1)) + { + std::cerr << "eval_interval length\n"; + return 1; } - std::cout << std::endl; - if (y == x) std::cout << "The orginal value is: " << x << std::endl; - else std::cout << "The evaluated inputs did not match the original value." << std::endl; - - std::cout << "Total PRG invocation: " << prg::count() << std::endl; + std::cout << dpf::reconstruct(*std::begin(iter0), *std::begin(iter1)) << "\n"; + (void)buf0; + (void)buf1; return 0; -} \ No newline at end of file +} diff --git a/examples/evaluation/eval_point.cpp b/examples/evaluation/eval_point.cpp index 98c8e2e..9f1b0b8 100644 --- a/examples/evaluation/eval_point.cpp +++ b/examples/evaluation/eval_point.cpp @@ -1,17 +1,54 @@ +#include #include #include "dpf.hpp" -int main(int arc, char * argv[]) +/// One input. `*eval_point` is that party's share of output 0. +/// Reconstruct with `dpf::reconstruct` (leaf outputs are subtractive). +int main() { - uint16_t x = 42; - auto [dpf0, dpf1] = dpf::make_dpf(x); + const std::uint8_t alpha = 42; + const std::uint64_t beta = 7; + auto [k0, k1] = dpf::make_dpf(alpha, beta); + using key_t = dpf::unwrap_party_key_t>; - auto res = dpf::eval_point(dpf0, x); + //! [eval-point] + auto y0 = *dpf::eval_point(k0, alpha); + auto y1 = *dpf::eval_point(k1, alpha); + std::uint64_t opened = dpf::reconstruct(y0, y1); + //! [eval-point] + if (opened != beta) + { + std::cerr << "eval_point at the programmed input\n"; + return 1; + } - std::cout << *dpf::eval_point(dpf0, 41) << " ^ " << *dpf::eval_point(dpf1, 41) << " = " << (*dpf::eval_point(dpf0, 41) ^ *dpf::eval_point(dpf1, 41)) << "\n"; // = 0 - std::cout << *dpf::eval_point(dpf0, x) << " ^ " << *dpf::eval_point(dpf1, x) << " = " << (*dpf::eval_point(dpf0, x) ^ *dpf::eval_point(dpf1, x)) << "\n"; // = 1 - std::cout << *dpf::eval_point(dpf0, 43) << " ^ " << *dpf::eval_point(dpf1, 43) << " = " << (*dpf::eval_point(dpf0, 43) ^ *dpf::eval_point(dpf1, 43)) << "\n"; // = 0 + auto off0 = *dpf::eval_point(k0, std::uint8_t{41}); + auto off1 = *dpf::eval_point(k1, std::uint8_t{41}); + if (dpf::reconstruct(off0, off1) != 0) + { + std::cerr << "eval_point off the programmed input\n"; + return 1; + } + //! [eval-point-memo] + // One mutable memoizer per key. Nearby points reuse the common prefix. + auto path0 = dpf::make_basic_path_memoizer(); + auto path1 = dpf::make_basic_path_memoizer(); + for (int x = 40; x <= 44; ++x) + { + auto s0 = *dpf::eval_point(k0, static_cast(x), path0); + auto s1 = *dpf::eval_point(k1, static_cast(x), path1); + std::uint64_t got = dpf::reconstruct(s0, s1); + std::uint64_t expect = (static_cast(x) == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "path memoizer\n"; + return 1; + } + } + //! [eval-point-memo] + + std::cout << opened << "\n"; return 0; -} \ No newline at end of file +} diff --git a/examples/evaluation/eval_sequence.cpp b/examples/evaluation/eval_sequence.cpp index 8cbc5c2..c36f7df 100644 --- a/examples/evaluation/eval_sequence.cpp +++ b/examples/evaluation/eval_sequence.cpp @@ -1,54 +1,114 @@ -#include +#include +#include #include + #include "dpf.hpp" -using std::chrono::high_resolution_clock; -using std::chrono::duration_cast; -using std::chrono::duration; -using std::chrono::milliseconds; - -int main(int argc, char * argv[]) +/// A sorted point list. `return_output_only_tag_` stores one share per point. +/// A `sequence_recipe` compiled from that list is reusable across keys. +int main() { - using input_type = uint8_t; - using prg = dpf::prg::counter_wrapper; + const std::uint8_t alpha = 42; + const std::uint64_t beta = 7; + auto [k0, k1] = dpf::make_dpf(alpha, beta); + using key_t = dpf::unwrap_party_key_t>; - constexpr int N = 50; - std::array keys{}; - for(int i=0; i points{1, 7, 42, 100, 200}; - // eval_sequence with recipe - input_type x = 42; - auto [dpf0, dpf1] = dpf::make_dpf(x); // First DPF to be able to create the recipe - auto t1 = high_resolution_clock::now(); // To measure the time of execution - auto before = prg::count(); // To count the number of PRG invocations - auto recipe0 = dpf::make_sequence_recipe(dpf0, std::begin(keys), std::end(keys)); // Create a recipe - auto recipe1 = dpf::make_sequence_recipe(dpf1, std::begin(keys), std::end(keys)); // Create a recipe - for (int i=0; i(i); // Make 50 DPFs - dpf::eval_sequence(dpf0, recipe0); // Evaluate the DPFs with the recipe - dpf::eval_sequence(dpf1, recipe0); // Evaluate the DPFs with the recipe + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "eval_sequence\n"; + return 1; + } + ++it0; + ++it1; } - auto after = prg::count(); // Count the number of PRG invocations - std::cout << "dpf::eval_sequence with recipe " << (after-before) << "\n"; + //! [eval-sequence] - // eval_sequence without the recipe - auto t2 = high_resolution_clock::now(); - duration ms_double = t2 - t1; - std::cout << "Time of execution: " << ms_double.count() << "ms\n"; - auto t3 = high_resolution_clock::now(); - before = prg::count(); - for (int i=0; i(points.begin(), points.end()); + auto memo0 = dpf::make_double_space_sequence_memoizer(recipe); + auto memo1 = dpf::make_double_space_sequence_memoizer(recipe); + auto sbuf0 = dpf::make_output_buffer_for_recipe_subsequence(k0, recipe, + dpf::return_output_only_tag_{}); + auto sbuf1 = dpf::make_output_buffer_for_recipe_subsequence(k1, recipe, + dpf::return_output_only_tag_{}); + auto seq0 = dpf::eval_sequence(k0, recipe, sbuf0, memo0, + dpf::return_output_only_tag_{}); + auto seq1 = dpf::eval_sequence(k1, recipe, sbuf1, memo1, + dpf::return_output_only_tag_{}); + //! [eval-sequence-recipe] + + // Omitting the memoizer allocates a double-space workspace for that call. + seq0 = dpf::eval_sequence(k0, recipe, sbuf0, dpf::return_output_only_tag_{}); + seq1 = dpf::eval_sequence(k1, recipe, sbuf1, dpf::return_output_only_tag_{}); + it0 = std::begin(seq0); + it1 = std::begin(seq1); + for (std::uint8_t x : points) { - auto [dpf00, dpf11] = dpf::make_dpf(i); - dpf::eval_sequence(dpf00, std::begin(keys), std::end(keys)); - dpf::eval_sequence(dpf11, std::begin(keys), std::end(keys)); + if (dpf::reconstruct(*it0, *it1) != ((x == alpha) ? beta : 0)) + { + std::cerr << "eval_sequence recipe, default memoizer\n"; + return 1; + } + ++it0; + ++it1; } - after = prg::count(); - std::cout << "dpf::eval_sequence used " << (after-before) << "\n"; - auto t4 = high_resolution_clock::now(); - duration ms_double2 = t4 - t3; - std::cout << "Time of execution with the memoizers: " << ms_double2.count() << "ms\n"; + seq0 = dpf::eval_sequence(k0, recipe, sbuf0, memo0, dpf::return_output_only_tag_{}); + seq1 = dpf::eval_sequence(k1, recipe, sbuf1, memo1, dpf::return_output_only_tag_{}); + + it0 = std::begin(seq0); + it1 = std::begin(seq1); + for (std::uint8_t x : points) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "eval_sequence recipe\n"; + return 1; + } + ++it0; + ++it1; + } + + // Same recipe, same memoizers, same buffers: a second key overwrites them. + auto [k0b, k1b] = dpf::make_dpf(std::uint8_t{100}, std::uint64_t{9}); + seq0 = dpf::eval_sequence(k0b, recipe, sbuf0, memo0, dpf::return_output_only_tag_{}); + seq1 = dpf::eval_sequence(k1b, recipe, sbuf1, memo1, dpf::return_output_only_tag_{}); + it0 = std::begin(seq0); + it1 = std::begin(seq1); + for (std::uint8_t x : points) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == 100) ? 9 : 0; + if (got != expect) + { + std::cerr << "eval_sequence recipe reuse\n"; + return 1; + } + ++it0; + ++it1; + } + + std::cout << beta << "\n"; + (void)buf0; + (void)buf1; return 0; -} \ No newline at end of file +} diff --git a/examples/evaluation/memoizers.cpp b/examples/evaluation/memoizers.cpp index 45ab8d3..b19a311 100644 --- a/examples/evaluation/memoizers.cpp +++ b/examples/evaluation/memoizers.cpp @@ -1,50 +1,128 @@ +#include +#include #include -#include + #include "dpf.hpp" -using std::chrono::high_resolution_clock; -using std::chrono::duration_cast; -using std::chrono::duration; -using std::chrono::milliseconds; - -int main(int arc, char * argv[]) +/// Memoizers are workspaces of interior nodes. Pass a mutable lvalue. +/// A temporary (including the default argument) cannot remember a prefix. +int main() { - // Making the DPF with an integer value - uint16_t x = 42; - using prg = dpf::prg::counter_wrapper; - auto [dpf0, dpf1] = dpf::make_dpf(x); + const std::uint8_t alpha = 42; + const std::uint64_t beta = 7; + auto [k0, k1] = dpf::make_dpf(alpha, beta); + using key_t = dpf::unwrap_party_key_t>; - // Evaluating the DPF and counting how much it cost without memoizers - auto t1 = high_resolution_clock::now(); - auto before = prg::count(); - for (int i = 0; i<1024*1024; i++) + //! [path-memoizer] + auto path0 = dpf::make_basic_path_memoizer(); + auto path1 = dpf::make_basic_path_memoizer(); + for (int x = 0; x < 256; ++x) { - dpf::eval_point(dpf0, i); - + auto y0 = *dpf::eval_point(k0, static_cast(x), path0); + auto y1 = *dpf::eval_point(k1, static_cast(x), path1); + std::uint64_t got = dpf::reconstruct(y0, y1); + std::uint64_t expect = (static_cast(x) == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "basic_path_memoizer\n"; + return 1; + } } - // Printing out the results - auto after = prg::count(); - std::cout << "Without memoizers: " << "\n"; - std::cout << "PRG invocation: " << after-before << "\n"; - auto t2 = high_resolution_clock::now(); - duration ms_double = t2 - t1; - std::cout << "Time of execution: " << ms_double.count() << "ms\n"; + //! [path-memoizer] - // Evaluating the DPF and counting how much it cost with memoizers - auto t3 = high_resolution_clock::now(); - before = prg::count(); - auto path = dpf::make_basic_path_memoizer(dpf0); - for (int i = 0; i<1024*1024; i++) + // One node, no prefix reuse. Correct for a single query. + auto once0 = dpf::make_nonmemoizing_path_memoizer(); + auto once1 = dpf::make_nonmemoizing_path_memoizer(); + if (dpf::reconstruct(*dpf::eval_point(k0, alpha, once0), + *dpf::eval_point(k1, alpha, once1)) != beta) { - dpf::eval_point(dpf0, i, path); + std::cerr << "nonmemoizing_path_memoizer\n"; + return 1; } - // Printing out the results - after = prg::count(); - std::cout << "With memoizers: " << "\n"; - std::cout << "PRG invocation: " << after-before << "\n"; - auto t4 = high_resolution_clock::now(); - duration ms_double2 = t4 - t3; - std::cout << "Time of execution with the memoizers: " << ms_double2.count() << "ms\n"; + //! [interval-memoizer] + const std::uint8_t from = 40; + const std::uint8_t to = 50; + // Sized for [from, to]. A wider interval throws std::length_error. + // `make_full_tree_interval_memoizer` keeps every level instead of two. + auto memo0 = dpf::make_basic_interval_memoizer(from, to); + auto memo1 = dpf::make_basic_interval_memoizer(from, to); + auto [ibuf0, i0] = dpf::eval_interval(k0, from, to, memo0); + auto [ibuf1, i1] = dpf::eval_interval(k1, from, to, memo1); + //! [interval-memoizer] + + auto it0 = std::begin(i0); + auto it1 = std::begin(i1); + for (std::uint8_t x = from; x <= to; ++x, ++it0, ++it1) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "basic_interval_memoizer\n"; + return 1; + } + } + + // A different key rebuilds into the same memoizer. + auto [k0b, k1b] = dpf::make_dpf(std::uint8_t{44}, std::uint64_t{9}); + std::tie(ibuf0, i0) = dpf::eval_interval(k0b, from, to, memo0); + std::tie(ibuf1, i1) = dpf::eval_interval(k1b, from, to, memo1); + it0 = std::begin(i0); + it1 = std::begin(i1); + for (std::uint8_t x = from; x <= to; ++x, ++it0, ++it1) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == 44) ? 9 : 0; + if (got != expect) + { + std::cerr << "interval memoizer reuse\n"; + return 1; + } + } + + //! [sequence-memoizer] + std::array points{1, 7, 42, 100, 200}; + // The memoizer stores a reference to this recipe and checks it by address. + auto recipe = dpf::make_sequence_recipe(points.begin(), points.end()); + auto seq0 = dpf::make_inplace_reversing_sequence_memoizer(recipe); + auto seq1 = dpf::make_inplace_reversing_sequence_memoizer(recipe); + auto [sbuf0, s0] = dpf::eval_sequence(k0, recipe, seq0, dpf::return_output_only_tag_{}); + auto [sbuf1, s1] = dpf::eval_sequence(k1, recipe, seq1, dpf::return_output_only_tag_{}); + //! [sequence-memoizer] + + it0 = std::begin(s0); + it1 = std::begin(s1); + for (std::uint8_t x : points) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "sequence memoizer\n"; + return 1; + } + ++it0; + ++it1; + } + + std::tie(sbuf0, s0) = dpf::eval_sequence(k0b, recipe, seq0, dpf::return_output_only_tag_{}); + std::tie(sbuf1, s1) = dpf::eval_sequence(k1b, recipe, seq1, dpf::return_output_only_tag_{}); + it0 = std::begin(s0); + it1 = std::begin(s1); + for (std::uint8_t x : points) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == 44) ? 9 : 0; + if (got != expect) + { + std::cerr << "sequence memoizer reuse\n"; + return 1; + } + ++it0; + ++it1; + } + + std::cout << beta << "\n"; return 0; -} \ No newline at end of file +} diff --git a/examples/evaluation/output_buffers.cpp b/examples/evaluation/output_buffers.cpp index ac823b7..0681e05 100644 --- a/examples/evaluation/output_buffers.cpp +++ b/examples/evaluation/output_buffers.cpp @@ -1,6 +1,82 @@ +#include +#include + #include "dpf.hpp" -int main(int argc, char * argv[]) +/// Output buffers are move-only. `eval_interval` takes the buffer by +/// non-const reference, so name it. The iterable points into that buffer; +/// read it only while the buffer is still alive, and only over the points +/// the iterable covers. +int main() { + const std::uint8_t alpha = 42; + const std::uint64_t beta = 7; + const std::uint8_t from = 40; + const std::uint8_t to = 50; + auto [k0, k1] = dpf::make_dpf(alpha, beta); + using key_t = dpf::unwrap_party_key_t>; + + //! [output-buffer] + auto buf0 = dpf::make_output_buffer_for_interval(k0, from, to); + auto buf1 = dpf::make_output_buffer_for_interval(k1, from, to); + auto memo0 = dpf::make_basic_interval_memoizer(from, to); + auto memo1 = dpf::make_basic_interval_memoizer(from, to); + + auto iter0 = dpf::eval_interval(k0, from, to, buf0, memo0); + auto iter1 = dpf::eval_interval(k1, from, to, buf1, memo1); + + auto it0 = std::begin(iter0); + auto it1 = std::begin(iter1); + for (std::uint8_t x = from; x <= to; ++x, ++it0, ++it1) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "output buffer\n"; + return 1; + } + } + //! [output-buffer] + + // The next evaluation overwrites the same slots. + auto [k0b, k1b] = dpf::make_dpf(std::uint8_t{44}, std::uint64_t{9}); + iter0 = dpf::eval_interval(k0b, from, to, buf0, memo0); + iter1 = dpf::eval_interval(k1b, from, to, buf1, memo1); + it0 = std::begin(iter0); + it1 = std::begin(iter1); + for (std::uint8_t x = from; x <= to; ++x, ++it0, ++it1) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (x == 44) ? 9 : 0; + if (got != expect) + { + std::cerr << "output buffer reuse\n"; + return 1; + } + } + + //! [output-buffer-full] + auto full0 = dpf::make_output_buffer_for_full(k0); + auto full1 = dpf::make_output_buffer_for_full(k1); + auto fmemo0 = dpf::make_basic_full_memoizer(); + auto fmemo1 = dpf::make_basic_full_memoizer(); + auto f0 = dpf::eval_full(k0, full0, fmemo0); + auto f1 = dpf::eval_full(k1, full1, fmemo1); + //! [output-buffer-full] + it0 = std::begin(f0); + it1 = std::begin(f1); + for (int x = 0; x < 256; ++x, ++it0, ++it1) + { + std::uint64_t got = dpf::reconstruct(*it0, *it1); + std::uint64_t expect = (static_cast(x) == alpha) ? beta : 0; + if (got != expect) + { + std::cerr << "full output buffer\n"; + return 1; + } + } + + std::cout << beta << "\n"; return 0; -} \ No newline at end of file +} diff --git a/include/dpf.hpp b/include/dpf.hpp index fd6c9fc..4bfb64e 100644 --- a/include/dpf.hpp +++ b/include/dpf.hpp @@ -116,4 +116,6 @@ #include "dpf/uint256_t.hpp" +#include "dpf/interval.hpp" + #endif // LIBDPF_INCLUDE_DPF_HPP__ diff --git a/include/dpf/advice_bit_iterable.hpp b/include/dpf/advice_bit_iterable.hpp index d449459..a50da4c 100644 --- a/include/dpf/advice_bit_iterable.hpp +++ b/include/dpf/advice_bit_iterable.hpp @@ -143,6 +143,7 @@ class advice_bit_iterable_const_iterator using difference_type = std::ptrdiff_t; using node_type = typename iterator_traits::value_type; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr explicit advice_bit_iterable_const_iterator(const wrapped_type & it) noexcept @@ -213,28 +214,33 @@ class advice_bit_iterable_const_iterator return *this; } + HEDLEY_NO_THROW advice_bit_iterable_const_iterator operator+(std::size_t n) const noexcept { return advice_bit_iterable_const_iterator(it_ + n); } + HEDLEY_NO_THROW advice_bit_iterable_const_iterator & operator-=(std::size_t n) noexcept { it_ -= n; return *this; } + HEDLEY_NO_THROW advice_bit_iterable_const_iterator operator-(std::size_t n) const noexcept { return advice_bit_iterable_const_iterator(it_ - n); } + HEDLEY_NO_THROW difference_type operator-(advice_bit_iterable_const_iterator rhs) const noexcept { return it_ - rhs.it_; } + HEDLEY_NO_THROW reference operator[](std::size_t i) const noexcept { return bit(it_ + i); diff --git a/include/dpf/aligned_allocator.hpp b/include/dpf/aligned_allocator.hpp index bce624e..9885893 100644 --- a/include/dpf/aligned_allocator.hpp +++ b/include/dpf/aligned_allocator.hpp @@ -57,6 +57,7 @@ class aligned_allocator template struct deleter { + HEDLEY_NO_THROW constexpr void operator()(Pointer p) const noexcept { free(p); } }; public: @@ -125,6 +126,7 @@ class aligned_allocator /// @note This function returns the maximum number of elements that can /// be allocated, not the maximum allocation size in bytes /// @return The maximum supported allocation size. + HEDLEY_NO_THROW constexpr size_type max_size() const noexcept { return std::numeric_limits::max() / sizeof(value_type); diff --git a/include/dpf/beaver.hpp b/include/dpf/beaver.hpp index 2cc79f1..4883a3a 100644 --- a/include/dpf/beaver.hpp +++ b/include/dpf/beaver.hpp @@ -166,17 +166,22 @@ class wire friend class session; public: + HEDLEY_NO_THROW constexpr wire() noexcept = default; + HEDLEY_NO_THROW constexpr std::uint32_t id() const noexcept { return id_; } + HEDLEY_NO_THROW constexpr session * owner() const noexcept { return sess_; } + HEDLEY_NO_THROW friend bool operator==(wire a, wire b) noexcept { return a.sess_ == b.sess_ && a.id_ == b.id_; } + HEDLEY_NO_THROW friend bool operator!=(wire a, wire b) noexcept { return !(a == b); @@ -240,13 +245,16 @@ public: static constexpr std::uint32_t mono_role_base = 0x40000000u; static constexpr std::uint32_t dot_role_base = 0x80000000u; + HEDLEY_NO_THROW static constexpr std::uint32_t wire_role(std::uint32_t id) noexcept { return id; } + HEDLEY_NO_THROW static constexpr std::uint32_t mono_role(std::uint32_t i) noexcept { return mono_role_base + i; } + HEDLEY_NO_THROW static constexpr std::uint32_t dot_role(std::uint32_t gate) noexcept { return dot_role_base + gate; @@ -255,6 +263,7 @@ public: /// Fused within-polynomial λ combinations (Appendix E groupings). static constexpr std::uint32_t bundle_role_base = 0xC0000000u; + HEDLEY_NO_THROW static constexpr std::uint32_t bundle_role(std::uint32_t i) noexcept { return bundle_role_base + i; @@ -268,6 +277,7 @@ public: : lanes_(std::move(seed), window) { } + HEDLEY_NO_THROW const seed_type & seed() const noexcept { return lanes_.seed(); } Ring blind(std::uint32_t role, std::uint64_t index) const @@ -822,11 +832,13 @@ public: return wires_[check(w)].ready_round; } + HEDLEY_NO_THROW std::size_t wire_count() const noexcept { return wires_.size(); } /// Product shares beyond the per-wire blinds: subset monomials from /// `product` gates, plus one fused bundle per public-δ class in a /// polynomial (Appendix E). A lone mask is not counted. + HEDLEY_NO_THROW std::size_t monomial_count() const noexcept { return monos_.size() + bundles_.size(); diff --git a/include/dpf/bit.hpp b/include/dpf/bit.hpp index 1180258..65944b2 100644 --- a/include/dpf/bit.hpp +++ b/include/dpf/bit.hpp @@ -183,12 +183,14 @@ operator>>(std::basic_istream & is, dpf::bit & value) /// @} +HEDLEY_NO_THROW inline constexpr dpf::bit operator+(dpf::bit lhs, dpf::bit rhs) noexcept { return static_cast(static_cast(lhs) ^ static_cast(rhs)); } /// @brief GF(2) subtraction. Identical to `operator+`. +HEDLEY_NO_THROW inline constexpr dpf::bit operator-(dpf::bit lhs, dpf::bit rhs) noexcept { return lhs + rhs; @@ -216,6 +218,7 @@ struct packed_lane_bits template <> struct make_from_integral_value { + HEDLEY_NO_THROW constexpr dpf::bit operator()(bool val) const noexcept { return val ? dpf::bit::one : dpf::bit::zero; diff --git a/include/dpf/bit_array.hpp b/include/dpf/bit_array.hpp index d58707a..fe81f86 100644 --- a/include/dpf/bit_array.hpp +++ b/include/dpf/bit_array.hpp @@ -41,6 +41,7 @@ namespace detail { template +HEDLEY_NO_THROW constexpr void check_one_bit(Word mask) noexcept { #if defined(__GNUC__) || defined(__clang__) @@ -135,6 +136,7 @@ class bit_array_base inline constexpr bit_array_base(const bit_array_base &) = default; /// @brief default move constructor + HEDLEY_NO_THROW inline constexpr bit_array_base(bit_array_base &&) noexcept = default; /// @brief default destructor @@ -145,6 +147,7 @@ class bit_array_base bit_array_base & operator=(const bit_array_base &) = default; /// @brief defaulted move assignment + HEDLEY_NO_THROW inline constexpr bit_array_base & operator=(bit_array_base &&) noexcept = default; @@ -163,6 +166,7 @@ class bit_array_base /// @note Does not perform bounds checking; behaviour is undefined if /// `pos` is out of bounds /// @return `data()[pos]` + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr word_type data(size_type pos) const noexcept { @@ -187,6 +191,7 @@ class bit_array_base return data()[pos]; } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr size_type data_length() const noexcept { return derived_from_this()->data_length(); } @@ -261,6 +266,7 @@ class bit_array_base /// @{ /// @returns iterator to the first element /// @complexity `O(1)` + HEDLEY_NO_THROW constexpr iterator begin() noexcept { auto *p = data(); @@ -269,6 +275,7 @@ class bit_array_base } /// @returns iterator to the first element /// @complexity `O(1)` + HEDLEY_NO_THROW constexpr const_iterator begin() const noexcept { auto *p = data(); @@ -277,6 +284,7 @@ class bit_array_base } /// @returns iterator to the first element /// @complexity `O(1)` + HEDLEY_NO_THROW constexpr const_iterator cbegin() const noexcept { return begin(); @@ -287,6 +295,7 @@ class bit_array_base /// @{ /// @returns iterator to the element following the last element /// @complexity `O(1)` + HEDLEY_NO_THROW constexpr iterator end() noexcept { auto *p = data(); @@ -296,6 +305,7 @@ class bit_array_base } /// @returns iterator to the element following the last element /// @complexity `O(1)` + HEDLEY_NO_THROW constexpr const_iterator end() const noexcept { auto *p = data(); @@ -305,6 +315,7 @@ class bit_array_base } /// @returns iterator to the element following the last element /// @complexity `O(1)` + HEDLEY_NO_THROW constexpr const_iterator cend() const noexcept { return end(); @@ -326,6 +337,7 @@ class bit_array_base /// @details checks if all bits are set to `true` /// @return `true` if all of the bits are set to `true`, otherwise `false` /// @complexity `O(size())` + HEDLEY_NO_THROW bool all() const noexcept { if (size() == 0) return true; @@ -351,6 +363,7 @@ class bit_array_base /// `true`, otherwise `false` /// @complexity `O(last-first)` template + HEDLEY_NO_THROW bool all(Iterator first, Iterator last) const noexcept { bool ok = true; @@ -365,6 +378,7 @@ class bit_array_base /// @details checks if any bits are set to `true` /// @return `true` if any of the bits are set to `true`, otherwise `false` /// @complexity `O(size())` + HEDLEY_NO_THROW bool any() const noexcept { const size_type n = data_length(); @@ -386,6 +400,7 @@ class bit_array_base /// `true`, otherwise `false` /// @complexity `O(last-first)` template + HEDLEY_NO_THROW bool any(Iterator first, Iterator last) const noexcept { bool found = false; @@ -400,6 +415,7 @@ class bit_array_base /// @details checks if none of the bits are set to `true` /// @return `true` if none of the bits are set to `true`, otherwise `false` /// @complexity `O(size())` + HEDLEY_NO_THROW bool none() const noexcept { return !any(); @@ -412,6 +428,7 @@ class bit_array_base /// `true`, otherwise `false` /// @complexity `O(last-first)` template + HEDLEY_NO_THROW bool none(Iterator first, Iterator last) const noexcept { return !any(first, last); @@ -423,6 +440,7 @@ class bit_array_base /// @details counts the number of bits that are set to `true` /// @returns the number of bits set to `true` /// @complexity `O(size())` + HEDLEY_NO_THROW size_type count() const noexcept { const size_type n = data_length(); @@ -443,6 +461,7 @@ class bit_array_base /// @return the number of bits in the given range that are set to `true` /// @complexity `O(last-first)` template + HEDLEY_NO_THROW size_type count(Iterator first, Iterator last) const noexcept { size_type sum = 0; @@ -460,6 +479,7 @@ class bit_array_base /// @details counts the parity of all stored bits /// @returns the parity of all stored bits /// @complexity `O(size())` + HEDLEY_NO_THROW size_type parity() const noexcept { const size_type n = data_length(); @@ -481,6 +501,7 @@ class bit_array_base /// @return the parity of all bits in the given range /// @complexity `O(last-first)` template + HEDLEY_NO_THROW size_type parity(Iterator first, Iterator last) const noexcept { word_type x = word_type{0}; @@ -496,6 +517,7 @@ class bit_array_base /// @brief returns the number of bits /// @returns number of bits that the `bit_array_base` holds /// @complexity `O(1)` + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr size_type size() const noexcept @@ -507,6 +529,7 @@ class bit_array_base /// @{ /// @brief sets all bits to `true` /// @complexity `O(size())` + HEDLEY_NO_THROW constexpr void set() noexcept { const size_type n = data_length(); @@ -551,6 +574,7 @@ class bit_array_base /// @{ /// @brief sets all bits to `false' /// @complexity `O(size())` + HEDLEY_NO_THROW constexpr void unset() noexcept { word_type *p = data(); @@ -578,6 +602,7 @@ class bit_array_base /// @{ /// @brief flips all bits (like `operator~`, but in-place) /// @complexity `O(size())` + HEDLEY_NO_THROW constexpr void flip() noexcept { const size_type n = data_length(); @@ -711,6 +736,7 @@ class bit_array_base return static_cast(static_cast(lhs) ^ static_cast(rhs)); } + HEDLEY_NO_THROW friend constexpr dpf::bit operator+(bit_reference lhs, bit_reference rhs) noexcept { return static_cast(static_cast(lhs) ^ static_cast(rhs)); @@ -820,6 +846,7 @@ class bit_array_base /// @brief Exchange the bits named by two proxies, including temporaries /// returned from `operator[]` and `operator*`. + HEDLEY_NO_THROW friend constexpr void swap(bit_reference a, bit_reference b) noexcept { const bool tmp = static_cast(a); @@ -870,6 +897,7 @@ class bit_array_base static constexpr word_type sentinel = ~word_type(0); /// @brief Low `n` bits set. `n == 0` yields 0. `n >= bits_per_word` yields all ones. + HEDLEY_NO_THROW static constexpr word_type low_bits_mask(size_type n) noexcept { if (n == 0) return word_type{0}; @@ -878,18 +906,21 @@ class bit_array_base static_cast(~word_type{0}) >> (bits_per_word - n)); } + HEDLEY_NO_THROW static constexpr size_type pop(word_type w) noexcept { return static_cast(utils::popcount(w)); } /// @brief Bits strictly below the single set bit in `mask`. + HEDLEY_NO_THROW static constexpr word_type bits_below(word_type mask) noexcept { return static_cast(mask - word_type{1}); } /// @brief Bits at and above the single set bit in `mask`. + HEDLEY_NO_THROW static constexpr word_type bits_at_and_above(word_type mask) noexcept { return static_cast(~bits_below(mask)); @@ -999,6 +1030,7 @@ class bit_iterator_base word_type mask_; /// @brief Singular iterator. Comparable, not dereferenceable. + HEDLEY_NO_THROW inline constexpr bit_iterator_base() noexcept : word_ptr_{nullptr}, mask_{lsb} @@ -1125,11 +1157,14 @@ class bit_iterator final using word_pointer = typename bit_array_base::word_pointer; /// @brief Singular iterator. Comparable, not dereferenceable. + HEDLEY_NO_THROW constexpr bit_iterator() noexcept = default; + HEDLEY_NO_THROW inline constexpr bit_iterator(const bit_iterator &) noexcept = default; + HEDLEY_NO_THROW inline constexpr bit_iterator(bit_iterator &&) noexcept = default; HEDLEY_NO_THROW @@ -1218,6 +1253,7 @@ class bit_iterator final return tmp -= amt; } + HEDLEY_NO_THROW friend constexpr iterator operator+(difference_type amt, iterator it) noexcept { return it + amt; @@ -1257,6 +1293,7 @@ class const_bit_iterator final using const_word_pointer = typename bit_array_base::const_word_pointer; /// @brief Singular iterator. Comparable, not dereferenceable. + HEDLEY_NO_THROW constexpr const_bit_iterator() noexcept = default; HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW @@ -1349,6 +1386,7 @@ class const_bit_iterator final return *this; } + HEDLEY_NO_THROW inline constexpr const_iterator & operator-=(difference_type amt) noexcept { @@ -1356,6 +1394,7 @@ class const_bit_iterator final return *this; } + HEDLEY_NO_THROW inline constexpr const_iterator operator+(difference_type amt) const noexcept { @@ -1363,6 +1402,7 @@ class const_bit_iterator final return tmp += amt; } + HEDLEY_NO_THROW inline constexpr const_iterator operator-(difference_type amt) const noexcept { @@ -1370,12 +1410,14 @@ class const_bit_iterator final return tmp -= amt; } + HEDLEY_NO_THROW friend constexpr const_iterator operator+(difference_type amt, const_iterator it) noexcept { return it + amt; } + HEDLEY_NO_THROW inline constexpr const_reference operator[](difference_type i) const noexcept { @@ -1402,10 +1444,14 @@ class alignas(utils::max_align_v) static_bit_array final /// `size()` bits static constexpr size_type data_length_ = utils::quotient_ceiling(Nbits, bits_per_word); public: + HEDLEY_NO_THROW constexpr static_bit_array(static_bit_array &&) noexcept = default; + HEDLEY_NO_THROW constexpr static_bit_array(const static_bit_array &) noexcept = default; ~static_bit_array() = default; + HEDLEY_NO_THROW constexpr static_bit_array & operator=(static_bit_array &&) noexcept = default; + HEDLEY_NO_THROW constexpr static_bit_array & operator=(const static_bit_array &) noexcept = default; /// @brief constructs a zeroed `static_bit_array` that holds `Nbits` bits inline constexpr static_bit_array() @@ -1459,6 +1505,7 @@ class alignas(utils::max_align_v) static_bit_array final /// @brief returns the number of bits /// @returns number of bits that the `static_bit_array` holds /// @complexity `O(1)` + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr size_type size() const noexcept @@ -1509,6 +1556,7 @@ class dynamic_bit_array std::copy_n(other.data_.get(), data_length_ + 1, data_.get()); } + HEDLEY_NO_THROW dynamic_bit_array(dynamic_bit_array && other) noexcept : num_bits_{std::exchange(other.num_bits_, 0)}, data_length_{std::exchange(other.data_length_, 0)}, @@ -1525,6 +1573,7 @@ class dynamic_bit_array return *this; } + HEDLEY_NO_THROW dynamic_bit_array & operator=(dynamic_bit_array && other) noexcept { if (this != &other) @@ -1537,6 +1586,7 @@ class dynamic_bit_array return *this; } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE ~dynamic_bit_array() noexcept { @@ -1595,6 +1645,7 @@ class dynamic_bit_array /// @brief returns the number of bits /// @returns number of bits that the `dynamic_bit_array` holds /// @complexity `O(1)` + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr size_type size() const noexcept @@ -1603,6 +1654,8 @@ class dynamic_bit_array } private: + /// Store zeros through `volatile` so the wipe is not deleted as a dead store. + HEDLEY_NO_THROW void wipe() noexcept { if (!data_) return; @@ -1620,6 +1673,7 @@ class dynamic_bit_array /// @brief template +HEDLEY_NO_THROW inline constexpr void swap(typename dynamic_bit_array::reference lhs, typename dynamic_bit_array::reference rhs) noexcept { @@ -1630,6 +1684,7 @@ inline constexpr void swap(typename dynamic_bit_array::reference lhs, template +HEDLEY_NO_THROW inline constexpr void swap(typename static_bit_array::reference lhs, typename static_bit_array::reference rhs) noexcept { diff --git a/include/dpf/bitstring.hpp b/include/dpf/bitstring.hpp index 3355c6c..80b929b 100644 --- a/include/dpf/bitstring.hpp +++ b/include/dpf/bitstring.hpp @@ -187,7 +187,9 @@ class bitstring : public bit_array_base, WordT> /// @} + HEDLEY_NO_THROW bitstring & operator=(const bitstring &) noexcept = default; + HEDLEY_NO_THROW bitstring & operator=(bitstring &&) noexcept = default; ~bitstring() = default; @@ -387,6 +389,7 @@ class bitstring : public bit_array_base, WordT> using base::flip; /// @brief Flips every defined bit and clears bits above `Nbits`. + HEDLEY_NO_THROW constexpr void flip() noexcept { base::flip(); @@ -468,6 +471,7 @@ class bitstring : public bit_array_base, WordT> /// @note Does not perform bounds checking; behaviour is undefined if /// `pos` is out of bounds /// @return `data()[pos]` + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr word_type data(size_type pos) const noexcept { @@ -480,6 +484,7 @@ class bitstring : public bit_array_base, WordT> /// @note Does not perform bounds checking; behaviour is undefined if /// `pos` is out of bounds /// @return `data()[pos]` + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr word_type & data(size_type pos) noexcept { @@ -488,6 +493,7 @@ class bitstring : public bit_array_base, WordT> /// @brief length of the underlying data array /// @return the number of elements in the underlying array + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr size_type data_length() const noexcept { @@ -497,6 +503,7 @@ class bitstring : public bit_array_base, WordT> /// @brief returns the number of bits /// @returns number of bits that the `bitstring` holds /// @complexity `O(1)` + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr size_type size() const noexcept @@ -510,6 +517,7 @@ class bitstring : public bit_array_base, WordT> std::array data_{}; /// @brief Mask of the bits that belong to this string in the high word. + HEDLEY_NO_THROW static constexpr word_type defined_high_mask() noexcept { constexpr auto rem = Nbits % bits_per_word; @@ -519,6 +527,7 @@ class bitstring : public bit_array_base, WordT> return static_cast((word_type{1} << rem) - word_type{1}); } + HEDLEY_NO_THROW constexpr void clear_unused() noexcept { if constexpr (Nbits % bits_per_word != 0) @@ -630,6 +639,7 @@ struct countl_zero_symmetric_difference> { using T = dpf::bitstring; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr @@ -706,6 +716,7 @@ struct make_from_integral_value> using integral_type = std::conditional_t, simde_uint128, T_integral_type>; static constexpr auto mod = utils::mod_pow_2{}; static constexpr auto bits_per_last_word = Nbits % T::bits_per_word; + HEDLEY_NO_THROW constexpr dpf::bitstring operator()(integral_type val) const noexcept { dpf::bitstring ret; @@ -731,6 +742,7 @@ struct mod_pow_2> using T = dpf::bitstring; static constexpr auto to_int = to_integral_type{}; static constexpr auto mod = mod_pow_2{}; + HEDLEY_NO_THROW std::size_t operator()(T val, std::size_t n) const noexcept { return mod(to_int(val), n); @@ -1141,14 +1153,23 @@ class numeric_limits> = std::numeric_limits::integral_type>::traps; static constexpr bool tinyness_before = false; + HEDLEY_NO_THROW static constexpr dpf::bitstring min() noexcept { return dpf::bitstring{}; } + HEDLEY_NO_THROW static constexpr dpf::bitstring lowest() noexcept { return dpf::bitstring{}; } + HEDLEY_NO_THROW static constexpr dpf::bitstring max() noexcept { return ~dpf::bitstring{}; } + HEDLEY_NO_THROW static constexpr dpf::bitstring epsilon() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::bitstring round_error() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::bitstring infinity() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::bitstring quiet_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::bitstring signaling_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::bitstring denorm_min() noexcept { return 0; } }; diff --git a/include/dpf/blocked_dcf.hpp b/include/dpf/blocked_dcf.hpp new file mode 100644 index 0000000..39f45a5 --- /dev/null +++ b/include/dpf/blocked_dcf.hpp @@ -0,0 +1,475 @@ +/// @file dpf/blocked_dcf.hpp +/// @brief Blocked-checkpoint comparison: one ring word per block of levels. +/// @details The seed spine stays dense. Ring words are published only at a +/// public checkpoint schedule. Point eval expands parked siblings +/// up to the next checkpoint; a full-domain memoizer already holds +/// those nodes. `q` tail bits, when the comparison sets the key +/// depth, are a residual table on the node at height `h`. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + +#ifndef LIBDPF_INCLUDE_DPF_BLOCKED_DCF_HPP__ +#define LIBDPF_INCLUDE_DPF_BLOCKED_DCF_HPP__ + +#include +#include +#include +#include +#include +#include + +#include "hedley/hedley.h" + +#include "dpf/dcf.hpp" +#include "dpf/path_memoizer.hpp" +#include "dpf/twiddle.hpp" +#include "dpf/utils.hpp" + +namespace dpf +{ +namespace detail +{ +namespace blocked +{ + +template +struct schedule +{ + static constexpr std::size_t count = + (B == 0 || H == 0) ? 0 : (H + B - 1) / B; + + static constexpr auto depths = [] { + std::array cs{}; + if constexpr (count == 0) + return cs; + const std::size_t base = H / count; + const std::size_t extra = H % count; + std::size_t acc = 0; + for (std::size_t i = 0; i < count; ++i) + { + acc += base + (i < extra ? 1 : 0); + cs[i] = acc; + } + return cs; + }(); + + HEDLEY_CONST + HEDLEY_NO_THROW + static constexpr bool contains(std::size_t depth) noexcept + { + for (std::size_t i = 0; i < count; ++i) + { + if (depths[i] == depth) + return true; + } + return false; + } + + HEDLEY_CONST + HEDLEY_NO_THROW + static constexpr std::size_t index(std::size_t depth) noexcept + { + for (std::size_t i = 0; i < count; ++i) + { + if (depths[i] == depth) + return i; + } + return static_cast(-1); + } +}; + +template +HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW +uint64_t rho_of(const Node & node, uint64_t mask) noexcept +{ + auto kids = PRG::eval01(dpf::unset_lo_2bits(node)); + return dcf_impl::convert_node(kids[0], mask); +} + +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr int control_sign(uint8_t t0, uint8_t t1) noexcept +{ + return static_cast(t0) - static_cast(t1); +} + +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t mul_sgn(int sgn, uint64_t v, uint64_t mask) noexcept +{ + if (sgn > 0) + return v & mask; + if (sgn < 0) + return dcf_impl::neg_m(v, mask); + return 0; +} + +/// Group element `sgn` (`+1`, `-1`, or `0`) used as an `assign_cmp` coefficient. +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t sgn_coeff(int sgn, uint64_t mask) noexcept +{ + if (sgn > 0) + return 1ULL & mask; + if (sgn < 0) + return dcf_impl::neg_m(1ULL, mask); + return 0; +} + +template +HEDLEY_NO_THROW +uint64_t checkpoint_word(const Node & n0, const Node & n1, uint64_t beta, + uint64_t mask) noexcept +{ + const int sgn = control_sign( + static_cast(dpf::get_lo_bit(n0)), + static_cast(dpf::get_lo_bit(n1))); + const uint64_t r0 = rho_of(n0, mask); + const uint64_t r1 = rho_of(n1, mask); + const uint64_t inner = + (beta + dcf_impl::neg_m(r0, mask) + r1) & mask; + return mul_sgn(sgn, inner, mask); +} + +template +HEDLEY_NO_THROW +uint64_t checkpoint_coeff(const Node & n0, const Node & n1, + uint64_t mask) noexcept +{ + return sgn_coeff(control_sign( + static_cast(dpf::get_lo_bit(n0)), + static_cast(dpf::get_lo_bit(n1))), mask); +} + +template +HEDLEY_NON_NULL(4) +HEDLEY_NO_THROW +void suffix_masks(const Node & seed, std::size_t q, uint64_t mask, + uint64_t * out) noexcept +{ + Node cur[4]{}; + Node nxt[8]{}; + cur[0] = seed; + std::size_t n = 1; + for (std::size_t lvl = 0; lvl < q; ++lvl) + { + std::size_t m = 0; + for (std::size_t i = 0; i < n; ++i) + { + auto kids = PRG::eval01(dpf::unset_lo_2bits(cur[i])); + nxt[m++] = kids[0]; + nxt[m++] = kids[1]; + } + for (std::size_t i = 0; i < m; ++i) + cur[i] = nxt[i]; + n = m; + } + for (std::size_t i = 0; i < n; ++i) + out[i] = dcf_impl::convert_node(cur[i], mask); +} + +template +HEDLEY_NON_NULL(8) +HEDLEY_NO_THROW +void tail_words(const Node & n0, const Node & n1, uint64_t beta, uint64_t mask, + bool include_eq, uint64_t suffix, std::size_t q, uint64_t * words, + uint64_t * coeffs) noexcept +{ + uint64_t u0[4]{}; + uint64_t u1[4]{}; + suffix_masks(n0, q, mask, u0); + suffix_masks(n1, q, mask, u1); + const int sgn = control_sign( + static_cast(dpf::get_lo_bit(n0)), + static_cast(dpf::get_lo_bit(n1))); + const uint64_t coeff = sgn_coeff(sgn, mask); + const std::size_t n = std::size_t{1} << q; + for (std::size_t z = 0; z < n; ++z) + { + const bool pred = include_eq + ? (z <= suffix) + : (z < suffix); + const uint64_t inner = + ((pred ? beta : 0ULL) + dcf_impl::neg_m(u0[z], mask) + u1[z]) & mask; + words[z] = mul_sgn(sgn, inner, mask); + if (coeffs != nullptr) + coeffs[z] = pred ? coeff : 0ULL; + } +} + +template +HEDLEY_NO_THROW +uint64_t add_membership(uint64_t acc, const typename KeyT::interior_node & node, + uint64_t word, uint64_t mask, int party) noexcept +{ + using prg = typename KeyT::interior_prg; + const uint8_t t = static_cast(dpf::get_lo_bit(node)); + const uint64_t y = + (rho_of(node, mask) + (t ? word : 0ULL)) & mask; + return (acc + (party ? dcf_impl::neg_m(y, mask) : y)) & mask; +} + +template +uint64_t add_frontier(uint64_t acc, const typename KeyT::interior_node & seed, + std::size_t from_depth, std::size_t to_depth, const KeyT & dpf, uint64_t word, + uint64_t mask, int party) +{ + using node = typename KeyT::interior_node; + std::vector cur; + std::vector nxt; + cur.push_back(seed); + for (std::size_t lvl = from_depth; lvl < to_depth; ++lvl) + { + nxt.clear(); + nxt.reserve(cur.size() * 2); + const node cw0 = dpf.correction_word(lvl, false); + const node cw1 = dpf.correction_word(lvl, true); + for (const node & fs : cur) + { + auto kids = KeyT::traverse_interior01(fs, cw0, cw1); + nxt.push_back(kids[0]); + nxt.push_back(kids[1]); + } + cur.swap(nxt); + } + for (const node & fs : cur) + acc = add_membership(acc, fs, word, mask, party); + return acc; +} + +template +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t query_suffix(InputT tx, std::size_t nbits, std::size_t q) noexcept +{ + if (q == 0) + return 0; + uint64_t z = 0; + auto bit_mask = KeyT::msb_mask >> (nbits - q); + for (std::size_t j = 0; j < q; ++j, bit_mask >>= 1) + z = (z << 1) | static_cast(!!(bit_mask & tx)); + return z; +} + +template +HEDLEY_NO_THROW +uint64_t finish_share(const KeyT & dpf, uint64_t suffix, uint64_t acc, + const typename KeyT::interior_node & at_h, int party) noexcept +{ + using namespace dcf_impl; + using prg = typename KeyT::interior_prg; + const auto & ch = dpf.cmp(); + const uint64_t mask = ch.mask; + constexpr std::size_t q = KeyT::cmp_q; + constexpr std::size_t h = KeyT::cmp_h; + if constexpr (q == 0) + { + if (ch.include_eq) + { + constexpr auto wi = schedule::index(h); + acc = add_membership(acc, at_h, dpf.value_cw(wi), mask, party); + } + (void)suffix; + } + else + { + const uint64_t z = suffix; + uint64_t u[4]{}; + suffix_masks(at_h, q, mask, u); + const uint8_t t = static_cast(dpf::get_lo_bit(at_h)); + const uint64_t y = (u[z] + (t ? dpf.tail_cw(z) : 0ULL)) & mask; + acc = (acc + (party ? neg_m(y, mask) : y)) & mask; + } + if (ch.eval_as_ge) + acc = neg_m(acc, mask); + const uint64_t add = [&]() -> uint64_t { + if constexpr (is_party_key_v) + return dpf.cmp_addend().raw(); + else + return dpf.cmp_addend(); + }(); + return (acc + add) & mask; +} + +template +uint64_t eval_share(const KeyT & dpf, InputT tx, PathMemoizer & path) +{ + using node = typename KeyT::interior_node; + const auto & ch = dpf.cmp(); + const uint64_t mask = ch.mask; + const uint64_t add = [&]() -> uint64_t { + if constexpr (is_party_key_v) + return dpf.cmp_addend().raw(); + else + return dpf.cmp_addend(); + }(); + if (ch.trivial == cmp_trivial::always_true + || ch.trivial == cmp_trivial::always_false) + return add & mask; + + constexpr std::size_t h = KeyT::cmp_h; + using sched = schedule; + const std::size_t nbits = static_cast(ch.nbits); + const int party = dpf::get_lo_bit(dpf.root()) ? 1 : 0; + + dpf::detail::ensure_level(dpf, tx, path, h); + + struct parked + { + node seed; + std::size_t depth; + }; + parked pend[128]; + std::size_t npend = 0; + + uint64_t acc = 0; + auto bit_mask = KeyT::msb_mask; + for (std::size_t level = 0; level < h; ++level, bit_mask >>= 1) + { + const bool xi = !!(bit_mask & tx); + const node & parent = path[level]; + const node right = KeyT::traverse_interior(parent, + dpf.correction_word(level, true), true); + if (!xi) + { + pend[npend].seed = right; + pend[npend].depth = level + 1; + ++npend; + } + const std::size_t c = level + 1; + if (sched::contains(c)) + { + const uint64_t word = dpf.value_cw(sched::index(c)); + for (std::size_t p = 0; p < npend; ++p) + { + acc = add_frontier(acc, pend[p].seed, pend[p].depth, c, + dpf, word, mask, party); + } + npend = 0; + } + } + const uint64_t suffix = query_suffix(tx, nbits, KeyT::cmp_q); + return finish_share(dpf, suffix, acc, path[h], party); +} + +template +HEDLEY_NO_THROW +bool memo_has(const Memo & memo, Integral prefix, std::size_t depth, + Integral from_lane, Integral to_excl) noexcept +{ + const auto shift = KeyT::cmp_depth - depth; + const auto from_p = from_lane >> shift; + if (prefix < from_p) + return false; + const auto idx = static_cast(prefix - from_p); + const auto count = memo.get_nodes_at_level(depth, from_lane, to_excl); + return idx < count; +} + +template +const typename KeyT::interior_node & memo_node(const Memo & memo, Integral prefix, + std::size_t depth, Integral from_lane) +{ + const auto shift = KeyT::cmp_depth - depth; + const auto from_p = from_lane >> shift; + const auto idx = static_cast(prefix - from_p); + return memo[depth][idx]; +} + +template +uint64_t eval_share_memo(const KeyT & dpf, Integral lane, + Integral from_lane, Integral to_excl, const Memo & memo) +{ + using node = typename KeyT::interior_node; + const auto & ch = dpf.cmp(); + const uint64_t mask = ch.mask; + const uint64_t add = [&]() -> uint64_t { + if constexpr (is_party_key_v) + return dpf.cmp_addend().raw(); + else + return dpf.cmp_addend(); + }(); + if (ch.trivial == cmp_trivial::always_true + || ch.trivial == cmp_trivial::always_false) + return add & mask; + + constexpr std::size_t h = KeyT::cmp_h; + using sched = schedule; + const int party = dpf::get_lo_bit(dpf.root()) ? 1 : 0; + + struct parked + { + node seed; + Integral prefix; + std::size_t depth; + }; + parked pend[128]; + std::size_t npend = 0; + + uint64_t acc = 0; + Integral path_pref = 0; + const std::size_t nbits = static_cast(ch.nbits); + for (std::size_t level = 0; level < h; ++level) + { + const bool xi = ((lane >> (nbits - 1 - level)) & Integral{1}) != 0; + const node & parent = memo_node(memo, path_pref, level, from_lane); + const Integral sib = static_cast((path_pref << 1) | Integral{1}); + if (!xi) + { + const node right = KeyT::traverse_interior(parent, + dpf.correction_word(level, true), true); + pend[npend].seed = right; + pend[npend].prefix = sib; + pend[npend].depth = level + 1; + ++npend; + } + path_pref = static_cast((path_pref << 1) | Integral{xi ? 1 : 0}); + const std::size_t c = level + 1; + if (!sched::contains(c)) + continue; + const uint64_t word = dpf.value_cw(sched::index(c)); + for (std::size_t p = 0; p < npend; ++p) + { + const std::size_t extra = c - pend[p].depth; + const Integral leftmost = + static_cast(pend[p].prefix << extra); + const Integral rightmost = static_cast( + leftmost + static_cast((Integral{1} << extra) - 1)); + const bool covered = + memo_has(memo, leftmost, c, from_lane, to_excl) + && memo_has(memo, rightmost, c, from_lane, to_excl); + if (covered && extra < 16) + { + const Integral nleaf = static_cast(Integral{1} << extra); + for (Integral k = 0; k < nleaf; ++k) + { + const auto pref = static_cast(leftmost + k); + acc = add_membership(acc, + memo_node(memo, pref, c, from_lane), word, mask, + party); + } + } + else + { + acc = add_frontier(acc, pend[p].seed, pend[p].depth, c, + dpf, word, mask, party); + } + } + npend = 0; + } + const node & at_h = memo_node(memo, path_pref, h, from_lane); + const uint64_t suffix = static_cast(lane) + & (KeyT::cmp_q == 0 ? 0ULL : ((1ULL << KeyT::cmp_q) - 1ULL)); + return finish_share(dpf, suffix, acc, at_h, party); +} + +} // namespace blocked +} // namespace detail +} // namespace dpf + +#endif // LIBDPF_INCLUDE_DPF_BLOCKED_DCF_HPP__ diff --git a/include/dpf/buffered_prg.hpp b/include/dpf/buffered_prg.hpp index c66c5c6..7e61cf7 100644 --- a/include/dpf/buffered_prg.hpp +++ b/include/dpf/buffered_prg.hpp @@ -39,6 +39,7 @@ namespace detail { template +HEDLEY_NO_THROW typename PRG::block_type mask_master(typename PRG::block_type master) noexcept { unsigned char raw[sizeof(master)]; @@ -140,6 +141,7 @@ struct buffered_slot return lane_codec::at(seed_, index); } + HEDLEY_NO_THROW std::uint64_t sampled() const noexcept { return absolute_pos_; } private: @@ -165,7 +167,13 @@ typename PRG::block_type sample_master_seed() return dpf::uniform_sample(); } -/// Fixed lanes. Lane `I` is `PRG::eval(master, I)`. +/// Forward cursor over one PRG stream per value type. +/// +/// `get()` and `fill()` consume the cursor. `at(index)` reads an +/// absolute index and leaves the cursor where it is. `sampled()` reports +/// how far `get` and `fill` have advanced. `per_stream_buffer_elems` is at +/// least 1. +/// @snippet evaluation/buffered_prg.cpp buffered-prg template class buffered_prg { @@ -184,6 +192,7 @@ public: buffers_(make_buffers(per_stream_buffer_elems)) { } + HEDLEY_NO_THROW const seed_type & seed() const noexcept { return seed_; } template @@ -208,6 +217,7 @@ public: } template + HEDLEY_NO_THROW std::uint64_t sampled() const noexcept { static_assert(I < stream_count, "stream index out of range"); @@ -238,9 +248,12 @@ private: template using aes_buffered_prg = buffered_prg; -/// Dynamic lanes of one value type. `value_at(role, index)` and -/// `mask_at(role, index)` are independent of call order. A window cache -/// refills from the requested index. +/// Seekable value and mask streams for a runtime set of roles. +/// +/// `value_at(role, index)` and `mask_at(role, index)` are independent of +/// call order. A repeated index returns the same element. `window` is at +/// least 1. +/// @snippet evaluation/buffered_prg.cpp lane-table template class lane_table { @@ -267,6 +280,7 @@ public: lane_table(lane_table &&) = default; lane_table & operator=(lane_table &&) = default; + HEDLEY_NO_THROW const seed_type & seed() const noexcept { return seed_; } T value_at(std::uint32_t role, std::uint64_t index) const diff --git a/include/dpf/dcf.hpp b/include/dpf/dcf.hpp index e12c717..9aae0b5 100644 --- a/include/dpf/dcf.hpp +++ b/include/dpf/dcf.hpp @@ -26,15 +26,61 @@ namespace dpf { +template +struct ic_pack; + +template +struct is_ic_pack : std::false_type {}; + +template +struct is_ic_pack> : std::true_type {}; + +template +inline constexpr bool no_ic_pack_v = + (!is_ic_pack>::value && ...); + /// Comparison kind for the optional DCF channel on a key. +/// `lt`/`leq`/`gt`/`geq` are the comparison predicates. The later kinds are +/// path paints: one constant on each sibling subtree of the secret point, +/// evaluated by the same value-correction walk. enum class cmp_kind : uint8_t { lt = 0, leq = 1, gt = 2, - geq = 3 + geq = 3, + lcp = 4, // common-prefix length + prefix = 5, // matched prefix, in the lane's high bits + mask = 6, // high-bit mask of that length + one_hot = 7, // 2^{length} (0 when the bit does not fit) + break_bit = 8, // secret bit at the first difference + prefix_with_length = 9, // (low-aligned prefix << length_bits) | length + paint = 10 // caller-supplied unit plant }; +/// True for the path-paint kinds. Comparisons stay `lt`/`leq`/`gt`/`geq`. +HEDLEY_NO_THROW +inline constexpr bool is_paint_kind(cmp_kind kind) noexcept +{ + switch (kind) + { + case cmp_kind::lcp: + case cmp_kind::prefix: + case cmp_kind::mask: + case cmp_kind::one_hot: + case cmp_kind::break_bit: + case cmp_kind::prefix_with_length: + case cmp_kind::paint: + return true; + default: + return false; + } +} + +/// Unit plant for `path_paint`. `prefix` is the in-lane matched prefix. +using paint_callback = uint64_t (*)(std::size_t matched, uint64_t prefix, + bool leaf, const void * ctx); + enum class cmp_trivial : uint8_t { none = 0, @@ -48,6 +94,7 @@ namespace dcf_impl { template +HEDLEY_NO_THROW Beta default_false() noexcept { if constexpr (std::is_same_v) @@ -57,6 +104,7 @@ Beta default_false() noexcept } template +HEDLEY_NO_THROW uint64_t beta_delta_u64(const Beta & if_true, const Beta & if_false, uint64_t mask) noexcept { @@ -79,6 +127,7 @@ uint64_t beta_delta_u64(const Beta & if_true, const Beta & if_false, } template +HEDLEY_NO_THROW uint64_t beta_to_u64_simple(const Beta & beta, uint64_t mask) noexcept { if constexpr (std::is_same_v) @@ -88,6 +137,7 @@ uint64_t beta_to_u64_simple(const Beta & beta, uint64_t mask) noexcept } template +HEDLEY_NO_THROW Beta sub_beta(const Beta & a, const Beta & b) noexcept { if constexpr (std::is_same_v) @@ -99,6 +149,7 @@ Beta sub_beta(const Beta & a, const Beta & b) noexcept } template +HEDLEY_NO_THROW Beta u64_to_beta(uint64_t v) noexcept { if constexpr (std::is_same_v) @@ -107,7 +158,10 @@ Beta u64_to_beta(uint64_t v) noexcept return static_cast(v); } -inline uint64_t default_mask_for_bits(std::size_t out_bits) noexcept +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t default_mask_for_bits(std::size_t out_bits) noexcept { if (out_bits >= 64) return ~0ULL; @@ -116,14 +170,18 @@ inline uint64_t default_mask_for_bits(std::size_t out_bits) noexcept return (1ULL << out_bits) - 1ULL; } +HEDLEY_CONST +HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE -uint64_t neg_m(uint64_t x, uint64_t mask) noexcept +constexpr uint64_t neg_m(uint64_t x, uint64_t mask) noexcept { return (0ULL - x) & mask; } +HEDLEY_CONST +HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE -uint64_t sgn_m(uint8_t t1, uint64_t x, uint64_t mask) noexcept +constexpr uint64_t sgn_m(uint8_t t1, uint64_t x, uint64_t mask) noexcept { return t1 ? neg_m(x, mask) : x; } @@ -143,6 +201,7 @@ uint64_t convert_node(simde__m128i n, uint64_t mask) noexcept /// same block source so their keys stay byte-identical (matched tapes). template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW uint64_t sample_addend_blind(uint64_t mask, BlockSampler && sample) noexcept { return convert_node(dpf::unset_lo_2bits(sample()), mask); @@ -150,6 +209,7 @@ uint64_t sample_addend_blind(uint64_t mask, BlockSampler && sample) noexcept /// One level of value CW on GGM children. Updates running `Va`. /// `ai` is the keep-path bit of the (effective) threshold. +HEDLEY_NO_THROW inline uint64_t make_value_cw(simde__m128i c0L, simde__m128i c0R, simde__m128i c1L, simde__m128i c1R, uint8_t t0, uint8_t t1, int ai, uint64_t & Va, uint64_t beta, uint64_t mask) noexcept @@ -179,8 +239,127 @@ inline uint64_t make_value_cw(simde__m128i c0L, simde__m128i c0R, return vcw; } +/// Same recurrence as `make_value_cw`, planting `plant` on the lose child +/// in both directions. `plant == 0` leaves the correction unchanged. +HEDLEY_NO_THROW +inline uint64_t make_value_cw_planted(simde__m128i c0L, simde__m128i c0R, + simde__m128i c1L, simde__m128i c1R, uint8_t t0, uint8_t t1, int ai, + uint64_t & Va, uint64_t plant, uint64_t mask) noexcept +{ + (void)t0; + uint64_t v0K, v1K, v0Lo, v1Lo; + if (ai == 0) + { + v0K = convert_node(c0L, mask); + v1K = convert_node(c1L, mask); + v0Lo = convert_node(c0R, mask); + v1Lo = convert_node(c1R, mask); + } + else + { + v0K = convert_node(c0R, mask); + v1K = convert_node(c1R, mask); + v0Lo = convert_node(c0L, mask); + v1Lo = convert_node(c1L, mask); + } + uint64_t vcw = sgn_m(t1, + (v1Lo + neg_m(v0Lo, mask) + neg_m(Va, mask)) & mask, mask); + vcw = (vcw + sgn_m(t1, plant, mask)) & mask; + Va = (Va + neg_m(v1K, mask) + v0K + sgn_m(t1, vcw, mask)) & mask; + return vcw; +} + +HEDLEY_NO_THROW +inline unsigned __int128 paint_lane_mask(std::size_t nbits) noexcept +{ + using u128 = unsigned __int128; + if (nbits == 0) + return 0; + if (nbits >= 128) + return ~u128{0}; + return (u128{1} << nbits) - 1; +} + +/// High `d` bits of an `nbits`-wide lane, in that lane's own positions. +HEDLEY_NO_THROW +inline unsigned __int128 paint_high_bits(unsigned __int128 alpha, + std::size_t nbits, std::size_t d) noexcept +{ + alpha &= paint_lane_mask(nbits); + if (d == 0 || nbits == 0) + return 0; + if (d >= nbits) + return alpha; + const std::size_t drop = nbits - d; + return (alpha >> drop) << drop; +} + +HEDLEY_NO_THROW +inline unsigned __int128 paint_low_aligned(unsigned __int128 alpha, + std::size_t nbits, std::size_t d) noexcept +{ + alpha &= paint_lane_mask(nbits); + if (d == 0 || nbits == 0) + return 0; + if (d >= nbits) + return alpha; + return alpha >> (nbits - d); +} + +/// Unit (β = 1) lose-subtree or leaf plant. The caller scales by δ. +/// `matched` is the number of leading bits already shared with α. A lose +/// subtree at that depth reconstructs to this value; `leaf` is the full match. +inline uint64_t paint_unit(cmp_kind kind, std::size_t matched, + unsigned __int128 alpha, std::size_t nbits, std::size_t length_bits, + bool leaf, paint_callback fn, const void * ctx) +{ + if (nbits == 0) + return 0; + const std::size_t d = leaf ? nbits : matched; + switch (kind) + { + case cmp_kind::lcp: + return static_cast(d); + case cmp_kind::prefix: + return static_cast(paint_high_bits(alpha, nbits, d)); + case cmp_kind::mask: + return static_cast( + paint_high_bits(paint_lane_mask(nbits), nbits, d)); + case cmp_kind::one_hot: + return d >= 64 ? 0ULL : (1ULL << d); + case cmp_kind::break_bit: + if (leaf || matched >= nbits) + return 0; + return static_cast( + (alpha >> (nbits - 1 - matched)) & 1); + case cmp_kind::prefix_with_length: + { + if (length_bits >= 128) + return static_cast(d); + unsigned __int128 packed = + paint_low_aligned(alpha, nbits, d) << length_bits; + packed |= static_cast(d); + return static_cast(packed); + } + case cmp_kind::paint: + if (fn == nullptr) + return 0; + return fn(d, static_cast(paint_high_bits(alpha, nbits, d)), + leaf, ctx); + default: + return 0; + } +} + +HEDLEY_NO_THROW +inline uint64_t scale_plant(uint64_t unit, uint64_t scale, uint64_t mask) noexcept +{ + return (unit * scale) & mask; +} + /// Final leaf value CW. `on_path` is the payload reconstructed when the query /// stays on α's path through all levels (0 for strict lt/geq; β for leq/gt). +HEDLEY_NO_THROW inline uint64_t make_final_cw(simde__m128i s0, simde__m128i s1, uint8_t t1, uint64_t Va, uint64_t mask, uint64_t on_path = 0) noexcept { @@ -206,7 +385,11 @@ struct cmp_meta bool eval_as_ge = false; // invert path-sum (geq / gt) bool include_eq = false; // plant δ on the α-path leaf (leq / gt) bool active = false; + bool incremental = false; // final correction saved at every depth + int block_width = 0; // 0 = per-level path-sum + int tail_bits = 0; // residual q; 0 when the tree covers every bit + HEDLEY_NO_THROW bool empty() const noexcept { return !active; } }; @@ -225,6 +408,7 @@ struct cmp_pack static constexpr bool is_cmp = true; static constexpr cmp_kind kind = Kind; static constexpr std::size_t prefix = 0; + static constexpr std::size_t block_width = 0; using beta_type = Beta; Beta if_true; Beta if_false; @@ -239,6 +423,7 @@ struct cmp_at_pack static constexpr bool is_cmp = true; static constexpr cmp_kind kind = Kind; static constexpr std::size_t prefix = N; + static constexpr std::size_t block_width = 0; using beta_type = Beta; Beta if_true; Beta if_false; @@ -289,6 +474,228 @@ inline auto geq_at(Beta t, Beta f = detail::dcf_impl::default_false>(std::move(t), std::move(f)); } +// --------------------------------------------------------------------------- +// Path paints. Same channel and same (if_true, if_false) scale as a comparison: +// the reconstructed value is if_false + (if_true − if_false) · unit(x). +// `unit` is the common-prefix length, the matched prefix, a mask, and so on. +// --------------------------------------------------------------------------- + +template +struct spec_is_incremental : std::false_type {}; +template +struct spec_is_incremental> + : std::bool_constant {}; + +template +struct spec_length_bits : std::integral_constant {}; +template +struct spec_length_bits> + : std::integral_constant {}; + +template +struct paint_pack +{ + static constexpr bool is_cmp = true; + static constexpr cmp_kind kind = Kind; + static constexpr std::size_t prefix = 0; + static constexpr std::size_t block_width = 0; + static constexpr std::size_t length_bits = LengthBits; + static constexpr bool incremental = false; + using beta_type = Beta; + Beta if_true; + Beta if_false; + + explicit paint_pack(Beta t, Beta f = detail::dcf_impl::default_false()) + : if_true{std::move(t)}, if_false{std::move(f)} { } +}; + +template +struct paint_at_pack +{ + static constexpr bool is_cmp = true; + static constexpr cmp_kind kind = Kind; + static constexpr std::size_t prefix = N; + static constexpr std::size_t block_width = 0; + static constexpr std::size_t length_bits = LengthBits; + static constexpr bool incremental = false; + using beta_type = Beta; + Beta if_true; + Beta if_false; + + explicit paint_at_pack(Beta t, Beta f = detail::dcf_impl::default_false()) + : if_true{std::move(t)}, if_false{std::move(f)} { } +}; + +template +inline auto lcp(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_pack(std::move(t), std::move(f)); +} +template +inline auto lcp_at(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_at_pack(std::move(t), std::move(f)); +} + +template +inline auto common_prefix(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_pack(std::move(t), std::move(f)); +} +template +inline auto common_prefix_at(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_at_pack(std::move(t), std::move(f)); +} + +template +inline auto prefix_mask(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_pack(std::move(t), std::move(f)); +} +template +inline auto prefix_mask_at(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_at_pack(std::move(t), std::move(f)); +} + +template +inline auto diverge_one_hot(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_pack(std::move(t), std::move(f)); +} +template +inline auto diverge_one_hot_at(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_at_pack(std::move(t), std::move(f)); +} + +template +inline auto break_bit(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_pack(std::move(t), std::move(f)); +} +template +inline auto break_bit_at(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_at_pack(std::move(t), std::move(f)); +} + +/// Low `LengthBits` hold the common-prefix length. Above them sits the +/// matched prefix packed into the low bits of the lane (`α >> (N − d)`). +template +inline auto prefix_with_length(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_pack( + std::move(t), std::move(f)); +} +template +inline auto prefix_with_length_at(Beta t = Beta{1}, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_at_pack( + std::move(t), std::move(f)); +} + +/// Arbitrary unit plant. `fn(matched, in_lane_prefix, leaf)` returns the β = 1 +/// value of that sibling subtree (`leaf` is the full match, `matched == N`). +/// The result is scaled by `if_true − if_false` like the canned recipes. +template +struct paint_fn_pack +{ + static constexpr bool is_cmp = true; + static constexpr cmp_kind kind = cmp_kind::paint; + static constexpr std::size_t prefix = 0; + static constexpr std::size_t block_width = 0; + static constexpr std::size_t length_bits = 0; + static constexpr bool incremental = false; + using beta_type = Beta; + Beta if_true; + Beta if_false; + Fn fn; + + paint_fn_pack(Beta t, Beta f, Fn g) + : if_true{std::move(t)}, if_false{std::move(f)}, fn{std::move(g)} { } +}; + +template +struct paint_fn_at_pack +{ + static constexpr bool is_cmp = true; + static constexpr cmp_kind kind = cmp_kind::paint; + static constexpr std::size_t prefix = N; + static constexpr std::size_t block_width = 0; + static constexpr std::size_t length_bits = 0; + static constexpr bool incremental = false; + using beta_type = Beta; + Beta if_true; + Beta if_false; + Fn fn; + + paint_fn_at_pack(Beta t, Beta f, Fn g) + : if_true{std::move(t)}, if_false{std::move(f)}, fn{std::move(g)} { } +}; + +template +inline auto path_paint(Fn fn, Beta t, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_fn_pack, std::decay_t>( + std::move(t), std::move(f), std::move(fn)); +} +template +inline auto path_paint(Fn fn) +{ + return path_paint(std::move(fn), uint64_t{1}); +} +template +inline auto path_paint_at(Fn fn, Beta t, + Beta f = detail::dcf_impl::default_false()) +{ + return paint_fn_at_pack, std::decay_t>( + std::move(t), std::move(f), std::move(fn)); +} + +/// Incremental comparison: the same predicate, correct at every prefix length. +/// Evaluate the full point with `cmp`, and a prefix with `cmp_prefix`. +template +struct idcf_pack +{ + static constexpr bool is_cmp = true; + static constexpr bool incremental = true; + static constexpr cmp_kind kind = Spec::kind; + static constexpr std::size_t prefix = Spec::prefix; + static constexpr std::size_t block_width = Spec::block_width; + static constexpr std::size_t length_bits = spec_length_bits::value; + using beta_type = typename Spec::beta_type; + beta_type if_true; + beta_type if_false; + + explicit idcf_pack(Spec spec) + : if_true{std::move(spec.if_true)}, if_false{std::move(spec.if_false)} { } +}; + +template +inline auto idcf(Spec spec) +{ + static_assert(is_paint_kind(Spec::kind) == false, + "idcf wraps lt/leq/gt/geq; path paints are already one full-domain value"); + static_assert(Spec::block_width == 0, + "idcf uses the per-level path, not blocked checkpoints"); + return idcf_pack{std::move(spec)}; +} + // --------------------------------------------------------------------------- // Equality specs: eq / eq_at // --------------------------------------------------------------------------- @@ -330,10 +737,54 @@ inline auto eq_at(Beta t, Beta f = detail::dcf_impl::default_false>(std::move(t), std::move(f)); } +template +struct block_width_pack +{ + static_assert(BlockWidth >= 1, "block_width needs B >= 1"); + static_assert(Spec::block_width == 0, "comparison is already block_width"); + static constexpr bool is_cmp = true; + static constexpr std::size_t block_width = BlockWidth; + static constexpr cmp_kind kind = Spec::kind; + static constexpr std::size_t prefix = Spec::prefix; + static constexpr bool incremental = spec_is_incremental::value; + static constexpr std::size_t length_bits = spec_length_bits::value; + using beta_type = typename Spec::beta_type; + beta_type if_true; + beta_type if_false; + + explicit block_width_pack(Spec spec) + : if_true{std::move(spec.if_true)}, if_false{std::move(spec.if_false)} { } +}; + +template +struct block_width_fn +{ + template + constexpr auto operator()(Spec spec) const + { + return block_width_pack{std::move(spec)}; + } +}; + +template +inline constexpr block_width_fn block_width{}; + template struct is_cmp_spec : std::false_type {}; template struct is_cmp_spec> : std::true_type {}; template struct is_cmp_spec> : std::true_type {}; +template +struct is_cmp_spec> : std::true_type {}; +template +struct is_cmp_spec> : std::true_type {}; +template +struct is_cmp_spec> : std::true_type {}; +template +struct is_cmp_spec> : std::true_type {}; +template +struct is_cmp_spec> : std::true_type {}; +template +struct is_cmp_spec> : std::true_type {}; template inline constexpr bool is_cmp_spec_v = is_cmp_spec::value; diff --git a/include/dpf/doerner_shelat.hpp b/include/dpf/doerner_shelat.hpp index f70b2d0..bdfd243 100644 --- a/include/dpf/doerner_shelat.hpp +++ b/include/dpf/doerner_shelat.hpp @@ -1,11 +1,12 @@ /// @file dpf/doerner_shelat.hpp /// @brief Doerner–Shelat generation of a dealer DPF key. -/// @details Two XOR shares of the point are walked level by level. Correction -/// words, advice bits, seeds, and leaves are the ones `make_dpf` -/// would emit for the XOR of those shares, the same roots, and the -/// same beaver coins. Beaver pads used to hide the path bit cancel -/// and are not part of the key. Pad randomness must not come from -/// `uniform_fill` if the beaver tape is being matched. +/// @details Two shares of the point are walked level by level — XOR shares by +/// default, or additive shares when tagged with `arith_input`. +/// Correction words, advice bits, seeds, and leaves are the ones +/// `make_dpf` would emit for the reconstructed point, the same roots, +/// and the same beaver coins. Beaver pads used to hide the path bit +/// cancel and are not part of the key. Pad randomness must not come +/// from `uniform_fill` if the beaver tape is being matched. /// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license; /// see [LICENSE.md](@ref license) for details. @@ -28,6 +29,14 @@ namespace dpf { +/// Tag: Doerner–Shelat / geneval takes additive shares of the point +/// (`x0 + x1` in the input ring). Default calls take XOR shares. +struct arith_input_t +{ +}; + +inline constexpr arith_input_t arith_input{}; + /// Roots and the Beaver-pad stream for one Doerner–Shelat generation. /// `root` is called twice, same as `make_dpf`: party 0 clears the low bit of /// the first sample, party 1 sets the low bit of the second. @@ -86,12 +95,14 @@ struct ds_and_shares simde__m128i z1; }; +HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE simde__m128i ds_xor(simde__m128i a, simde__m128i b) noexcept { return simde_mm_xor_si128(a, b); } +HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE simde__m128i ds_gate(uint8_t bit, simde__m128i block) noexcept { @@ -128,6 +139,7 @@ ds_and_pads ds_sample_and(PadRng & pad) return p; } +HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE simde__m128i ds_cw_share(simde__m128i L, simde__m128i R, uint8_t my_bit, const ds_cw_party & mine, const ds_blind & their) noexcept @@ -144,6 +156,7 @@ simde__m128i ds_cw_share(simde__m128i L, simde__m128i R, uint8_t my_bit, return out; } +HEDLEY_NO_THROW inline void ds_cw_blinds(const ds_cw_pads & p, simde__m128i L0, simde__m128i R0, uint8_t bit0, simde__m128i L1, simde__m128i R1, uint8_t bit1, @@ -155,6 +168,7 @@ inline void ds_cw_blinds(const ds_cw_pads & p, b1.msg = ds_xor(ds_xor(L1, R1), p.p1.rand); } +HEDLEY_NO_THROW inline simde__m128i ds_cw_outs(const ds_cw_pads & p, simde__m128i L0, simde__m128i R0, uint8_t bit0, simde__m128i L1, simde__m128i R1, uint8_t bit1, @@ -165,6 +179,7 @@ inline simde__m128i ds_cw_outs(const ds_cw_pads & p, ds_cw_share(L1, R1, bit1, p.p1, b0)); } +HEDLEY_NO_THROW inline uint8_t ds_open_advice(simde__m128i L0, simde__m128i R0, uint8_t bit0, simde__m128i L1, simde__m128i R1, uint8_t bit1) noexcept { @@ -177,6 +192,7 @@ inline uint8_t ds_open_advice(simde__m128i L0, simde__m128i R0, uint8_t bit0, return static_cast((t1 << 1) | (t0 & 1u)); } +HEDLEY_NO_THROW inline void ds_next_terms(simde__m128i L, simde__m128i R, uint8_t advice, simde__m128i cw, uint8_t tpack, simde__m128i & M, simde__m128i & base) noexcept { @@ -190,6 +206,7 @@ inline void ds_next_terms(simde__m128i L, simde__m128i R, uint8_t advice, base = (advice & 1u) ? ds_xor(L, cw_base) : L; } +HEDLEY_NO_THROW inline ds_and_shares ds_and_open(const ds_and_pads & p, simde__m128i M, uint8_t b_recv) noexcept { @@ -202,6 +219,7 @@ inline ds_and_shares ds_and_open(const ds_and_pads & p, simde__m128i M, return z; } +HEDLEY_NO_THROW inline simde__m128i ds_deliver(uint8_t b_exp, simde__m128i base, simde__m128i M, const ds_and_shares & z) noexcept { @@ -249,6 +267,11 @@ struct ds_cmp_gen_state bool track_coeff = false; uint64_t Va1 = 0; uint64_t last_vcw_coeff = 0; + cmp_kind kind = cmp_kind::lt; + bool paint = false; + std::size_t length_bits = 0; + paint_callback paint_cb = nullptr; + const void * paint_ctx = nullptr; }; /// Local joint simulation: today's `ds_cw_outs` / `ds_open_advice` / `ds_and_open`. @@ -288,6 +311,7 @@ struct local_cw_protocol } /// Open CW + advice only (AND pads stay in `blinds` for a later open). + HEDLEY_NO_THROW std::pair open_cw(const ds_level_blinds & b) noexcept { return {ds_cw_outs(b.cwp, b.L0, b.R0, b.bit0, b.L1, b.R1, b.bit1, @@ -297,6 +321,7 @@ struct local_cw_protocol /// Open the public value CW for this level (local: clear convert+make_value_cw). /// MPC backends open additive shares of the same word. + HEDLEY_NO_THROW uint64_t open_value_cw(const ds_level_blinds & b, uint8_t adv0, uint8_t adv1, int ai, uint64_t & Va, uint64_t beta, uint64_t mask) noexcept { @@ -304,6 +329,16 @@ struct local_cw_protocol adv1, ai, Va, beta, mask); } + /// Open a path-paint value CW. `plant` is the scaled lose-subtree constant. + HEDLEY_NO_THROW + uint64_t open_planted_cw(const ds_level_blinds & b, uint8_t adv0, uint8_t adv1, + int ai, uint64_t & Va, uint64_t plant, uint64_t mask) noexcept + { + return dcf_impl::make_value_cw_planted(b.L0, b.R0, b.L1, b.R1, adv0, + adv1, ai, Va, plant, mask); + } + + HEDLEY_NO_THROW ds_and_shares open_and(const ds_and_pads & p, simde__m128i M, uint8_t b_recv) noexcept { @@ -313,6 +348,7 @@ struct local_cw_protocol /// Open the final comparison leaf CW. Wraps `make_final_cw` so the /// Doerner–Shelat gen does not call it directly on reconstructed seeds; /// an MPC backend would open additive shares of the same word. + HEDLEY_NO_THROW uint64_t open_final_cw(simde__m128i s0, simde__m128i s1, uint8_t t1, uint64_t Va, uint64_t mask, uint64_t on_path) noexcept { @@ -323,18 +359,81 @@ struct local_cw_protocol /// the shared root sampler so the blind matches the dealer's; an MPC /// backend would instead pull a group-width element from the pad stream. template + HEDLEY_NO_THROW uint64_t sample_addend_blind(uint64_t mask, BlockSampler && sample) noexcept { return dcf_impl::sample_addend_blind(mask, std::forward(sample)); } + /// Majority of three bits (next carry of a full adder). + HEDLEY_NO_THROW + static constexpr uint8_t majority(uint8_t a, uint8_t b, uint8_t c) noexcept + { + return static_cast((a & b) | (a & c) | (b & c)); + } + + /// One additive digit: sum bit `a XOR b XOR cin`, carry out = majority. + HEDLEY_NO_THROW + static constexpr uint8_t open_sum_bit(uint8_t a, uint8_t b, uint8_t cin, + uint8_t & cout) noexcept + { + cout = majority(a, b, cin); + return static_cast(a ^ b ^ cin); + } + + /// Reconstruct `a0 + a1` via a LSB→MSB carry chain, then flip the MSB + /// when the domain is signed — matching `make_dpf` on the sum. The call + /// site never forms the sum; an MPC backend would open the same bits. + template + InputT open_arith_point(InputT a0, InputT a1) const + { + constexpr auto to_int = utils::to_integral_type{}; + using FromI = typename utils::make_from_integral_value::integral_type; + using U = std::make_unsigned_t; + const U u0 = static_cast(to_int(a0)); + const U u1 = static_cast(to_int(a1)); + U sum = 0; + uint8_t carry = 0; + constexpr std::size_t nbits = utils::bitlength_of_v; + for (std::size_t i = 0; i < nbits; ++i) + { + const uint8_t b0 = static_cast((u0 >> i) & U{1}); + const uint8_t b1 = static_cast((u1 >> i) & U{1}); + const uint8_t s = open_sum_bit(b0, b1, carry, carry); + sum = static_cast(sum | (static_cast(s) << i)); + } + InputT out = utils::make_from_integral_value{}( + static_cast(sum)); + utils::flip_msb_if_signed_integral(out); + return out; + } + + /// Encode shares for the XOR-style CW walk. XOR mode flips party 0's MSB + /// (linear over XOR). Arithmetic mode opens the sum (carry + signed MSB) + /// and returns `(alpha, 0)` so the walk matches `make_dpf(alpha)`. + template + void encode_walk_shares(InputT & x0, InputT & x1, bool arith) const + { + if (arith) + { + const InputT alpha = open_arith_point(x0, x1); + x0 = alpha; + x1 = InputT{}; + } + else + { + utils::flip_msb_if_signed_integral(x0); + } + } + /// Open a group of leaf correction words for one prefix group. In this /// local joint simulation both XOR shares of the point are present, so the /// point is reconstructed *inside* the protocol and handed to `leaf_fn` /// (which runs `make_leaves` for the group). The Doerner–Shelat gen never /// forms `x = x0 ^ x1` at its own call site; an MPC backend would instead - /// run a per-group leaf CW exchange that never reveals `x`. + /// run a per-group leaf CW exchange that never reveals `x`. After + /// `encode_walk_shares`, arithmetic inputs are already `(alpha, 0)`. template void open_leaf_group(InputT x0, InputT x1, LeafFn && leaf_fn) { @@ -351,6 +450,7 @@ struct ds_gen_state NodeT root0; NodeT root1; + HEDLEY_NO_THROW void init(NodeT r0, NodeT r1) noexcept { root0 = r0; @@ -361,9 +461,13 @@ struct ds_gen_state home[1] = 1; } + HEDLEY_NO_THROW NodeT & seed0() noexcept { return inbox[home[0]]; } + HEDLEY_NO_THROW NodeT & seed1() noexcept { return inbox[home[1]]; } + HEDLEY_NO_THROW const NodeT & seed0() const noexcept { return inbox[home[0]]; } + HEDLEY_NO_THROW const NodeT & seed1() const noexcept { return inbox[home[1]]; } }; @@ -403,15 +507,37 @@ void ds_advance_level(ds_gen_state & st, InputT x0, InputT x1, { const int ai = static_cast( (cmp->thresh >> (cmp->nbits - 1 - level)) & 1); - *value_cw_out = proto.open_value_cw(blinds, adv0, adv1, ai, cmp->Va, - cmp->beta, cmp->mask); - if (cmp->track_coeff) + if (cmp->paint) { - // Affine coefficient: same level with β = 1 on a parallel Va. - const uint64_t v1 = proto.open_value_cw(blinds, adv0, adv1, ai, - cmp->Va1, 1ULL, cmp->mask); - cmp->last_vcw_coeff = - (v1 + dcf_impl::neg_m(*value_cw_out, cmp->mask)) & cmp->mask; + const uint64_t unit = dcf_impl::paint_unit(cmp->kind, level, + cmp->thresh, cmp->nbits, cmp->length_bits, false, + cmp->paint_cb, cmp->paint_ctx); + const uint64_t plant = dcf_impl::scale_plant(unit, cmp->beta, + cmp->mask); + *value_cw_out = proto.open_planted_cw(blinds, adv0, adv1, ai, + cmp->Va, plant, cmp->mask); + if (cmp->track_coeff) + { + const uint64_t plant1 = dcf_impl::scale_plant(unit, 1ULL, + cmp->mask); + const uint64_t v1 = proto.open_planted_cw(blinds, adv0, adv1, + ai, cmp->Va1, plant1, cmp->mask); + cmp->last_vcw_coeff = + (v1 + dcf_impl::neg_m(*value_cw_out, cmp->mask)) & cmp->mask; + } + } + else + { + *value_cw_out = proto.open_value_cw(blinds, adv0, adv1, ai, cmp->Va, + cmp->beta, cmp->mask); + if (cmp->track_coeff) + { + // Affine coefficient: same level with β = 1 on a parallel Va. + const uint64_t v1 = proto.open_value_cw(blinds, adv0, adv1, ai, + cmp->Va1, 1ULL, cmp->mask); + cmp->last_vcw_coeff = + (v1 + dcf_impl::neg_m(*value_cw_out, cmp->mask)) & cmp->mask; + } } } @@ -475,14 +601,14 @@ template -auto make_dpf_doerner_shelat_impl(InputT x0, InputT x1, +auto make_dpf_doerner_shelat_impl(bool arith, InputT x0, InputT x1, RootSampler & root_sampler, CwProtocol & proto, OutputT && y, OutputTs && ...ys) { static_assert(!dpf::is_wildcard_v, - "Doerner–Shelat gen takes XOR shares of a concrete point"); + "Doerner–Shelat gen takes shares of a concrete point"); static_assert(!dpf::is_secret_share_v, - "Doerner–Shelat: pass additive_share of xor_wrapper, or raw XOR shares"); + "Doerner–Shelat: pass additive_share of xor_wrapper, or raw shares"); static_assert(sizeof(typename InteriorPRG::block_type) == sizeof(simde__m128i), "Doerner–Shelat gen uses the AES-block interior node"); @@ -492,7 +618,7 @@ auto make_dpf_doerner_shelat_impl(InputT x0, InputT x1, using input_type = typename dpf_type::input_type; constexpr auto depth = dpf_type::depth; - utils::flip_msb_if_signed_integral(x0); + proto.encode_walk_shares(x0, x1, arith); const node root0 = dpf::unset_lo_bit(static_cast(root_sampler())); const node root1 = dpf::set_lo_bit(static_cast(root_sampler())); diff --git a/include/dpf/dpf_key.hpp b/include/dpf/dpf_key.hpp index 486adc1..8f4d7e1 100644 --- a/include/dpf/dpf_key.hpp +++ b/include/dpf/dpf_key.hpp @@ -150,6 +150,12 @@ HEDLEY_PRAGMA(GCC diagnostic pop) static constexpr std::size_t num_outputs = 1 + sizeof...(OutputTs); static constexpr std::size_t cmp_depth = 0; static constexpr std::size_t cmp_out_bits = 0; + static constexpr std::size_t cmp_block = 0; + static constexpr bool cmp_idcf = false; + static constexpr std::size_t cmp_q = 0; + static constexpr std::size_t cmp_h = 0; + static constexpr std::size_t cmp_checkpoints = 0; + static constexpr std::size_t cmp_tail = 0; /// Classic keys are single-level; the unified eval surface keeps routing /// them through the classic `eval_*` fast paths (see `is_multilevel_key`). static constexpr bool is_multilevel = false; @@ -286,6 +292,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) return std::get(leaf_nodes).beaver(); } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_PURE constexpr bool is_wildcard(std::size_t i) const noexcept @@ -442,41 +449,68 @@ namespace incr /// / `cw_last` are affine in the payload δ, so after keygen with δ = 0 the /// concrete values are `base[i] + coeff[i]·δ`; `assign_cmp` patches them in /// place with no tree re-walk / re-PRG. -template +template struct cmp_wild_state { }; -template -struct cmp_wild_state +template +struct cmp_wild_state { std::array value_cw_coeff{}; ValueCwWord cw_last_coeff{0}; + std::array tail_coeff{}; + std::array prefix_cw_coeff{}; bool assigned{false}; }; -template +template struct cmp_storage { using value_cw_word = ValueCwWord; using value_cw_array = std::array; + using tail_array = std::array; + static constexpr std::size_t prefix_cw_len = Idcf ? Depth + 1 : 0; + using prefix_cw_array = std::array; cmp_storage() = default; cmp_storage(detail::cmp_meta cmp, value_cw_array value_cws, - value_cw_word cw_last_in, value_cw_word cmp_addend_in) + value_cw_word cw_last_in, value_cw_word cmp_addend_in, + tail_array tail = {}, tail_array tail_coeff = {}, + prefix_cw_array prefix = {}, prefix_cw_array prefix_coeff = {}) : cmp_{cmp}, value_cw_{value_cws}, cw_last_{cw_last_in}, - cmp_addend_{cmp_addend_in} { } + cmp_addend_{cmp_addend_in}, tail_{tail}, prefix_cw_{prefix} + { + if constexpr (Wild && Blocked) + wild_.tail_coeff = tail_coeff; + else + (void)tail_coeff; + if constexpr (Wild && Idcf) + wild_.prefix_cw_coeff = prefix_coeff; + else + (void)prefix_coeff; + } cmp_storage(detail::cmp_meta cmp, value_cw_array value_cws, value_cw_word cw_last_in, value_cw_word cmp_addend_in, - value_cw_array coeff, value_cw_word cw_last_coeff) + value_cw_array coeff, value_cw_word cw_last_coeff, + tail_array tail = {}, tail_array tail_coeff = {}, + prefix_cw_array prefix = {}, prefix_cw_array prefix_coeff = {}) : cmp_{cmp}, value_cw_{value_cws}, cw_last_{cw_last_in}, - cmp_addend_{cmp_addend_in} + cmp_addend_{cmp_addend_in}, tail_{tail}, prefix_cw_{prefix} { if constexpr (Wild) { wild_.value_cw_coeff = coeff; wild_.cw_last_coeff = cw_last_coeff; + if constexpr (Blocked) + wild_.tail_coeff = tail_coeff; + if constexpr (Idcf) + wild_.prefix_cw_coeff = prefix_coeff; } else { (void)coeff; (void)cw_last_coeff; + (void)tail_coeff; + (void)prefix_coeff; } } @@ -485,15 +519,32 @@ struct cmp_storage { return static_cast(value_cw_[level]); } + HEDLEY_NO_THROW uint64_t cw_last() const noexcept { return static_cast(cw_last_); } + HEDLEY_NO_THROW + const tail_array & tail_cw() const noexcept { return tail_; } + uint64_t tail_cw(std::size_t i) const + { + return static_cast(tail_[i]); + } + HEDLEY_NO_THROW uint64_t cmp_addend() const noexcept { return static_cast(cmp_addend_); } + HEDLEY_NO_THROW const detail::cmp_meta & cmp() const noexcept { return cmp_; } + HEDLEY_NO_THROW bool has_cmp() const noexcept { return cmp_.active; } + HEDLEY_NO_THROW + const prefix_cw_array & prefix_cws() const noexcept { return prefix_cw_; } + uint64_t prefix_cw(std::size_t i) const + { + return static_cast(prefix_cw_[i]); + } static constexpr bool is_wildcard = Wild; + HEDLEY_NO_THROW bool cmp_assigned() const noexcept { if constexpr (Wild) @@ -518,9 +569,27 @@ struct cmp_storage const uint64_t c = static_cast(wild_.value_cw_coeff[i]); value_cw_[i] = static_cast((base + c * delta) & mask); } + if constexpr (Blocked) + { + for (std::size_t i = 0; i < TailLen; ++i) + { + const uint64_t base = static_cast(tail_[i]); + const uint64_t c = static_cast(wild_.tail_coeff[i]); + tail_[i] = static_cast((base + c * delta) & mask); + } + } const uint64_t lbase = static_cast(cw_last_); const uint64_t lc = static_cast(wild_.cw_last_coeff); cw_last_ = static_cast((lbase + lc * delta) & mask); + if constexpr (Idcf) + { + for (std::size_t i = 0; i < prefix_cw_len; ++i) + { + const uint64_t base = static_cast(prefix_cw_[i]); + const uint64_t c = static_cast(wild_.prefix_cw_coeff[i]); + prefix_cw_[i] = static_cast((base + c * delta) & mask); + } + } cmp_addend_ = static_cast(addend_share & mask); wild_.assigned = true; } @@ -531,14 +600,17 @@ struct cmp_storage value_cw_array value_cw_{}; value_cw_word cw_last_{0}; value_cw_word cmp_addend_{0}; - cmp_wild_state wild_{}; + tail_array tail_{}; + prefix_cw_array prefix_cw_{}; + cmp_wild_state wild_{}; }; /// Multi-level / comparison DPF key body. `PlacedTuple` is a tuple of /// `placed` slots; `CmpDepth > 0` activates the comparison channel. template + std::size_t CmpOutBits = 0, bool CmpWild = false, + std::size_t CmpBlock = 0, bool CmpIdcf = false> struct incr_key_base { public: @@ -554,6 +626,26 @@ struct incr_key_base static constexpr std::size_t cmp_out_bits = CmpOutBits; /// True when the comparison payload is an unassigned wildcard. static constexpr bool cmp_is_wildcard = CmpWild; + /// 0 = per-level path-sum. `B >= 1` = blocked checkpoints of width `B`. + static constexpr std::size_t cmp_block = CmpBlock; + static constexpr bool cmp_idcf = CmpIdcf; + static constexpr std::size_t max_output_level = + detail::incr::max_tree_level_v; + /// Residual tail width. 2 only when dropping those levels does not cut an + /// output and the comparison itself is what sets the tree height. + static constexpr std::size_t cmp_q = [] { + if (CmpBlock == 0 || CmpDepth <= 2) + return std::size_t{0}; + if (max_output_level > CmpDepth - 2) + return std::size_t{0}; + return std::size_t{2}; + }(); + static constexpr std::size_t cmp_h = + (CmpBlock == 0) ? CmpDepth : (CmpDepth - cmp_q); + static constexpr std::size_t cmp_checkpoints = + (CmpBlock == 0 || cmp_h == 0) ? 0 : (cmp_h + CmpBlock - 1) / CmpBlock; + static constexpr std::size_t cmp_tail = + (CmpBlock == 0 || cmp_q == 0) ? 0 : (std::size_t{1} << cmp_q); /// Multi-level / comparison keys route through the slot-aware eval path. static constexpr bool is_multilevel = true; /// Narrowest unsigned word that holds `cmp_out_bits` bits (1 byte for a @@ -564,8 +656,11 @@ struct incr_key_base static constexpr std::size_t num_outputs = std::tuple_size_v; static constexpr std::size_t input_bits = utils::bitlength_of_v; - static constexpr std::size_t depth = std::max( - detail::incr::max_tree_level_v, CmpDepth); + static constexpr std::size_t depth = std::max(max_output_level, + (CmpBlock == 0) ? CmpDepth : cmp_h); + static constexpr std::size_t value_cw_len = + (CmpBlock == 0) ? depth + : (cmp_checkpoints == 0 ? std::size_t{1} : cmp_checkpoints); static constexpr auto msb_mask = utils::msb_of_v; using integral_type = utils::integral_type_from_bitlength_t< input_bits, utils::bitlength_of_v>; @@ -577,7 +672,10 @@ struct incr_key_base using correction_words_array = std::array; using correction_advice_array = std::array; - using value_cw_array = std::array; + using value_cw_array = std::array; + using tail_array = std::array; + static constexpr std::size_t prefix_cw_len = CmpIdcf ? depth + 1 : 0; + using prefix_cw_array = std::array; using meta_array = std::array; static constexpr meta_array meta = detail::incr::build_meta(); @@ -675,14 +773,17 @@ struct incr_key_base detail::cmp_meta cmp = {}, value_cw_array value_cws = {}, uint64_t cw_last_in = 0, uint64_t cmp_addend_in = 0, addend_tuple addends = {}, value_cw_array value_cw_coeff = {}, - uint64_t cw_last_coeff_in = 0) + uint64_t cw_last_coeff_in = 0, tail_array tail_in = {}, + tail_array tail_coeff_in = {}, prefix_cw_array prefix_in = {}, + prefix_cw_array prefix_coeff_in = {}) : leaf_nodes{std::move(leaves)}, offset_x{offset_share}, cmp_store_{cmp, value_cws, static_cast(cw_last_in), static_cast(cmp_addend_in), value_cw_coeff, - static_cast(cw_last_coeff_in)}, + static_cast(cw_last_coeff_in), + tail_in, tail_coeff_in, prefix_in, prefix_coeff_in}, public_addends{std::move(addends)}, root_{root}, correction_words_{correction_words}, @@ -706,10 +807,19 @@ struct incr_key_base return correction_advice_; } const value_cw_array & value_cw() const { return cmp_store_.value_cw(); } + HEDLEY_NO_THROW uint64_t cw_last() const noexcept { return cmp_store_.cw_last(); } + HEDLEY_NO_THROW + const prefix_cw_array & prefix_cws() const noexcept + { + return cmp_store_.prefix_cws(); + } + uint64_t prefix_cw(std::size_t i) const { return cmp_store_.prefix_cw(i); } /// Party-local share of the constant absorb (`if_false`, or /// `δ + if_false` when `eval_as_ge`). Reconstructs with the peer share. + HEDLEY_NO_THROW uint64_t cmp_addend() const noexcept { return cmp_store_.cmp_addend(); } + HEDLEY_NO_THROW const detail::cmp_meta & cmp() const noexcept { return cmp_store_.cmp(); } const digest_type & common_part_hash() const { return common_part_hash_; } const leaf_wrapper_tuple & leaves() const { return leaf_nodes; } @@ -728,6 +838,8 @@ struct incr_key_base (correction_advice_[level] >> direction) & 1); } uint64_t value_cw(std::size_t level) const { return cmp_store_.value_cw(level); } + const tail_array & tail_cw() const { return cmp_store_.tail_cw(); } + uint64_t tail_cw(std::size_t i) const { return cmp_store_.tail_cw(i); } template const auto & leaf() const @@ -812,6 +924,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) } template + HEDLEY_NO_THROW auto traverse_exterior(const interior_node & node) const noexcept { static_assert(num_outputs > 0, "cmp-only key has no exterior outputs"); @@ -836,9 +949,11 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// Public `if_false` addends for `eq` / `eq_at` slots. addend_tuple public_addends{}; + HEDLEY_NO_THROW bool has_cmp() const noexcept { return cmp_store_.has_cmp(); } /// True once a wildcard comparison payload has been assigned (always true /// for concrete cmp keys and for keys without a comparison channel). + HEDLEY_NO_THROW bool cmp_assigned() const noexcept { return cmp_store_.cmp_assigned(); } /// Patch the value CWs / `cw_last` for a resolved payload δ and install @@ -850,7 +965,9 @@ HEDLEY_PRAGMA(GCC diagnostic pop) } private: - cmp_storage cmp_store_{}; + cmp_storage 0), + CmpIdcf> + cmp_store_{}; interior_node root_; correction_words_array correction_words_; correction_advice_array correction_advice_; @@ -890,7 +1007,13 @@ using dpf_key_base_t = std::conditional_t< OutputT, OutputTs...>::cmp_out_bits, dpf::detail::incr::normalize_pack< utils::bitlength_of_v>, - OutputT, OutputTs...>::cmp_wild>>; + OutputT, OutputTs...>::cmp_wild, + dpf::detail::incr::normalize_pack< + utils::bitlength_of_v>, + OutputT, OutputTs...>::cmp_block, + dpf::detail::incr::normalize_pack< + utils::bitlength_of_v>, + OutputT, OutputTs...>::cmp_idcf>>; } // namespace detail @@ -919,39 +1042,40 @@ namespace incr // expanding the placed slots into the output pack and appending the phantom // cmp tag when a comparison channel is present. template struct assemble_key { using type = dpf::dpf_key>; + dpf::cmp_channel_tag>; }; -template -struct assemble_key<0, CmpOutBits, CmpWild, InteriorPRG, ExteriorPRG, InputT, - Ps...> +struct assemble_key<0, CmpOutBits, CmpWild, CmpBlock, CmpIdcf, InteriorPRG, + ExteriorPRG, InputT, Ps...> { using type = dpf::dpf_key; }; template + bool CmpWild = false, std::size_t CmpBlock = 0, bool CmpIdcf = false> struct incr_dpf_key_of; template + bool CmpWild, std::size_t CmpBlock, bool CmpIdcf> struct incr_dpf_key_of, - CmpDepth, CmpOutBits, CmpWild> + CmpDepth, CmpOutBits, CmpWild, CmpBlock, CmpIdcf> { - using type = typename assemble_key::type; + using type = typename assemble_key::type; }; template + bool CmpWild = false, std::size_t CmpBlock = 0, bool CmpIdcf = false> using incr_dpf_key_of_t = typename incr_dpf_key_of::type; + InputT, PlacedTuple, CmpDepth, CmpOutBits, CmpWild, CmpBlock, CmpIdcf>::type; } // namespace incr } // namespace detail diff --git a/include/dpf/eval_common.hpp b/include/dpf/eval_common.hpp index fdf3e8a..c029ccc 100644 --- a/include/dpf/eval_common.hpp +++ b/include/dpf/eval_common.hpp @@ -67,8 +67,10 @@ struct alignas(utils::max_align_v) dpf_output subtractive_share>; dpf_output(const dpf_output &) = default; + HEDLEY_NO_THROW dpf_output(dpf_output &&) noexcept = default; dpf_output & operator=(const dpf_output &) = default; + HEDLEY_NO_THROW dpf_output & operator=(dpf_output &&) noexcept = default; ~dpf_output() = default; @@ -156,6 +158,7 @@ auto make_eval_dpf_output(const Node & node, Input x) /// Wrap a raw comparison `Beta` value as an additive share when `KeyT` is a /// `party_key`. template +HEDLEY_NO_THROW auto make_eval_cmp_result(Beta raw) noexcept { if constexpr (is_party_key_v) diff --git a/include/dpf/eval_full.hpp b/include/dpf/eval_full.hpp index adf860c..52d208e 100644 --- a/include/dpf/eval_full.hpp +++ b/include/dpf/eval_full.hpp @@ -1,6 +1,8 @@ /// @file dpf/eval_full.hpp -/// @brief -/// @details +/// @brief Evaluate every input in the DPF domain. +/// @details Equivalent to `eval_interval` from +/// `std::numeric_limits::min()` through `max()`. +/// @snippet evaluation/eval_full.cpp eval-full /// @author Ryan Henry /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) @@ -135,6 +137,7 @@ auto eval_full(const DpfKey & dpf, return std::make_pair(std::move(outbufs), std::move(iterable)); } +/// Evaluate the whole domain, allocating a basic full memoizer and a buffer. template (from); integral_type to_node = utils::get_to_node(to); - auto segs = utils::split_leaf_nodes(from_node, to_node, dpf.depth); + constexpr auto to_int = utils::to_integral_type{}; + const bool wraps = utils::interval_wraps( + static_cast(to_int(from)), + static_cast(to_int(to)), + utils::bitlength_of_v); + auto segs = utils::split_leaf_nodes(from_node, to_node, dpf.depth, wraps); // The memoizer keeps one interval. A wrap is two intervals, and walking // the first clobbers the second, so only a single segment can be cached. if (segs.n == 1) @@ -359,7 +364,12 @@ auto eval_inner_product_impl(const DpfKey & dpf, InputT from, InputT to, integral_type from_node = utils::get_from_node(from); integral_type to_node = utils::get_to_node(to); - auto segs = utils::split_leaf_nodes(from_node, to_node, dpf.depth); + constexpr auto to_int = utils::to_integral_type{}; + const bool wraps = utils::interval_wraps( + static_cast(to_int(from)), + static_cast(to_int(to)), + utils::bitlength_of_v); + auto segs = utils::split_leaf_nodes(from_node, to_node, dpf.depth, wraps); auto accs = std::make_tuple( ip_accum>{}...); diff --git a/include/dpf/eval_interval.hpp b/include/dpf/eval_interval.hpp index 4c3eb39..8243a1c 100644 --- a/include/dpf/eval_interval.hpp +++ b/include/dpf/eval_interval.hpp @@ -1,6 +1,10 @@ /// @file dpf/eval_interval.hpp -/// @brief -/// @details +/// @brief Evaluate every input in a closed interval. +/// @details `[from, to]` is inclusive. The returned iterable yields one +/// share per input, in that order. Pass a named output buffer; +/// this overload binds it as a non-const reference. An interval +/// memoizer is optional and comes after the buffer. +/// @snippet evaluation/eval_interval.cpp eval-interval /// @author Ryan Henry /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) @@ -339,7 +343,12 @@ auto eval_interval_impl(const DpfKey & dpf, InputT from, InputT to, integral_type from_node = utils::get_from_node(from), to_node = utils::get_to_node(to); - auto segs = utils::split_leaf_nodes(from_node, to_node, dpf.depth); + constexpr auto to_int = utils::to_integral_type{}; + const bool wraps = utils::interval_wraps( + static_cast(to_int(from)), + static_cast(to_int(to)), + utils::bitlength_of_v); + auto segs = utils::split_leaf_nodes(from_node, to_node, dpf.depth, wraps); auto idxs = std::index_sequence{}; std::size_t start = 0; @@ -388,6 +397,10 @@ auto eval_interval(const DpfKey & dpf, InputT from, InputT to, } // namespace internal +/// Write outputs `I, Is...` for `[from, to]` into `outbufs`. +/// @param outbufs Named buffer, or a tuple of buffers when several outputs +/// are selected. Must outlive the returned iterable. +/// @param memoizer Workspace sized for at least this interval. template (dpf, dpf.offset_x(from), dpf.offset_x(to), outbufs, memoizer, std::make_index_sequence<1+sizeof...(Is)>()); } +/// Evaluate `[from, to]` into `outbufs`, allocating a basic interval memoizer. template (from, to)); } +/// Evaluate `[from, to]` with a caller-supplied memoizer. +/// @return `std::pair` of a new buffer (or tuple of buffers) and an iterable +/// into that buffer. template ` selects another output. +/// `eval_point` returns a tuple of shares. +/// Pass a `basic_path_memoizer` lvalue to resume a previous path. +/// An unassigned wildcard output throws `std::runtime_error`. +/// @snippet evaluation/eval_point.cpp eval-point /// @author Ryan Henry /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) @@ -72,6 +77,10 @@ auto eval_point(const DpfKey & dpf, InputT && x, PathMemoizer && path) } // namespace internal +/// Evaluate output `I` at `x`. +/// @param path Mutable path memoizer. The default is a fresh +/// nonmemoizing workspace for this call. +/// @return Handle whose `operator*` is the party's share. template (dpf, tx, path), tx); } +/// Evaluate several outputs at `x`. +/// @return Tuple of shares, already dereferenced. template /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) @@ -143,6 +148,9 @@ inline auto eval_sequence(const DpfKey & dpf, ForwardIterator begin, ForwardIter } } +/// Evaluate the sorted range `[begin, end)`, allocating a buffer. +/// @param return_type `return_entire_node_tag_{}` or `return_output_only_tag_{}`. +/// @return Pair of buffer (or tuple of buffers) and an iterable in list order. template +struct cmp_prefix_t +{ + static constexpr std::size_t length = L; +}; +template +inline constexpr cmp_prefix_t cmp_prefix{}; + template struct is_out : std::false_type { @@ -53,10 +62,22 @@ struct is_cmp_target : std::bool_constant, cmp_t> template inline constexpr bool is_cmp_target_v = is_cmp_target::value; +template +struct is_cmp_prefix_target : std::false_type +{ +}; +template +struct is_cmp_prefix_target> : std::true_type +{ +}; +template +inline constexpr bool is_cmp_prefix_target_v = + is_cmp_prefix_target>::value; + /// True for channel tags that must not bind as the key in classic eval_*. template inline constexpr bool is_eval_channel_tag_v = - is_out_v || is_cmp_target_v; + is_out_v || is_cmp_target_v || is_cmp_prefix_target_v; template struct looks_like_dpf_key : std::false_type diff --git a/include/dpf/eval_unified.hpp b/include/dpf/eval_unified.hpp index 6e1ddf3..4193275 100644 --- a/include/dpf/eval_unified.hpp +++ b/include/dpf/eval_unified.hpp @@ -1,6 +1,11 @@ /// @file dpf/eval_unified.hpp /// @brief Target-first eval surface for DPF / iDPF / DCF channels. -/// @details `eval_*(out, …)` and `eval_*(cmp, …)` are the public API. +/// @details `eval_*(out, …)` selects point-output slot `I`. +/// `eval_*(cmp, …)` selects the comparison channel. Memoizer and +/// buffer arguments match the classic overloads: a path memoizer +/// on `eval_point`, an output buffer then an interval memoizer on +/// `eval_interval`. `make_output_buffer(out, key, from, to)` and +/// `make_output_buffer(cmp, key, n)` size the buffer for that channel. /// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license. @@ -38,6 +43,7 @@ namespace detail { template +HEDLEY_NO_THROW constexpr std::size_t resolved_out_prefix() noexcept { if constexpr (is_multilevel_key_v) @@ -91,6 +97,15 @@ auto eval_point(cmp_t, const KeyT & key, QueryT && x, std::forward(path)); } +template > +auto eval_point(cmp_prefix_t, const KeyT & key, QueryT && x, + PathMemoizer && path = PathMemoizer{}) +{ + return detail::incr::eval_cmp_prefix_point_impl(key, + std::forward(x), std::forward(path)); +} + // --------------------------------------------------------------------------- // eval_interval(target, key, from, to [, buf [, memo]]) // --------------------------------------------------------------------------- @@ -388,11 +403,12 @@ auto eval_out_inner_product_impl(const KeyT & dpf, LaneT from, LaneT to, utils::flip_msb_if_signed_integral(from); utils::flip_msb_if_signed_integral(to); - integral_type from_node = utils::leaf_node_floor( - static_cast(to_int(from)), lg_opl); - integral_type to_node = utils::leaf_node_ceil_exclusive( - static_cast(to_int(to)), lg_opl); - const auto segs = utils::split_leaf_nodes(from_node, to_node, to_level); + const auto from_i = static_cast(to_int(from)); + const auto to_i = static_cast(to_int(to)); + integral_type from_node = utils::leaf_node_floor(from_i, lg_opl); + integral_type to_node = utils::leaf_node_ceil_exclusive(to_i, lg_opl); + const bool wraps = utils::interval_wraps(from_i, to_i, N); + const auto segs = utils::split_leaf_nodes(from_node, to_node, to_level, wraps); ml_ip_accum acc{}; std::size_t start = 0; @@ -437,15 +453,27 @@ Beta eval_cmp_inner_product_impl(const KeyT & dpf, LaneT from, LaneT to, constexpr std::size_t stop = KeyT::cmp_depth == 0 ? KeyT::depth : KeyT::cmp_depth; detail::incr::cmp_full_interval_memo memo{count}; + const std::size_t levels = unwrap_party_key_t::cmp_block > 0 + ? unwrap_party_key_t::cmp_h : nbits; detail::incr::eval_cmp_interval_impl_interior(dpf, a, cmp_exclusive_end(b), - nbits, memo); + nbits, memo, levels); uint64_t dot = 0; for (std::size_t i = 0; i < count; ++i) { const auto q = static_cast(a + static_cast(i)); - const uint64_t raw = - detail::incr::eval_cmp_from_interval_memo(dpf, q, a, nbits, memo); + const uint64_t raw = [&] { + if constexpr (unwrap_party_key_t::cmp_block > 0) + { + return detail::blocked::eval_share_memo(dpf, q, a, + cmp_exclusive_end(b), memo); + } + else + { + return detail::incr::eval_cmp_from_interval_memo( + dpf, q, a, nbits, memo); + } + }(); const uint64_t wt = static_cast(weights[i]) & mask; dot = (dot + ((raw & mask) * wt)) & mask; } diff --git a/include/dpf/geneval.hpp b/include/dpf/geneval.hpp index f1422f2..865dc0a 100644 --- a/include/dpf/geneval.hpp +++ b/include/dpf/geneval.hpp @@ -10,10 +10,10 @@ /// nodes are identical across the two parties, so a dummy word /// cancels. /// -/// A wildcard-input call takes additive shares of the real point and -/// a public query. It samples a random target, runs geneval there, -/// and shifts the query by `target - x`, which is what -/// `offset_x` does after a wildcard key is bound to `x`. +/// Default calls take XOR shares of the point. Tagged with +/// `arith_input`, the point is the ring sum of the two shares; path +/// bits are opened by a carry chain inside the local CW protocol so +/// the words match `make_dpf(x0 + x1)` at the caller's query. /// /// `geneval_cmp` is the comparison-channel form. The value-correction /// word is a function of the secret path at every level, so the walk @@ -50,13 +50,6 @@ namespace dpf { -/// Tag for a geneval whose point is known only as additive shares. -struct wildcard_input_t -{ -}; - -inline constexpr wildcard_input_t wildcard_input{}; - /// Shares and the correction words opened along the query trie. /// `correction_words[i]` / `correction_advice[i]` match a reusable key at /// the same target for every `i < live_levels`. `leaf_live` means the @@ -78,6 +71,7 @@ namespace detail template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW T geneval_mod_add(T a, T b) noexcept { using U = std::make_unsigned_t; @@ -87,17 +81,6 @@ T geneval_mod_add(T a, T b) noexcept return out; } -template -HEDLEY_ALWAYS_INLINE -T geneval_mod_sub(T a, T b) noexcept -{ - using U = std::make_unsigned_t; - U diff = static_cast(static_cast(a) - static_cast(b)); - T out; - std::memcpy(&out, &diff, sizeof(out)); - return out; -} - template T geneval_flipped(T x) { @@ -151,8 +134,9 @@ template -auto geneval_run(InputT x0, InputT x1, const std::vector & queries, - RootSampler & root_sampler, PadRng & pads, OutputT y) +auto geneval_run(bool arith, InputT x0, InputT x1, + const std::vector & queries, RootSampler & root_sampler, + PadRng & pads, OutputT y) { static_assert(std::is_integral_v, "geneval input shares are an integral domain"); @@ -172,9 +156,10 @@ auto geneval_run(InputT x0, InputT x1, const std::vector & queries, if (queries.size() > (std::size_t{1} << 22)) throw std::length_error("geneval query is too large"); + local_cw_protocol proto{pads}; InputT x0c = x0; InputT x1c = x1; - utils::flip_msb_if_signed_integral(x0c); + proto.encode_walk_shares(x0c, x1c, arith); const InputT alpha = utils::xor_input_shares(x0c, x1c); std::vector flipped; @@ -196,7 +181,6 @@ auto geneval_run(InputT x0, InputT x1, const std::vector & queries, const uint64_t secret_leaf = geneval_leaf_id(alpha); - local_cw_protocol proto{pads}; constexpr auto to_int = utils::to_integral_type{}; const node root0 = dpf::unset_lo_bit(static_cast(root_sampler())); @@ -342,6 +326,19 @@ auto geneval_run(InputT x0, InputT x1, const std::vector & queries, return result; } +template +auto geneval_run(InputT x0, InputT x1, const std::vector & queries, + RootSampler & root_sampler, PadRng & pads, OutputT y) +{ + return geneval_run(false, x0, x1, queries, + root_sampler, pads, y); +} + template InputT geneval_from_bits(uint64_t bits) { @@ -401,22 +398,6 @@ std::vector geneval_inclusive(InputT from, InputT to) return qs; } -template -InputT geneval_sample_target(TargetSampler & sample) -{ - return static_cast(sample()); -} - -template -std::vector geneval_shift_all(const std::vector & qs, InputT delta) -{ - std::vector out; - out.reserve(qs.size()); - for (const InputT & q : qs) - out.push_back(geneval_mod_add(q, delta)); - return out; -} - } // namespace detail /// Geneval at one public point. The secret point is `x0 XOR x1`. @@ -430,7 +411,22 @@ HEDLEY_WARN_UNUSED_RESULT auto geneval_point(InputT x0, InputT x1, InputT query, ds_randomness rng, OutputT y) { - return detail::geneval_run(x0, x1, + return detail::geneval_run(false, x0, x1, + std::vector{query}, rng.root, rng.pad, y); +} + +/// Geneval at one public point. The secret point is `x0 + x1`. +template +HEDLEY_WARN_UNUSED_RESULT +auto geneval_point(arith_input_t, InputT x0, InputT x1, InputT query, + ds_randomness rng, OutputT y) +{ + return detail::geneval_run(true, x0, x1, std::vector{query}, rng.root, rng.pad, y); } @@ -445,7 +441,21 @@ HEDLEY_WARN_UNUSED_RESULT auto geneval_interval(InputT x0, InputT x1, InputT from, InputT to, ds_randomness rng, OutputT y) { - return detail::geneval_run(x0, x1, + return detail::geneval_run(false, x0, x1, + detail::geneval_inclusive(from, to), rng.root, rng.pad, y); +} + +template +HEDLEY_WARN_UNUSED_RESULT +auto geneval_interval(arith_input_t, InputT x0, InputT x1, InputT from, + InputT to, ds_randomness rng, OutputT y) +{ + return detail::geneval_run(true, x0, x1, detail::geneval_inclusive(from, to), rng.root, rng.pad, y); } @@ -460,7 +470,21 @@ HEDLEY_WARN_UNUSED_RESULT auto geneval_full(InputT x0, InputT x1, ds_randomness rng, OutputT y) { - return detail::geneval_run(x0, x1, + return detail::geneval_run(false, x0, x1, + detail::geneval_full_domain(), rng.root, rng.pad, y); +} + +template +HEDLEY_WARN_UNUSED_RESULT +auto geneval_full(arith_input_t, InputT x0, InputT x1, + ds_randomness rng, OutputT y) +{ + return detail::geneval_run(true, x0, x1, detail::geneval_full_domain(), rng.root, rng.pad, y); } @@ -477,154 +501,10 @@ auto geneval_sequence(InputT x0, InputT x1, ForwardIterator begin, ForwardIterator end, ds_randomness rng, OutputT y) { std::vector qs(begin, end); - return detail::geneval_run(x0, x1, + return detail::geneval_run(false, x0, x1, std::move(qs), rng.root, rng.pad, y); } -/// Wildcard-input geneval. `x0 + x1` is the real point (additive shares). -/// `sample_target()` is the random DPF target; the public query is shifted -/// by `target - (x0 + x1)` before the walk. -template -HEDLEY_WARN_UNUSED_RESULT -auto geneval_point(wildcard_input_t, InputT x0, InputT x1, InputT query, - ds_randomness rng, TargetSampler sample_target, - OutputT y) -{ - const InputT alpha = detail::geneval_sample_target(sample_target); - const InputT delta = detail::geneval_mod_sub(alpha, - detail::geneval_mod_add(x0, x1)); - const InputT shifted = detail::geneval_mod_add(query, delta); - InputT zero{}; - return detail::geneval_run(zero, alpha, - std::vector{shifted}, rng.root, rng.pad, y); -} - -template -HEDLEY_WARN_UNUSED_RESULT -auto geneval_point(wildcard_input_t, InputT x0, InputT x1, InputT query, - ds_randomness rng, OutputT y) -{ - return geneval_point(wildcard_input, x0, x1, query, - std::move(rng), [] { return dpf::uniform_sample(); }, y); -} - -template -HEDLEY_WARN_UNUSED_RESULT -auto geneval_interval(wildcard_input_t, InputT x0, InputT x1, InputT from, - InputT to, ds_randomness rng, TargetSampler sample_target, - OutputT y) -{ - const InputT alpha = detail::geneval_sample_target(sample_target); - const InputT delta = detail::geneval_mod_sub(alpha, - detail::geneval_mod_add(x0, x1)); - auto shifted = detail::geneval_shift_all( - detail::geneval_inclusive(from, to), delta); - InputT zero{}; - return detail::geneval_run(zero, alpha, - std::move(shifted), rng.root, rng.pad, y); -} - -template -HEDLEY_WARN_UNUSED_RESULT -auto geneval_interval(wildcard_input_t, InputT x0, InputT x1, InputT from, - InputT to, ds_randomness rng, OutputT y) -{ - return geneval_interval(wildcard_input, x0, x1, - from, to, std::move(rng), [] { return dpf::uniform_sample(); }, y); -} - -template -HEDLEY_WARN_UNUSED_RESULT -auto geneval_full(wildcard_input_t, InputT x0, InputT x1, - ds_randomness rng, TargetSampler sample_target, OutputT y) -{ - const InputT alpha = detail::geneval_sample_target(sample_target); - const InputT delta = detail::geneval_mod_sub(alpha, - detail::geneval_mod_add(x0, x1)); - InputT zero{}; - auto full = detail::geneval_run(zero, alpha, - detail::geneval_full_domain(), rng.root, rng.pad, y); - constexpr auto to_int = utils::to_integral_type{}; - const std::size_t n = full.party0.size(); - std::vector p0(n), p1(n); - for (std::size_t i = 0; i < n; ++i) - { - InputT q = detail::geneval_from_bits(i); - InputT s = detail::geneval_mod_add(q, delta); - const std::size_t si = static_cast(to_int(s)); - p0[i] = full.party0[si]; - p1[i] = full.party1[si]; - } - full.party0 = std::move(p0); - full.party1 = std::move(p1); - return full; -} - -template -HEDLEY_WARN_UNUSED_RESULT -auto geneval_full(wildcard_input_t, InputT x0, InputT x1, - ds_randomness rng, OutputT y) -{ - return geneval_full(wildcard_input, x0, x1, - std::move(rng), [] { return dpf::uniform_sample(); }, y); -} - -template -HEDLEY_WARN_UNUSED_RESULT -auto geneval_sequence(wildcard_input_t, InputT x0, InputT x1, - ForwardIterator begin, ForwardIterator end, - ds_randomness rng, TargetSampler sample_target, OutputT y) -{ - const InputT alpha = detail::geneval_sample_target(sample_target); - const InputT delta = detail::geneval_mod_sub(alpha, - detail::geneval_mod_add(x0, x1)); - std::vector qs(begin, end); - auto shifted = detail::geneval_shift_all(qs, delta); - InputT zero{}; - return detail::geneval_run(zero, alpha, - std::move(shifted), rng.root, rng.pad, y); -} - template HEDLEY_WARN_UNUSED_RESULT -auto geneval_sequence(wildcard_input_t, InputT x0, InputT x1, +auto geneval_sequence(arith_input_t, InputT x0, InputT x1, ForwardIterator begin, ForwardIterator end, ds_randomness rng, OutputT y) { - return geneval_sequence(wildcard_input, x0, x1, - begin, end, std::move(rng), [] { return dpf::uniform_sample(); }, y); + std::vector qs(begin, end); + return detail::geneval_run(true, x0, x1, + std::move(qs), rng.root, rng.pad, y); } /// Opened comparison key material and one prefix share per endpoint. @@ -651,6 +532,7 @@ struct geneval_cmp_result std::vector> correction_words; std::vector correction_advice; std::vector value_cw; + std::vector tail_cw; uint64_t cw_last = 0; uint64_t addend0 = 0; uint64_t addend1 = 0; @@ -689,12 +571,77 @@ geneval_cmp_result geneval_cmp(InputT x0, InputT x1, out.addend1 = k1.cmp_addend().raw(); out.correction_words.resize(depth); out.correction_advice.resize(depth); - out.value_cw.resize(depth); + if constexpr (key_type::cmp_block > 0) + { + out.value_cw.resize(key_type::cmp_checkpoints); + for (std::size_t i = 0; i < key_type::cmp_checkpoints; ++i) + out.value_cw[i] = k0.value_cw(i); + out.tail_cw.resize(key_type::cmp_tail); + for (std::size_t z = 0; z < key_type::cmp_tail; ++z) + out.tail_cw[z] = k0.tail_cw(z); + } + else + out.value_cw.resize(depth); for (std::size_t level = 0; level < depth; ++level) { out.correction_words[level] = k0.correction_word(level); out.correction_advice[level] = static_cast(k0.correction_advice(level)); - out.value_cw[level] = k0.value_cw(level); + if constexpr (key_type::cmp_block == 0) + out.value_cw[level] = k0.value_cw(level); + } + for (auto it = begin; it != end; ++it) + { + out.party0.push_back(eval_point(dpf::cmp, k0, *it).raw()); + out.party1.push_back(eval_point(dpf::cmp, k1, *it).raw()); + } + return out; +} + +/// Comparison geneval with additive shares of the point (`x0 + x1`). +template +HEDLEY_WARN_UNUSED_RESULT +geneval_cmp_result geneval_cmp(arith_input_t, InputT x0, InputT x1, + ForwardIterator begin, ForwardIterator end, + ds_randomness rng, Spec spec) +{ + geneval_cmp_result out; + if (begin == end) + return out; + + auto keys = make_dpf_doerner_shelat(arith_input, std::move(x0), std::move(x1), + std::move(rng), std::move(spec)); + const auto & k0 = keys.first; + const auto & k1 = keys.second; + using key_type = std::decay_t; + constexpr std::size_t depth = key_type::depth; + out.live_levels = depth; + out.mask = k0.cmp().mask; + out.cw_last = k0.cw_last(); + out.addend0 = k0.cmp_addend().raw(); + out.addend1 = k1.cmp_addend().raw(); + out.correction_words.resize(depth); + out.correction_advice.resize(depth); + if constexpr (key_type::cmp_block > 0) + { + out.value_cw.resize(key_type::cmp_checkpoints); + for (std::size_t i = 0; i < key_type::cmp_checkpoints; ++i) + out.value_cw[i] = k0.value_cw(i); + out.tail_cw.resize(key_type::cmp_tail); + for (std::size_t z = 0; z < key_type::cmp_tail; ++z) + out.tail_cw[z] = k0.tail_cw(z); + } + else + out.value_cw.resize(depth); + for (std::size_t level = 0; level < depth; ++level) + { + out.correction_words[level] = k0.correction_word(level); + out.correction_advice[level] = static_cast(k0.correction_advice(level)); + if constexpr (key_type::cmp_block == 0) + out.value_cw[level] = k0.value_cw(level); } for (auto it = begin; it != end; ++it) { @@ -718,6 +665,19 @@ geneval_cmp_result geneval_cmp(InputT x0, InputT x1, std::move(rng), dpf::gt(beta)); } +template +HEDLEY_WARN_UNUSED_RESULT +geneval_cmp_result geneval_cmp(arith_input_t, InputT x0, InputT x1, + ForwardIterator begin, ForwardIterator end, + ds_randomness rng, uint64_t beta) +{ + return geneval_cmp(arith_input, std::move(x0), std::move(x1), begin, end, + std::move(rng), dpf::gt(beta)); +} + } // namespace dpf #endif // LIBDPF_INCLUDE_DPF_GENEVAL_HPP__ diff --git a/include/dpf/incremental.hpp b/include/dpf/incremental.hpp index 91223bc..e22d1ec 100644 --- a/include/dpf/incremental.hpp +++ b/include/dpf/incremental.hpp @@ -37,6 +37,7 @@ #include "dpf/subinterval_iterable.hpp" #include "dpf/subsequence_iterable.hpp" #include "dpf/dcf.hpp" +#include "dpf/blocked_dcf.hpp" namespace dpf { @@ -53,9 +54,11 @@ inline constexpr bool args_have_eq_v = (is_eq_spec_v> || ...); template +HEDLEY_NO_THROW constexpr std::size_t forced_cmp_depth_sum() noexcept { return 0; } template +HEDLEY_NO_THROW constexpr std::size_t forced_cmp_depth_sum() noexcept { std::size_t m = 0; @@ -70,9 +73,11 @@ inline constexpr std::size_t forced_cmp_depth_v = forced_cmp_depth_sum(); template +HEDLEY_NO_THROW constexpr std::size_t forced_cmp_out_bits_sum() noexcept { return 0; } template +HEDLEY_NO_THROW constexpr std::size_t forced_cmp_out_bits_sum() noexcept { std::size_t m = 0; @@ -106,6 +111,39 @@ template inline constexpr bool forced_cmp_wild_v = (arg_is_wild_cmp::value || ...); +template >> +struct arg_cmp_block : std::integral_constant {}; +template +struct arg_cmp_block + : std::integral_constant::block_width> {}; + +template +inline constexpr std::size_t forced_cmp_block_v = + (std::size_t{0} + ... + arg_cmp_block::value); + +template +struct arg_cmp_idcf : std::false_type {}; +template +struct arg_cmp_idcf>>> + : spec_is_incremental> {}; + +template +inline constexpr bool forced_cmp_idcf_v = + (false || ... || arg_cmp_idcf::value); + +template +struct spec_has_paint_fn : std::false_type {}; +template +struct spec_has_paint_fn().fn)>> + : std::true_type {}; + +inline uint64_t paint_fn_adapter(std::size_t matched, uint64_t prefix, bool leaf, + const void * ctx) +{ + using fn_type = std::function; + return (*static_cast(ctx))(matched, prefix, leaf); +} + /// Runtime description of one comparison channel peeled from `make_dpf` args. struct dcf_runtime_spec { @@ -115,6 +153,9 @@ struct dcf_runtime_spec uint64_t mask = ~0ULL; cmp_kind kind = cmp_kind::lt; bool is_wildcard = false; // payload assigned after keygen (δ unknown now) + std::size_t length_bits = 0; + bool incremental = false; + std::function paint; }; namespace detail @@ -126,6 +167,16 @@ namespace incr // `filter_group`, `meta_holder`, and the `out_bits_v` / `lg_opl_v` / // `level_of_v` traits now live in `dpf/placement.hpp`. +template +auto expand_idpf(idpf_pack, Betas...> pack) +{ + return std::apply([](auto && ...ys) { + return std::make_tuple( + placed>{ + std::forward(ys)}...); + }, std::move(pack.values)); +} + template auto flatten_one(Arg && arg) { @@ -134,6 +185,10 @@ auto flatten_one(Arg && arg) { return std::tuple<>{}; } + else if constexpr (is_idpf_v) + { + return expand_idpf(std::forward(arg)); + } else if constexpr (is_eq_spec_v) { constexpr auto pref = A::prefix == 0 ? BitLen : A::prefix; @@ -172,6 +227,10 @@ void collect_cmp_one(dcf_runtime_spec & spec, bool & found, Arg && arg) else spec.prefix = 0; spec.kind = A::kind; + spec.length_bits = spec_length_bits::value; + spec.incremental = spec_is_incremental::value; + if constexpr (spec_has_paint_fn::value) + spec.paint = arg.fn; using Beta = typename A::beta_type; constexpr auto bits = [] { if constexpr (std::is_same_v) @@ -360,7 +419,8 @@ auto wrap_leaves(LeavesT & leaves, BeaversT & beavers, template + std::size_t CmpOutBits = 0, bool CmpWild = false, + std::size_t CmpBlock = 0> auto make_incremental_impl(InputT x, PlacedTuple placed, root_sampler_t root_sampler, const dcf_runtime_spec * cmp_spec = nullptr); @@ -384,6 +444,11 @@ inline void adjust_cmp_threshold(detail::cmp_meta & ch, : (((unsigned __int128)1 << nbits) - 1); ch.include_eq = false; ch.trivial = cmp_trivial::none; + if (is_paint_kind(ch.kind)) + { + ch.eval_as_ge = false; + return; + } switch (ch.kind) { case cmp_kind::lt: @@ -405,12 +470,15 @@ inline void adjust_cmp_threshold(detail::cmp_meta & ch, ch.trivial = cmp_trivial::always_false; break; } + if (ch.incremental) + ch.trivial = cmp_trivial::none; } /// Split the constant absorb into party shares. `target` is `if_false` for /// lt/leq, or `δ + if_false` when the path-sum is inverted (geq/gt), or the /// full if_true / if_false for trivial domain edges. Clears nothing — caller /// must not put δ on the key. +HEDLEY_NO_THROW inline void split_cmp_addend(uint64_t target, uint64_t mask, uint64_t r, uint64_t & add0, uint64_t & add1) noexcept { @@ -422,6 +490,7 @@ inline void split_cmp_addend(uint64_t target, uint64_t mask, uint64_t r, } /// Typed overload: write party-0 / party-1 additive shares of the absorb. +HEDLEY_NO_THROW inline void split_cmp_addend(uint64_t target, uint64_t mask, uint64_t r, additive_share & add0, additive_share & add1) noexcept @@ -440,14 +509,16 @@ auto extract_addends(const PlacedTuple & placed, std::index_sequence) template + bool CmpWild, std::size_t CmpBlock, bool CmpIdcf = false> auto make_incremental_impl(InputT x, PlacedTuple placed, root_sampler_t root_sampler, const dcf_runtime_spec * cmp_spec) { + static_assert(!(CmpIdcf && CmpBlock > 0), + "idcf uses the per-level path, not blocked checkpoints"); using key_type = incr_dpf_key_of_t; + CmpOutBits, CmpWild, CmpBlock, CmpIdcf>; using interior_node = typename key_type::interior_node; using input_type = typename key_type::input_type; constexpr auto depth = key_type::depth; @@ -484,9 +555,12 @@ auto make_incremental_impl(InputT x, PlacedTuple placed, detail::cmp_meta cmp{}; typename key_type::value_cw_array value_cws{}; + typename key_type::tail_array tail{}; + typename key_type::tail_array tail_coeff{}; uint64_t cw_last = 0; // Wildcard payload: per-level δ-coefficients (`value_cw(1) − value_cw(0)`), // computed with a parallel β = 1 accumulator `Va1`. Unused when !CmpWild. + // Blocked keys store one coefficient per checkpoint and tail slot instead. typename key_type::value_cw_array value_cw_coeff{}; uint64_t cw_last_coeff = 0; uint64_t Va1 = 0; @@ -505,12 +579,61 @@ auto make_incremental_impl(InputT x, PlacedTuple placed, false_value = cmp_spec->false_value & cmp_spec->mask; cmp.kind = cmp_spec->kind; cmp.active = true; + cmp.incremental = cmp_spec->incremental; cmp.eval_as_ge = false; cmp.trivial = cmp_trivial::none; + cmp.block_width = static_cast(CmpBlock); + cmp.tail_bits = static_cast(key_type::cmp_q); auto lane = lane_input(x, cmp_nbits, utils::bitlength_of_v); thresh = static_cast( utils::to_integral_type{}(lane)); adjust_cmp_threshold(cmp, thresh, cmp_nbits); + if (CmpBlock > 0 && is_paint_kind(cmp.kind)) + throw std::invalid_argument( + "path recipes use the per-level comparison channel"); + } + + const paint_callback paint_cb = + (cmp_spec != nullptr && cmp_spec->paint) + ? &paint_fn_adapter : nullptr; + const void * paint_ctx = + (cmp_spec != nullptr && cmp_spec->paint) ? &cmp_spec->paint : nullptr; + const std::size_t paint_length_bits = + cmp_spec != nullptr ? cmp_spec->length_bits : 0; + typename key_type::prefix_cw_array prefix_cw{}; + typename key_type::prefix_cw_array prefix_coeff{}; + const auto on_path_scaled = [&](uint64_t scale) -> uint64_t { + if (!is_paint_kind(cmp.kind)) + return (cmp.include_eq ? scale : 0ULL) & cmp.mask; + const uint64_t unit = detail::dcf_impl::paint_unit(cmp.kind, cmp_nbits, + thresh, cmp_nbits, paint_length_bits, true, paint_cb, paint_ctx); + return detail::dcf_impl::scale_plant(unit, scale, cmp.mask); + }; + const uint64_t on_path = on_path_scaled(delta); + const uint64_t on_path_unit = on_path_scaled(1ULL); + auto snap_prefix = [&](std::size_t at) { + if constexpr (CmpIdcf) + { + if (!cmp.incremental || cmp.trivial != cmp_trivial::none) + return; + const uint64_t word = detail::dcf_impl::make_final_cw(parent[0], + parent[1], static_cast(dpf::get_lo_bit(parent[1])), + Va, cmp.mask, on_path); + prefix_cw[at] = static_cast(word); + if constexpr (CmpWild) + { + const uint64_t w1 = detail::dcf_impl::make_final_cw(parent[0], + parent[1], static_cast(dpf::get_lo_bit(parent[1])), + Va1, cmp.mask, on_path_unit); + prefix_coeff[at] = static_cast( + (w1 + detail::dcf_impl::neg_m(word, cmp.mask)) & cmp.mask); + } + } + }; + if constexpr (CmpIdcf) + { + if (cmp.active) + snap_prefix(0); } for (std::size_t level = 0; level < depth; ++level, mask >>= 1) @@ -534,29 +657,101 @@ auto make_incremental_impl(InputT x, PlacedTuple placed, correction_advice[level] = static_cast(t[1] << 1) | t[0]; + if constexpr (CmpBlock == 0) + { if (cmp.active && cmp.trivial == cmp_trivial::none && level < cmp_nbits) { const int ai = static_cast( (thresh >> (cmp_nbits - 1 - level)) & 1); - const uint64_t base = make_value_cw(child0[0], child0[1], - child1[0], child1[1], - static_cast(advice[0]), - static_cast(advice[1]), ai, Va, delta, cmp.mask); - value_cws[level] = - static_cast(base); - if constexpr (CmpWild) + uint64_t base = 0; + if (is_paint_kind(cmp.kind)) { - // Same recurrence with β = 1 on a parallel accumulator; the - // value CW is affine in β so `coeff = value_cw(1) − base`. - const uint64_t v1 = make_value_cw(child0[0], child0[1], + const uint64_t unit = detail::dcf_impl::paint_unit(cmp.kind, + level, thresh, cmp_nbits, paint_length_bits, false, + paint_cb, paint_ctx); + base = detail::dcf_impl::make_value_cw_planted(child0[0], + child0[1], child1[0], child1[1], + static_cast(advice[0]), + static_cast(advice[1]), ai, Va, + detail::dcf_impl::scale_plant(unit, delta, cmp.mask), + cmp.mask); + if constexpr (CmpWild) + { + const uint64_t v1 = detail::dcf_impl::make_value_cw_planted( + child0[0], child0[1], child1[0], child1[1], + static_cast(advice[0]), + static_cast(advice[1]), ai, Va1, + detail::dcf_impl::scale_plant(unit, 1ULL, cmp.mask), + cmp.mask); + value_cw_coeff[level] = + static_cast( + (v1 + detail::dcf_impl::neg_m(base, cmp.mask)) + & cmp.mask); + } + } + else + { + base = make_value_cw(child0[0], child0[1], child1[0], child1[1], static_cast(advice[0]), - static_cast(advice[1]), ai, Va1, 1ULL, cmp.mask); - value_cw_coeff[level] = - static_cast( - (v1 + detail::dcf_impl::neg_m(base, cmp.mask)) - & cmp.mask); + static_cast(advice[1]), ai, Va, delta, cmp.mask); + if constexpr (CmpWild) + { + // Same recurrence with β = 1 on a parallel accumulator; the + // value CW is affine in β so `coeff = value_cw(1) − base`. + const uint64_t v1 = make_value_cw(child0[0], child0[1], + child1[0], child1[1], + static_cast(advice[0]), + static_cast(advice[1]), ai, Va1, 1ULL, cmp.mask); + value_cw_coeff[level] = + static_cast( + (v1 + detail::dcf_impl::neg_m(base, cmp.mask)) + & cmp.mask); + } + } + value_cws[level] = + static_cast(base); + snap_prefix(level + 1); + } + } + else if (cmp.active && cmp.trivial == cmp_trivial::none) + { + using sched = detail::blocked::schedule; + const std::size_t c = level + 1; + if (c <= key_type::cmp_h && sched::contains(c)) + { + const auto wi = sched::index(c); + const uint64_t word = detail::blocked::checkpoint_word( + parent[0], parent[1], delta, cmp.mask); + value_cws[wi] = + static_cast(word); + if constexpr (CmpWild) + { + value_cw_coeff[wi] = + static_cast( + detail::blocked::checkpoint_coeff( + parent[0], parent[1], cmp.mask)); + } + } + if (c == key_type::cmp_h && key_type::cmp_q > 0) + { + uint64_t words[4]{}; + uint64_t coeffs[4]{}; + const uint64_t suffix = static_cast(thresh) + & ((1ULL << key_type::cmp_q) - 1ULL); + detail::blocked::tail_words(parent[0], parent[1], + delta, cmp.mask, cmp.include_eq, suffix, key_type::cmp_q, + words, CmpWild ? coeffs : nullptr); + for (std::size_t z = 0; z < key_type::cmp_tail; ++z) + { + tail[z] = static_cast(words[z]); + if constexpr (CmpWild) + { + tail_coeff[z] = + static_cast(coeffs[z]); + } + } } } @@ -567,23 +762,24 @@ auto make_incremental_impl(InputT x, PlacedTuple placed, snap_sign[level + 1] = dpf::get_lo_bit(parent[0]); } + if constexpr (CmpBlock == 0) + { if (cmp.active && cmp.trivial == cmp_trivial::none && level + 1 == cmp_nbits) { - const uint64_t on_path = cmp.include_eq ? delta : 0ULL; cw_last = make_final_cw(parent[0], parent[1], static_cast(dpf::get_lo_bit(parent[1])), Va, cmp.mask, on_path); if constexpr (CmpWild) { - const uint64_t on_path1 = cmp.include_eq ? 1ULL : 0ULL; const uint64_t l1 = make_final_cw(parent[0], parent[1], static_cast(dpf::get_lo_bit(parent[1])), Va1, - cmp.mask, on_path1); + cmp.mask, on_path_unit); cw_last_coeff = (l1 + detail::dcf_impl::neg_m(cw_last, cmp.mask)) & cmp.mask; } } + } } constexpr std::size_t ngroups = [] { @@ -648,27 +844,31 @@ auto make_incremental_impl(InputT x, PlacedTuple placed, return dpf::make_party_key_pair( key_type{root[0], correction_words, correction_advice, std::move(wrap0), off0, cmp, value_cws, cw_last, cmp_add0, adds, value_cw_coeff, - cw_last_coeff}, + cw_last_coeff, tail, tail_coeff, prefix_cw, prefix_coeff}, key_type{root[1], correction_words, correction_advice, std::move(wrap1), off1, cmp, value_cws, cw_last, cmp_add1, adds, value_cw_coeff, - cw_last_coeff}); + cw_last_coeff, tail, tail_coeff, prefix_cw, prefix_coeff}); } template -auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, - CwProtocol & proto, PlacedTuple placed, + bool CmpWild, std::size_t CmpBlock, bool CmpIdcf = false, + typename RootSampler, + typename CwProtocol> +auto make_incremental_ds_impl(bool arith, InputT x0, InputT x1, + RootSampler & root_sampler, CwProtocol & proto, PlacedTuple placed, const dcf_runtime_spec * cmp_spec) { static_assert(!dpf::is_wildcard_v, - "Doerner–Shelat gen takes XOR shares of a concrete point"); + "Doerner–Shelat gen takes shares of a concrete point"); static_assert(sizeof(typename InteriorPRG::block_type) == sizeof(simde__m128i), "Doerner–Shelat gen uses the AES-block interior node"); + static_assert(!(CmpIdcf && CmpBlock > 0), + "idcf uses the per-level path, not blocked checkpoints"); using key_type = incr_dpf_key_of_t; + CmpOutBits, CmpWild, CmpBlock, CmpIdcf>; using interior_node = typename key_type::interior_node; using input_type = typename key_type::input_type; constexpr auto depth = key_type::depth; @@ -676,7 +876,7 @@ auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, using MetaHolder = meta_holder; using namespace detail::dcf_impl; - utils::flip_msb_if_signed_integral(x0); + proto.encode_walk_shares(x0, x1, arith); const interior_node root0 = dpf::unset_lo_bit(static_cast(root_sampler())); @@ -708,6 +908,8 @@ auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, detail::cmp_meta cmp{}; typename key_type::value_cw_array value_cws{}; typename key_type::value_cw_array value_cw_coeff{}; + typename key_type::tail_array tail{}; + typename key_type::tail_array tail_coeff{}; uint64_t cw_last = 0; uint64_t cw_last_coeff = 0; ds_cmp_gen_state cmp_st{}; @@ -728,13 +930,19 @@ auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, false_value = cmp_spec->false_value & cmp_spec->mask; cmp.kind = cmp_spec->kind; cmp.active = true; + cmp.incremental = cmp_spec->incremental; cmp.eval_as_ge = false; cmp.trivial = cmp_trivial::none; + cmp.block_width = static_cast(CmpBlock); + cmp.tail_bits = static_cast(key_type::cmp_q); auto lane = lane_input(cmp_x, cmp_nbits, utils::bitlength_of_v); unsigned __int128 thresh = static_cast( utils::to_integral_type{}(lane)); adjust_cmp_threshold(cmp, thresh, cmp_nbits); + if (CmpBlock > 0 && is_paint_kind(cmp.kind)) + throw std::invalid_argument( + "path recipes use the per-level comparison channel"); cmp_st.active = cmp.active; cmp_st.nbits = cmp_nbits; cmp_st.mask = cmp.mask; @@ -742,6 +950,11 @@ auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, cmp_st.include_eq = cmp.include_eq; cmp_st.trivial = cmp.trivial; cmp_st.thresh = thresh; + cmp_st.kind = cmp.kind; + cmp_st.paint = is_paint_kind(cmp.kind); + cmp_st.length_bits = cmp_spec->length_bits; + cmp_st.paint_cb = cmp_spec->paint ? &paint_fn_adapter : nullptr; + cmp_st.paint_ctx = cmp_spec->paint ? &cmp_spec->paint : nullptr; cmp_st.Va = 0; // Wildcard payload: open CWs for δ = 0 and stash β = 1 coefficients // (same dual-accumulator scheme as dealer `make_incremental_impl`). @@ -752,24 +965,118 @@ auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, } } + typename key_type::prefix_cw_array prefix_cw{}; + typename key_type::prefix_cw_array prefix_coeff{}; + const uint64_t on_path = [&]() -> uint64_t { + if (!is_paint_kind(cmp.kind)) + return (cmp.include_eq ? delta : 0ULL) & cmp.mask; + const uint64_t unit = paint_unit(cmp.kind, cmp_st.nbits, cmp_st.thresh, + cmp_st.nbits, cmp_st.length_bits, true, cmp_st.paint_cb, + cmp_st.paint_ctx); + return scale_plant(unit, delta, cmp.mask); + }(); + const uint64_t on_path_unit = [&]() -> uint64_t { + if (!is_paint_kind(cmp.kind)) + return (cmp.include_eq ? 1ULL : 0ULL) & cmp.mask; + const uint64_t unit = paint_unit(cmp.kind, cmp_st.nbits, cmp_st.thresh, + cmp_st.nbits, cmp_st.length_bits, true, cmp_st.paint_cb, + cmp_st.paint_ctx); + return scale_plant(unit, 1ULL, cmp.mask); + }(); + auto snap_prefix = [&](std::size_t at) { + if constexpr (CmpIdcf) + { + if (!cmp.incremental || cmp.trivial != cmp_trivial::none) + return; + const uint64_t word = proto.open_final_cw(st.seed0(), st.seed1(), + static_cast(dpf::get_lo_bit(st.seed1())), cmp_st.Va, + cmp.mask, on_path); + prefix_cw[at] = static_cast(word); + if constexpr (CmpWild) + { + const uint64_t w1 = proto.open_final_cw(st.seed0(), st.seed1(), + static_cast(dpf::get_lo_bit(st.seed1())), + cmp_st.Va1, cmp.mask, on_path_unit); + prefix_coeff[at] = static_cast( + (w1 + neg_m(word, cmp.mask)) & cmp.mask); + } + } + }; + if constexpr (CmpIdcf) + { + if (cmp.active) + snap_prefix(0); + } + auto mask = key_type::msb_mask; for (std::size_t level = 0; level < depth; ++level, mask >>= 1) { uint64_t vcw = 0; - ds_advance_level(st, x0, x1, mask, level, proto, - correction_words[level], correction_advice[level], - cmp_st.active ? &vcw : nullptr, - cmp_st.active ? &cmp_st : nullptr); - if (cmp_st.active && cmp_st.trivial == cmp_trivial::none - && level < cmp_st.nbits) + if constexpr (CmpBlock == 0) { - value_cws[level] = - static_cast(vcw); - if constexpr (CmpWild) + ds_advance_level(st, x0, x1, mask, level, proto, + correction_words[level], correction_advice[level], + cmp_st.active ? &vcw : nullptr, + cmp_st.active ? &cmp_st : nullptr); + if (cmp_st.active && cmp_st.trivial == cmp_trivial::none + && level < cmp_st.nbits) { - value_cw_coeff[level] = - static_cast( - cmp_st.last_vcw_coeff); + value_cws[level] = + static_cast(vcw); + if constexpr (CmpWild) + { + value_cw_coeff[level] = + static_cast( + cmp_st.last_vcw_coeff); + } + snap_prefix(level + 1); + } + } + else + { + ds_advance_level(st, x0, x1, mask, level, proto, + correction_words[level], correction_advice[level]); + if (cmp_st.active && cmp_st.trivial == cmp_trivial::none) + { + using sched = detail::blocked::schedule; + const std::size_t c = level + 1; + if (c <= key_type::cmp_h && sched::contains(c)) + { + const auto wi = sched::index(c); + const uint64_t word = + detail::blocked::checkpoint_word( + st.seed0(), st.seed1(), delta, cmp.mask); + value_cws[wi] = + static_cast(word); + if constexpr (CmpWild) + { + value_cw_coeff[wi] = + static_cast( + detail::blocked::checkpoint_coeff( + st.seed0(), st.seed1(), cmp.mask)); + } + } + if (c == key_type::cmp_h && key_type::cmp_q > 0) + { + uint64_t words[4]{}; + uint64_t coeffs[4]{}; + const uint64_t suffix = static_cast(cmp_st.thresh) + & ((1ULL << key_type::cmp_q) - 1ULL); + detail::blocked::tail_words(st.seed0(), + st.seed1(), delta, cmp.mask, cmp.include_eq, suffix, + key_type::cmp_q, words, CmpWild ? coeffs : nullptr); + for (std::size_t z = 0; z < key_type::cmp_tail; ++z) + { + tail[z] = + static_cast(words[z]); + if constexpr (CmpWild) + { + tail_coeff[z] = + static_cast( + coeffs[z]); + } + } + } } } @@ -780,23 +1087,24 @@ auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, snap_sign[level + 1] = dpf::get_lo_bit(st.seed0()); } + if constexpr (CmpBlock == 0) + { if (cmp_st.active && cmp_st.trivial == cmp_trivial::none && level + 1 == cmp_st.nbits) { - const uint64_t on_path = cmp_st.include_eq ? cmp_st.beta : 0ULL; cw_last = proto.open_final_cw(st.seed0(), st.seed1(), static_cast(dpf::get_lo_bit(st.seed1())), cmp_st.Va, cmp_st.mask, on_path); if constexpr (CmpWild) { - const uint64_t on_path1 = cmp_st.include_eq ? 1ULL : 0ULL; const uint64_t l1 = proto.open_final_cw(st.seed0(), st.seed1(), static_cast(dpf::get_lo_bit(st.seed1())), - cmp_st.Va1, cmp_st.mask, on_path1); + cmp_st.Va1, cmp_st.mask, on_path_unit); cw_last_coeff = (l1 + neg_m(cw_last, cmp_st.mask)) & cmp_st.mask; } } + } } constexpr std::size_t ngroups = [] { @@ -872,10 +1180,10 @@ auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, return dpf::make_party_key_pair( key_type{root0, correction_words, correction_advice, std::move(wrap0), off0, cmp, value_cws, cw_last, cmp_add0, adds, value_cw_coeff, - cw_last_coeff}, + cw_last_coeff, tail, tail_coeff, prefix_cw, prefix_coeff}, key_type{root1, correction_words, correction_advice, std::move(wrap1), off1, cmp, value_cws, cw_last, cmp_add1, adds, value_cw_coeff, - cw_last_coeff}); + cw_last_coeff, tail, tail_coeff, prefix_cw, prefix_coeff}); } @@ -890,7 +1198,8 @@ auto make_incremental_ds_impl(InputT x0, InputT x1, RootSampler & root_sampler, template + typename ...OutputTs, + typename = std::enable_if_t>> HEDLEY_WARN_UNUSED_RESULT auto make_dpf(InputT && x, OutputTs && ...ys) { @@ -904,6 +1213,8 @@ auto make_dpf(InputT && x, OutputTs && ...ys) constexpr std::size_t CD = forced_cmp_depth_v; constexpr std::size_t CB = forced_cmp_out_bits_v; constexpr bool CW = forced_cmp_wild_v; + constexpr std::size_t BK = forced_cmp_block_v; + constexpr bool ID = forced_cmp_idcf_v; dcf_runtime_spec dcf_spec{}; bool has_cmp = false; auto placed = detail::incr::flatten_args_and_cmp(dcf_spec, has_cmp, @@ -916,7 +1227,7 @@ auto make_dpf(InputT && x, OutputTs && ...ys) if (!has_cmp) throw std::invalid_argument("make_dpf: no outputs"); return detail::incr::make_incremental_impl(x, std::move(placed), + input_type, placed_tuple, CD, CB, CW, BK, ID>(x, std::move(placed), dpf::uniform_sample, cs); } else if constexpr (!args_have_cmp_v && !args_have_eq_v @@ -934,7 +1245,7 @@ auto make_dpf(InputT && x, OutputTs && ...ys) else { return detail::incr::make_incremental_impl(x, std::move(placed), + input_type, placed_tuple, CD, CB, CW, BK, ID>(x, std::move(placed), dpf::uniform_sample, cs); } } @@ -943,7 +1254,8 @@ template + typename ...OutputTs, + typename = std::enable_if_t>> HEDLEY_WARN_UNUSED_RESULT auto make_dpf(InputT && x, OutputTs && ...ys) { @@ -957,6 +1269,8 @@ auto make_dpf(InputT && x, OutputTs && ...ys) constexpr std::size_t CD = forced_cmp_depth_v; constexpr std::size_t CB = forced_cmp_out_bits_v; constexpr bool CW = forced_cmp_wild_v; + constexpr std::size_t BK = forced_cmp_block_v; + constexpr bool ID = forced_cmp_idcf_v; dcf_runtime_spec dcf_spec{}; bool has_cmp = false; auto placed = detail::incr::flatten_args_and_cmp(dcf_spec, has_cmp, @@ -970,7 +1284,7 @@ auto make_dpf(InputT && x, OutputTs && ...ys) if (!has_cmp) throw std::invalid_argument("make_dpf: no outputs"); return detail::incr::make_incremental_impl(x, std::move(placed), seed, cs); + input_type, placed_tuple, CD, CB, CW, BK, ID>(x, std::move(placed), seed, cs); } else if constexpr (!args_have_cmp_v && !args_have_eq_v && detail::incr::is_classic_placed()) @@ -988,14 +1302,15 @@ auto make_dpf(InputT && x, OutputTs && ...ys) else { return detail::incr::make_incremental_impl(x, std::move(placed), seed, cs); + input_type, placed_tuple, CD, CB, CW, BK, ID>(x, std::move(placed), seed, cs); } } template + typename ...OutputTs, + typename = std::enable_if_t>> HEDLEY_WARN_UNUSED_RESULT auto make_dpf(InputT && x, root_sampler_t root_sampler, OutputTs && ...ys) @@ -1009,6 +1324,8 @@ auto make_dpf(InputT && x, root_sampler_t root_sampler, constexpr std::size_t CD = forced_cmp_depth_v; constexpr std::size_t CB = forced_cmp_out_bits_v; constexpr bool CW = forced_cmp_wild_v; + constexpr std::size_t BK = forced_cmp_block_v; + constexpr bool ID = forced_cmp_idcf_v; dcf_runtime_spec dcf_spec{}; bool has_cmp = false; auto placed = detail::incr::flatten_args_and_cmp(dcf_spec, has_cmp, @@ -1022,7 +1339,7 @@ auto make_dpf(InputT && x, root_sampler_t root_sampler, if (!has_cmp) throw std::invalid_argument("make_dpf: no outputs"); return detail::incr::make_incremental_impl(x, std::move(placed), root_sampler, + input_type, placed_tuple, CD, CB, CW, BK, ID>(x, std::move(placed), root_sampler, cs); } else if constexpr (!args_have_cmp_v && !args_have_eq_v @@ -1041,7 +1358,7 @@ auto make_dpf(InputT && x, root_sampler_t root_sampler, else { return detail::incr::make_incremental_impl(x, std::move(placed), root_sampler, + input_type, placed_tuple, CD, CB, CW, BK, ID>(x, std::move(placed), root_sampler, cs); } } @@ -1059,19 +1376,56 @@ template + typename ...OutputTs, + typename = std::enable_if_t>> HEDLEY_WARN_UNUSED_RESULT auto make_dpf_doerner_shelat(InputT x0, InputT x1, ds_randomness rng, OutputTs && ...ys) +{ + return make_dpf_doerner_shelat( + false, std::move(x0), std::move(x1), std::move(rng), + std::forward(ys)...); +} + +/// Doerner–Shelat with additive shares: the point is `x0 + x1` in the input +/// ring (unsigned wrap; signed MSB flipped after the carry chain). +template >> +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(arith_input_t, InputT x0, InputT x1, + ds_randomness rng, OutputTs && ...ys) +{ + return make_dpf_doerner_shelat( + true, std::move(x0), std::move(x1), std::move(rng), + std::forward(ys)...); +} + +template >> +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(bool arith, InputT x0, InputT x1, + ds_randomness rng, OutputTs && ...ys) { using input_type = std::decay_t; static_assert(!is_secret_share_v, - "Doerner–Shelat: use additive_share of xor_wrapper, or raw XOR shares"); + "Doerner–Shelat: use additive_share of xor_wrapper, or raw shares"); using node = typename ExteriorPRG::block_type; constexpr auto bitlen = utils::bitlength_of_v; constexpr std::size_t CD = forced_cmp_depth_v; constexpr std::size_t CB = forced_cmp_out_bits_v; constexpr bool CW = forced_cmp_wild_v; + constexpr std::size_t BK = forced_cmp_block_v; + constexpr bool ID = forced_cmp_idcf_v; dcf_runtime_spec dcf_spec{}; bool has_cmp = false; auto placed = detail::incr::flatten_args_and_cmp(dcf_spec, has_cmp, @@ -1085,8 +1439,8 @@ auto make_dpf_doerner_shelat(InputT x0, InputT x1, throw std::invalid_argument("make_dpf_doerner_shelat: no outputs"); local_cw_protocol proto{rng.pad}; return detail::incr::make_incremental_ds_impl(std::move(x0), std::move(x1), - rng.root, proto, std::move(placed), cs); + input_type, placed_tuple, CD, CB, CW, BK, ID>(arith, std::move(x0), + std::move(x1), rng.root, proto, std::move(placed), cs); } else if constexpr (!args_have_cmp_v && !args_have_eq_v && detail::incr::is_classic_placed()) @@ -1097,7 +1451,7 @@ auto make_dpf_doerner_shelat(InputT x0, InputT x1, return std::apply( [&](auto && ...zs) { return detail::make_dpf_doerner_shelat_impl(std::move(x0), std::move(x1), rng.root, + ExteriorPRG>(arith, std::move(x0), std::move(x1), rng.root, proto, std::forward(zs)...); }, std::move(vals)); @@ -1106,8 +1460,8 @@ auto make_dpf_doerner_shelat(InputT x0, InputT x1, { local_cw_protocol proto{rng.pad}; return detail::incr::make_incremental_ds_impl(std::move(x0), std::move(x1), - rng.root, proto, std::move(placed), cs); + input_type, placed_tuple, CD, CB, CW, BK, ID>(arith, std::move(x0), + std::move(x1), rng.root, proto, std::move(placed), cs); } } @@ -1122,10 +1476,51 @@ template >::value>> + detail::is_cw_protocol>::value + && no_ic_pack_v>> HEDLEY_WARN_UNUSED_RESULT auto make_dpf_doerner_shelat(InputT x0, InputT x1, RootSampler root_sampler, CwProtocol & proto, OutputT && y, OutputTs && ...ys) +{ + return make_dpf_doerner_shelat( + false, std::move(x0), std::move(x1), std::move(root_sampler), proto, + std::forward(y), std::forward(ys)...); +} + +template >::value + && no_ic_pack_v>> +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(arith_input_t, InputT x0, InputT x1, + RootSampler root_sampler, CwProtocol & proto, OutputT && y, + OutputTs && ...ys) +{ + return make_dpf_doerner_shelat( + true, std::move(x0), std::move(x1), std::move(root_sampler), proto, + std::forward(y), std::forward(ys)...); +} + +template >::value + && no_ic_pack_v>> +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(bool arith, InputT x0, InputT x1, + RootSampler root_sampler, CwProtocol & proto, OutputT && y, + OutputTs && ...ys) { using input_type = std::decay_t; using node = typename ExteriorPRG::block_type; @@ -1133,6 +1528,8 @@ auto make_dpf_doerner_shelat(InputT x0, InputT x1, RootSampler root_sampler, constexpr std::size_t CD = forced_cmp_depth_v; constexpr std::size_t CB = forced_cmp_out_bits_v; constexpr bool CW = forced_cmp_wild_v; + constexpr std::size_t BK = forced_cmp_block_v; + constexpr bool ID = forced_cmp_idcf_v; dcf_runtime_spec dcf_spec{}; bool has_cmp = false; auto placed = detail::incr::flatten_args_and_cmp(dcf_spec, has_cmp, @@ -1145,8 +1542,8 @@ auto make_dpf_doerner_shelat(InputT x0, InputT x1, RootSampler root_sampler, if (!has_cmp) throw std::invalid_argument("make_dpf_doerner_shelat: no outputs"); return detail::incr::make_incremental_ds_impl(std::move(x0), std::move(x1), - root_sampler, proto, std::move(placed), cs); + input_type, placed_tuple, CD, CB, CW, BK, ID>(arith, std::move(x0), + std::move(x1), root_sampler, proto, std::move(placed), cs); } else if constexpr (!args_have_cmp_v && !args_have_eq_v @@ -1157,16 +1554,16 @@ auto make_dpf_doerner_shelat(InputT x0, InputT x1, RootSampler root_sampler, return std::apply( [&](auto && ...zs) { return detail::make_dpf_doerner_shelat_impl(std::move(x0), std::move(x1), root_sampler, - proto, std::forward(zs)...); + ExteriorPRG>(arith, std::move(x0), std::move(x1), + root_sampler, proto, std::forward(zs)...); }, std::move(vals)); } else { return detail::incr::make_incremental_ds_impl(std::move(x0), std::move(x1), - root_sampler, proto, std::move(placed), cs); + input_type, placed_tuple, CD, CB, CW, BK, ID>(arith, std::move(x0), + std::move(x1), root_sampler, proto, std::move(placed), cs); } } @@ -1179,7 +1576,8 @@ template >::value && !detail::is_cw_protocol>::value - && !detail::first_is_cw_protocol::value>> + && !detail::first_is_cw_protocol::value + && no_ic_pack_v>> HEDLEY_WARN_UNUSED_RESULT auto make_dpf_doerner_shelat(InputT x0, InputT x1, OutputT && y, OutputTs && ...ys) { @@ -1191,6 +1589,28 @@ auto make_dpf_doerner_shelat(InputT x0, InputT x1, OutputT && y, OutputTs && ... std::forward(ys)...); } +template >::value + && !detail::is_cw_protocol>::value + && !detail::first_is_cw_protocol::value + && no_ic_pack_v>> +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(arith_input_t, InputT x0, InputT x1, OutputT && y, + OutputTs && ...ys) +{ + using block = typename InteriorPRG::block_type; + ds_randomness rng{ + dpf::uniform_sample, {}}; + return make_dpf_doerner_shelat( + arith_input, std::move(x0), std::move(x1), rng, std::forward(y), + std::forward(ys)...); +} + /// Doerner–Shelat from party-tagged additive XOR shares of the point. template ; - const integral_type from_node = utils::leaf_node_floor( - static_cast(to_int(from)), lg); - const integral_type to_node = utils::leaf_node_ceil_exclusive( - static_cast(to_int(to)), lg); + const auto from_i = static_cast(to_int(from)); + const auto to_i = static_cast(to_int(to)); + const integral_type from_node = utils::leaf_node_floor(from_i, lg); + const integral_type to_node = utils::leaf_node_ceil_exclusive(to_i, lg); + const bool wraps = utils::interval_wraps(from_i, to_i, N); const auto segs = utils::split_leaf_nodes(from_node, to_node, - KeyT::meta[I].tree_level); + KeyT::meta[I].tree_level, wraps); return dpf::output_buffer(segs.total * opl); } @@ -1582,12 +2004,13 @@ auto eval_out_interval_impl( utils::flip_msb_if_signed_integral(from); utils::flip_msb_if_signed_integral(to); - integral_type from_node = utils::leaf_node_floor( - static_cast(to_int(from)), lg_opl); - integral_type to_node = utils::leaf_node_ceil_exclusive( - static_cast(to_int(to)), lg_opl); + const auto from_i = static_cast(to_int(from)); + const auto to_i = static_cast(to_int(to)); + integral_type from_node = utils::leaf_node_floor(from_i, lg_opl); + integral_type to_node = utils::leaf_node_ceil_exclusive(to_i, lg_opl); + const bool wraps = utils::interval_wraps(from_i, to_i, N); const auto segs = utils::split_leaf_nodes(from_node, to_node, - key_type::meta[I].tree_level); + key_type::meta[I].tree_level, wraps); std::size_t start = 0; for (std::size_t s = 0; s < segs.n; ++s) @@ -1601,11 +2024,11 @@ auto eval_out_interval_impl( } constexpr auto mod_pow_2 = utils::mod_pow_2{}; - auto from_i = to_int(from); - auto span = to_int(to) - from_i; + auto from_bits = to_int(from); + auto span = to_int(to) - from_bits; if constexpr (N < utils::bitlength_of_v) span &= (decltype(span){1} << N) - 1; - auto from_sz = static_cast(from_i); + auto from_sz = static_cast(from_bits); auto to_sz = from_sz + static_cast(span); return subinterval_iterable(std::begin(outbuf), utils::size(outbuf), from_sz, to_sz, mod_pow_2(from, lg_opl), opl); @@ -1625,11 +2048,12 @@ auto eval_out_interval_impl( utils::flip_msb_if_signed_integral(f); utils::flip_msb_if_signed_integral(t); constexpr auto lg = key_type::template lg_outputs_per_leaf_of; - const integral_type from_node = utils::leaf_node_floor( - static_cast(to_int(f)), lg); - const integral_type to_node = utils::leaf_node_ceil_exclusive( - static_cast(to_int(t)), lg); - const auto segs = utils::split_leaf_nodes(from_node, to_node, L); + const auto from_i = static_cast(to_int(f)); + const auto to_i = static_cast(to_int(t)); + const integral_type from_node = utils::leaf_node_floor(from_i, lg); + const integral_type to_node = utils::leaf_node_ceil_exclusive(to_i, lg); + const bool wraps = utils::interval_wraps(from_i, to_i, N); + const auto segs = utils::split_leaf_nodes(from_node, to_node, L, wraps); auto memo = basic_interval_memoizer_at(segs.total); return eval_out_interval_impl(dpf, from, to, outbuf, memo); } @@ -1754,12 +2178,11 @@ namespace incr template uint64_t eval_cmp_path_sum(const KeyT & dpf, typename KeyT::input_type tx, - PathMemoizer & path) + PathMemoizer & path, bool as_prefix = false, std::size_t prefix_len = 0) { using namespace detail::dcf_impl; const auto & ch = dpf.cmp(); const uint64_t mask = ch.mask; - const std::size_t nbits = static_cast(ch.nbits); const uint64_t add = [&]() -> uint64_t { if constexpr (is_party_key_v) return dpf.cmp_addend().raw(); @@ -1767,8 +2190,19 @@ uint64_t eval_cmp_path_sum(const KeyT & dpf, typename KeyT::input_type tx, return dpf.cmp_addend(); }(); - if (ch.trivial == cmp_trivial::always_true - || ch.trivial == cmp_trivial::always_false) + if constexpr (unwrap_party_key_t::cmp_block > 0) + { + if (as_prefix) + throw std::invalid_argument( + "cmp_prefix is not defined for a blocked comparison"); + return detail::blocked::eval_share(dpf, tx, path); + } + + const std::size_t full_bits = static_cast(ch.nbits); + const std::size_t nbits = as_prefix ? prefix_len : full_bits; + if ((ch.trivial == cmp_trivial::always_true + || ch.trivial == cmp_trivial::always_false) + && (!as_prefix || prefix_len == full_bits)) return add; dpf::detail::ensure_level(dpf, tx, path, nbits); @@ -1791,7 +2225,8 @@ uint64_t eval_cmp_path_sum(const KeyT & dpf, typename KeyT::input_type tx, const auto & leaf = path[nbits]; const uint8_t t = static_cast(dpf::get_lo_bit(leaf)); const uint64_t c = convert_node(leaf, mask); - const uint64_t contrib = (c + (t ? dpf.cw_last() : 0ULL)) & mask; + const uint64_t last = as_prefix ? dpf.prefix_cw(nbits) : dpf.cw_last(); + const uint64_t contrib = (c + (t ? last : 0ULL)) & mask; V = (V + (party ? neg_m(contrib, mask) : contrib)) & mask; if (ch.eval_as_ge) @@ -1869,6 +2304,7 @@ struct cmp_full_interval_memo - utils::shift_right(from_node, offset) + 1; } + HEDLEY_NO_THROW return_type operator[](std::size_t level) const noexcept { return Allocator::assume_aligned(&buf[level_endpoints[level]]); @@ -1902,8 +2338,11 @@ struct cmp_full_interval_memo /// Expand interval interior nodes for the comparison prefix (stop = nbits). template void eval_cmp_interval_impl_interior(const KeyT & dpf, IntegralT from_node, - IntegralT to_node, std::size_t nbits, IntervalMemoizer & memoizer) + IntegralT to_node, std::size_t nbits, IntervalMemoizer & memoizer, + std::size_t tree_levels = static_cast(-1)) { + if (tree_levels == static_cast(-1)) + tree_levels = nbits; using node_type = typename KeyT::interior_node; using input_type = typename KeyT::input_type; const input_type lane_msb = @@ -1914,7 +2353,7 @@ void eval_cmp_interval_impl_interior(const KeyT & dpf, IntegralT from_node, IntegralT mask = static_cast( utils::to_integral_type{}(lane_msb) >> (level_index - 1)); - for (; level_index <= nbits; + for (; level_index <= tree_levels; level_index = memoizer.advance_level(), nodes_at_level = memoizer.get_nodes_at_level(), mask >>= 1) { @@ -2019,6 +2458,30 @@ auto eval_cmp_point_impl(const KeyT & dpf, QueryT && x, detail::dcf_impl::u64_to_beta(raw)); } +template > +auto eval_cmp_prefix_point_impl(const KeyT & dpf, QueryT && x, + PathMemoizer && path = PathMemoizer{}) +{ + static_assert(unwrap_party_key_t::cmp_idcf, + "cmp_prefix requires an idcf key"); + if (!dpf.has_cmp()) + throw std::invalid_argument("cmp_prefix: key has no comparison channel"); + if (!dpf.cmp().incremental) + throw std::invalid_argument("cmp_prefix: key was not built with idcf"); + if (!dpf.cmp_assigned()) + throw std::invalid_argument( + "cmp_prefix: wildcard comparison payload not assigned (call assign_cmp)"); + if (L > static_cast(dpf.cmp().nbits)) + throw std::invalid_argument("cmp_prefix: prefix is longer than the comparison"); + auto tx = dpf.offset_x(std::forward(x)); + utils::flip_msb_if_signed_integral(tx); + const uint64_t raw = + detail::incr::eval_cmp_path_sum(dpf, tx, path, true, L); + return make_eval_cmp_result( + detail::dcf_impl::u64_to_beta(raw)); +} + // --------------------------------------------------------------------------- // Buffered / interval / sequence comparison evals // --------------------------------------------------------------------------- @@ -2112,16 +2575,31 @@ void eval_cmp_interval_impl(const KeyT & dpf, LaneT from, LaneT to, KeyT::cmp_depth == 0 ? KeyT::depth : KeyT::cmp_depth; // Rebind key depth for the memoizer by using a stop-level wrapper built on // the same assign/advance API as basic_interval_memoizer_at, but allocate - // per-level storage. + // per-level storage. Stop stays the logical comparison width so prefix + // indexes match `nbits`, even when a blocked key's seed spine is shorter. detail::incr::cmp_full_interval_memo memo{count}; - detail::incr::eval_cmp_interval_impl_interior(dpf, a, cmp_exclusive_end(b), nbits, memo); + const std::size_t levels = unwrap_party_key_t::cmp_block > 0 + ? unwrap_party_key_t::cmp_h : nbits; + detail::incr::eval_cmp_interval_impl_interior(dpf, a, cmp_exclusive_end(b), + nbits, memo, levels); for (std::size_t i = 0; i < count; ++i) { const auto q = static_cast(a + static_cast(i)); + const uint64_t raw = [&] { + if constexpr (unwrap_party_key_t::cmp_block > 0) + { + return detail::blocked::eval_share_memo(dpf, q, a, + cmp_exclusive_end(b), memo); + } + else + { + return detail::incr::eval_cmp_from_interval_memo( + dpf, q, a, nbits, memo); + } + }(); outbuf[i] = make_eval_cmp_result( - detail::dcf_impl::u64_to_beta( - detail::incr::eval_cmp_from_interval_memo(dpf, q, a, nbits, memo))); + detail::dcf_impl::u64_to_beta(raw)); } } @@ -2147,14 +2625,28 @@ void eval_cmp_interval_impl(const KeyT & dpf, LaneT from, LaneT to, const auto b = static_cast(to_int(to)); const auto count = cmp_inclusive_count(a, b); - detail::incr::eval_cmp_interval_impl_interior(dpf, a, cmp_exclusive_end(b), nbits, memo); + const std::size_t levels = unwrap_party_key_t::cmp_block > 0 + ? unwrap_party_key_t::cmp_h : nbits; + detail::incr::eval_cmp_interval_impl_interior(dpf, a, cmp_exclusive_end(b), + nbits, memo, levels); for (std::size_t i = 0; i < count; ++i) { const auto q = static_cast(a + static_cast(i)); + const uint64_t raw = [&] { + if constexpr (unwrap_party_key_t::cmp_block > 0) + { + return detail::blocked::eval_share_memo(dpf, q, a, + cmp_exclusive_end(b), memo); + } + else + { + return detail::incr::eval_cmp_from_interval_memo( + dpf, q, a, nbits, memo); + } + }(); outbuf[i] = make_eval_cmp_result( - detail::dcf_impl::u64_to_beta( - detail::incr::eval_cmp_from_interval_memo(dpf, q, a, nbits, memo))); + detail::dcf_impl::u64_to_beta(raw)); } } diff --git a/include/dpf/interval.hpp b/include/dpf/interval.hpp new file mode 100644 index 0000000..6d848d8 --- /dev/null +++ b/include/dpf/interval.hpp @@ -0,0 +1,674 @@ +/// @file dpf/interval.hpp +/// @brief Public-bound interval containment on one comparison key. +/// @details `make_dpf(r, ic(p, q, β))` hides the mask `r` and the payload `β`. +/// The bounds are public. Reconstruction is `β` when +/// `p ≤ (x − r) mod 2^n ≤ q`, and the false payload otherwise. +/// +/// The key is one `lt` comparison at `γ = r − 1`, the Boyle–Chandran– +/// Gilboa–Gupta–Ishai–Kumar–Rathee reduction (EUROCRYPT 2021, Fig. 3). +/// Evaluation walks that key at the two public shifts of `x` and adds +/// a secret-shared correction. Seed corrections, advice bits, leaves, +/// and the path memoizer stay single-path. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + +#ifndef LIBDPF_INCLUDE_DPF_INTERVAL_HPP__ +#define LIBDPF_INCLUDE_DPF_INTERVAL_HPP__ + +#include +#include +#include +#include +#include + +#include "hedley/hedley.h" + +#include "dpf/dcf.hpp" +#include "dpf/eval_unified.hpp" +#include "dpf/geneval.hpp" +#include "dpf/incremental.hpp" +#include "dpf/output_buffer.hpp" +#include "dpf/secret_share.hpp" +#include "dpf/utils.hpp" + +namespace dpf +{ + +template +struct ic_pack +{ + static constexpr bool is_ic = true; + using beta_type = Beta; + uint64_t lo = 0; + uint64_t hi = 0; + Beta if_true{}; + Beta if_false{}; +}; + +/// Spec tag and factory. `dpf::ic(p, q, beta)` builds a pack; +/// `eval_point(dpf::ic, key, x)` evaluates it. +struct ic_fn +{ + template + HEDLEY_WARN_UNUSED_RESULT + ic_pack> operator()(Lo lo, Hi hi, Beta t, + Beta f = detail::dcf_impl::default_false>()) const + { + ic_pack> spec; + spec.lo = static_cast(lo); + spec.hi = static_cast(hi); + spec.if_true = std::move(t); + spec.if_false = std::move(f); + return spec; + } +}; + +inline constexpr ic_fn ic{}; + +template +struct is_ic_key : std::false_type {}; + +template +struct ic_key +{ + static constexpr std::size_t party = Party; + static constexpr bool wildcard = Key::cmp_is_wildcard; + using input_type = Input; + using key_type = party_key; + using beta_type = Beta; + + key_type key; + uint64_t lo = 0; + uint64_t hi = 0; + uint64_t input_mask = 0; + uint64_t group_mask = 0; + /// Share of `δ`. Public `c_x ∈ {-1,0,1}` scales it locally. + uint64_t delta_share = 0; + /// Share of `δ · c_r + if_false`. + uint64_t cr_share = 0; + /// Wildcard only: shares of `1` and of `c_r`, scaled by `δ` in `assign_cmp`. + uint64_t delta_coeff = 0; + uint64_t cr_coeff = 0; + bool assigned = !wildcard; + + ic_key(key_type k, uint64_t lo_in, uint64_t hi_in, uint64_t nmask, + uint64_t gmask, uint64_t dshare, uint64_t cshare, uint64_t dcoeff, + uint64_t ccoeff) + : key(std::move(k)) + , lo(lo_in) + , hi(hi_in) + , input_mask(nmask) + , group_mask(gmask) + , delta_share(dshare) + , cr_share(cshare) + , delta_coeff(dcoeff) + , cr_coeff(ccoeff) + {} +}; + +template +struct is_ic_key> : std::true_type {}; + +template +inline constexpr bool is_ic_key_v = is_ic_key>::value; + +namespace detail +{ +namespace ic_impl +{ + +template +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t input_mask_of() noexcept +{ + constexpr auto n = utils::bitlength_of_v; + if constexpr (n >= 64) + return ~uint64_t{0}; + else + return (uint64_t{1} << n) - 1ULL; +} + +template +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t bits_of(Input x) noexcept +{ + constexpr auto to_int = utils::to_integral_type{}; + return static_cast(to_int(x)) & input_mask_of(); +} + +template +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr Input input_from_bits(uint64_t u) noexcept +{ + return static_cast(u & input_mask_of()); +} + +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t embed_small(int s, uint64_t mask) noexcept +{ + if (s >= 0) + return static_cast(s) & mask; + return dcf_impl::neg_m(static_cast(-s), mask); +} + +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t mul_mask(uint64_t a, uint64_t b, uint64_t mask) noexcept +{ + return static_cast(static_cast(a) * b) & mask; +} + +/// Dealer correction in Fig. 3, as an element of the payload group. +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t correction(uint64_t r, uint64_t p, uint64_t q, + uint64_t nmask, uint64_t gmask) noexcept +{ + const uint64_t aq = (q + r) & nmask; + const uint64_t ap = (p + r) & nmask; + const uint64_t q0 = (q + 1ULL) & nmask; + const uint64_t aq0 = (q0 + r) & nmask; + const int s = (ap > aq ? 1 : 0) - (ap > p ? 1 : 0) + + (aq0 > q0 ? 1 : 0) + (aq == nmask ? 1 : 0); + return embed_small(s, gmask); +} + +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr int public_cx(uint64_t x, uint64_t p, uint64_t q, uint64_t nmask) noexcept +{ + const uint64_t q0 = (q + 1ULL) & nmask; + return (x > p ? 1 : 0) - (x > q0 ? 1 : 0); +} + +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t shift_p(uint64_t x, uint64_t p, uint64_t nmask) noexcept +{ + return (x + (nmask - p)) & nmask; +} + +HEDLEY_CONST +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr uint64_t shift_q0(uint64_t x, uint64_t q, uint64_t nmask) noexcept +{ + const uint64_t q0 = (q + 1ULL) & nmask; + return (x + (nmask - q0)) & nmask; +} + +template +HEDLEY_PURE +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +uint64_t opened_u64(const T & v, uint64_t mask) noexcept +{ + if constexpr (is_secret_share_v>) + return dcf_impl::beta_to_u64_simple(v.raw(), mask); + else + return dcf_impl::beta_to_u64_simple(v, mask); +} + +inline void split_target(uint64_t target, uint64_t mask, + uint64_t & s0, uint64_t & s1) +{ + const uint64_t blind = dcf_impl::sample_addend_blind(mask, + [] { return dpf::uniform_sample(); }); + incr::split_cmp_addend(target, mask, blind, s0, s1); +} + +template +void check_input() +{ + static_assert(std::is_unsigned_v && !std::is_same_v, + "ic: input type must be an unsigned integer of at most 64 bits"); + static_assert(utils::bitlength_of_v <= 64, + "ic: input type must be an unsigned integer of at most 64 bits"); + static_assert(utils::bitlength_of_v > 0, + "ic: input type must be an unsigned integer of at most 64 bits"); +} + +template +void check_bounds(const ic_pack & spec) +{ + const uint64_t nmask = input_mask_of(); + if (spec.lo > nmask || spec.hi > nmask) + throw std::invalid_argument("ic: bound does not fit in the input domain"); + if (spec.lo > spec.hi) + throw std::invalid_argument("ic: require lo <= hi (the interval does not wrap)"); +} + +template +HEDLEY_NO_THROW +uint64_t group_mask_of() noexcept +{ + using B = concrete_type_t; + if constexpr (std::is_same_v) + return 1ULL; + else + return dcf_impl::default_mask_for_bits(utils::bitlength_of_v); +} + +template +ic_key make_side(party_key key, + uint64_t lo, uint64_t hi, uint64_t nmask, uint64_t gmask, + uint64_t delta_share, uint64_t cr_share, + uint64_t delta_coeff, uint64_t cr_coeff) +{ + return ic_key(std::move(key), lo, hi, nmask, gmask, + delta_share, cr_share, delta_coeff, cr_coeff); +} + +template +auto finish(uint64_t r_bits, const ic_pack & spec, Pair && inner) +{ + using in_type = std::decay_t; + using party0 = std::decay_t; + using raw_key = typename party0::key_type; + using out_beta = concrete_type_t; + const uint64_t nmask = input_mask_of(); + const uint64_t gmask = group_mask_of(); + constexpr bool wild = is_wildcard_v; + uint64_t delta = 0; + uint64_t fval = 0; + if constexpr (!wild) + { + delta = dcf_impl::beta_delta_u64(spec.if_true, spec.if_false, gmask); + fval = dcf_impl::beta_to_u64_simple(spec.if_false, gmask); + } + const uint64_t cr = correction(r_bits, spec.lo, spec.hi, nmask, gmask); + + uint64_t d0 = 0, d1 = 0, c0 = 0, c1 = 0; + uint64_t dc0 = 0, dc1 = 0, cc0 = 0, cc1 = 0; + if constexpr (wild) + { + split_target(1ULL & gmask, gmask, dc0, dc1); + split_target(cr, gmask, cc0, cc1); + } + else + { + split_target(delta, gmask, d0, d1); + const uint64_t absorb = + (mul_mask(delta, cr, gmask) + fval) & gmask; + split_target(absorb, gmask, c0, c1); + } + + auto k0 = make_side<0, raw_key, in_type, out_beta>(std::move(inner.first), + spec.lo, spec.hi, nmask, gmask, d0, c0, dc0, cc0); + auto k1 = make_side<1, raw_key, in_type, out_beta>(std::move(inner.second), + spec.lo, spec.hi, nmask, gmask, d1, c1, dc1, cc1); + return std::make_pair(std::move(k0), std::move(k1)); +} + +template +auto inner_lt(const ic_pack & spec) +{ + using B = std::decay_t; + if constexpr (is_wildcard_v) + return lt(spec.if_true, spec.if_false); + else + { + const uint64_t gmask = group_mask_of(); + const uint64_t delta = + dcf_impl::beta_delta_u64(spec.if_true, spec.if_false, gmask); + return lt(dcf_impl::u64_to_beta(delta), dcf_impl::u64_to_beta(0)); + } +} + +template +Input gamma_of(Input r) +{ + const uint64_t nmask = input_mask_of(); + const uint64_t ru = bits_of(r); + return input_from_bits((ru - 1ULL) & nmask); +} + +template +auto eval_one(const IcKey & k, Query && x, Memo & memo) +{ + if (!k.assigned) + throw std::invalid_argument( + "ic eval: wildcard payload not assigned (call assign_cmp)"); + using in_type = typename IcKey::input_type; + const uint64_t xu = bits_of(in_type(std::forward(x))); + const uint64_t xp = shift_p(xu, k.lo, k.input_mask); + const uint64_t xq = shift_q0(xu, k.hi, k.input_mask); + const uint64_t a = opened_u64( + eval_point(dpf::cmp, k.key, input_from_bits(xp), memo), + k.group_mask); + const uint64_t b = opened_u64( + eval_point(dpf::cmp, k.key, input_from_bits(xq), memo), + k.group_mask); + const int cx = public_cx(xu, k.lo, k.hi, k.input_mask); + uint64_t scaled = 0; + if (cx == 1) + scaled = k.delta_share & k.group_mask; + else if (cx == -1) + scaled = dcf_impl::neg_m(k.delta_share, k.group_mask); + const uint64_t y = (dcf_impl::neg_m(a, k.group_mask) + b + k.cr_share + + scaled) & k.group_mask; + return make_eval_cmp_result( + dcf_impl::u64_to_beta(y)); +} + +} // namespace ic_impl +} // namespace detail + +/// Dealer key for public bounds `spec` and secret mask `r`. +template +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf(InputT && r, const ic_pack & spec) +{ + using input_type = std::decay_t; + detail::ic_impl::check_input(); + detail::ic_impl::check_bounds(spec); + const uint64_t r_bits = detail::ic_impl::bits_of(input_type(r)); + const input_type gamma = detail::ic_impl::gamma_of(input_type(r)); + auto inner = make_dpf(gamma, + detail::ic_impl::inner_lt(spec)); + return detail::ic_impl::finish(r_bits, spec, std::move(inner)); +} + +/// Doerner–Shelat key. `r0 XOR r1` is the secret mask. +template +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(InputT r0, InputT r1, + ds_randomness rng, const ic_pack & spec) +{ + using input_type = std::decay_t; + detail::ic_impl::check_input(); + detail::ic_impl::check_bounds(spec); + const input_type r = utils::xor_input_shares(r0, r1); + const uint64_t r_bits = detail::ic_impl::bits_of(r); + const input_type gamma = detail::ic_impl::gamma_of(r); + const input_type g0 = r0; + const input_type g1 = utils::xor_input_shares(g0, gamma); + auto inner = make_dpf_doerner_shelat(g0, g1, + std::move(rng), detail::ic_impl::inner_lt(spec)); + return detail::ic_impl::finish(r_bits, spec, std::move(inner)); +} + +/// Doerner–Shelat IC key. `r0 + r1` is the secret mask; γ = (r0 + r1) − 1. +template +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(arith_input_t, InputT r0, InputT r1, + ds_randomness rng, const ic_pack & spec) +{ + using input_type = std::decay_t; + detail::ic_impl::check_input(); + detail::ic_impl::check_bounds(spec); + const uint64_t nmask = detail::ic_impl::input_mask_of(); + const uint64_t r_bits = + (detail::ic_impl::bits_of(r0) + detail::ic_impl::bits_of(r1)) & nmask; + const input_type r = detail::ic_impl::input_from_bits(r_bits); + // Additive shares of γ = r − 1: (r0 − 1, r1). + const input_type g0 = detail::ic_impl::input_from_bits( + (detail::ic_impl::bits_of(r0) - 1ULL) & nmask); + const input_type g1 = r1; + auto inner = make_dpf_doerner_shelat( + arith_input, g0, g1, std::move(rng), detail::ic_impl::inner_lt(spec)); + return detail::ic_impl::finish( + detail::ic_impl::bits_of(r), spec, std::move(inner)); +} + +/// Doerner–Shelat key sampled from the library entropy source. +template +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(InputT r0, InputT r1, const ic_pack & spec) +{ + using block = typename InteriorPRG::block_type; + ds_randomness rng{ + dpf::uniform_sample, {}}; + return make_dpf_doerner_shelat( + std::move(r0), std::move(r1), rng, spec); +} + +template +HEDLEY_WARN_UNUSED_RESULT +auto make_dpf_doerner_shelat(arith_input_t, InputT r0, InputT r1, + const ic_pack & spec) +{ + using block = typename InteriorPRG::block_type; + ds_randomness rng{ + dpf::uniform_sample, {}}; + return make_dpf_doerner_shelat( + arith_input, std::move(r0), std::move(r1), rng, spec); +} + +/// Open a wildcard interval payload onto an existing key pair. +template +void assign_cmp(ic_key<0, Key, Input, Beta> & k0, + ic_key<1, Key, Input, Beta> & k1, const Payload & if_true, + const Payload & if_false = Payload{}) +{ + static_assert(Key::cmp_is_wildcard, + "assign_cmp: interval payload is not a wildcard"); + const uint64_t mask = k0.group_mask; + const uint64_t delta = + detail::dcf_impl::beta_delta_u64(if_true, if_false, mask); + const uint64_t fval = + detail::dcf_impl::beta_to_u64_simple(if_false, mask); + assign_cmp(k0.key, k1.key, + detail::dcf_impl::u64_to_beta(delta), + detail::dcf_impl::u64_to_beta(0)); + k0.delta_share = detail::ic_impl::mul_mask(k0.delta_coeff, delta, mask); + k1.delta_share = detail::ic_impl::mul_mask(k1.delta_coeff, delta, mask); + k0.cr_share = detail::ic_impl::mul_mask(k0.cr_coeff, delta, mask); + k1.cr_share = detail::ic_impl::mul_mask(k1.cr_coeff, delta, mask); + uint64_t f0 = 0, f1 = 0; + detail::ic_impl::split_target(fval, mask, f0, f1); + k0.cr_share = (k0.cr_share + f0) & mask; + k1.cr_share = (k1.cr_share + f1) & mask; + k0.assigned = true; + k1.assigned = true; +} + +/// Point evaluation. `memo` is a path memoizer for the inner comparison key. +template , + typename = std::enable_if_t>> +HEDLEY_WARN_UNUSED_RESULT +auto eval_point(ic_fn, const IcKey & key, Query && x, Memo && memo = Memo{}) +{ + return detail::ic_impl::eval_one(key, std::forward(x), memo); +} + +/// Inclusive interval `[from, to]` on the input domain. +template >> +void eval_interval(ic_fn, const IcKey & key, Lane from, Lane to, + Buffer && buf, Memo && memo) +{ + using in_type = typename IcKey::input_type; + const uint64_t nmask = key.input_mask; + const uint64_t a = detail::ic_impl::bits_of(in_type(from)); + const uint64_t b = detail::ic_impl::bits_of(in_type(to)); + if (a > b) + throw std::invalid_argument("ic interval: to < from"); + std::size_t i = 0; + for (uint64_t x = a;; ++x) + { + buf[i++] = detail::ic_impl::eval_one(key, + detail::ic_impl::input_from_bits(x), memo); + if (x == b) + break; + if (x == nmask) + throw std::invalid_argument("ic interval: to < from"); + } +} + +template >> +void eval_interval(ic_fn, const IcKey & key, Lane from, Lane to, Buffer && buf) +{ + basic_path_memoizer memo; + eval_interval(ic, key, from, to, std::forward(buf), memo); +} + +/// Evaluate the points in `[begin, end)`. +template >> +void eval_sequence(ic_fn, const IcKey & key, Iter begin, Iter end, + Buffer && buf, Memo && memo) +{ + std::size_t i = 0; + for (auto it = begin; it != end; ++it, ++i) + buf[i] = detail::ic_impl::eval_one(key, *it, memo); +} + +template >> +void eval_sequence(ic_fn, const IcKey & key, Iter begin, Iter end, Buffer && buf) +{ + basic_path_memoizer memo; + eval_sequence(ic, key, begin, end, std::forward(buf), memo); +} + +/// Buffer of `n` interval shares. +template >> +HEDLEY_WARN_UNUSED_RESULT +auto make_output_buffer(ic_fn, const IcKey &, std::size_t n) +{ + using beta = typename IcKey::beta_type; + using elem = cmp_buffer_elem_t; + return output_buffer(n); +} + +/// Buffer large enough for the inclusive interval `[from, to]`. +template >> +HEDLEY_WARN_UNUSED_RESULT +auto make_output_buffer(ic_fn, const IcKey & key, Lane from, Lane to) +{ + using in_type = typename IcKey::input_type; + const uint64_t a = detail::ic_impl::bits_of(in_type(from)); + const uint64_t b = detail::ic_impl::bits_of(in_type(to)); + if (a > b) + throw std::invalid_argument("ic interval: to < from"); + const uint64_t n = b - a + 1ULL; + return make_output_buffer(ic, key, static_cast(n)); +} + +/// Doerner–Shelat geneval. `r0 XOR r1` is the secret mask. Each query is +/// returned already combined into the interval share. +template +HEDLEY_WARN_UNUSED_RESULT +geneval_cmp_result geneval_ic(InputT r0, InputT r1, Iter begin, Iter end, + ds_randomness rng, const ic_pack & spec) +{ + static_assert(!is_wildcard_v, + "geneval_ic: payload must be concrete (assign_cmp is a separate step)"); + geneval_cmp_result out; + if (begin == end) + return out; + + auto keys = make_dpf_doerner_shelat(std::move(r0), std::move(r1), + std::move(rng), spec); + const auto & k0 = keys.first; + const auto & k1 = keys.second; + using key_type = unwrap_party_key_t::key_type>; + constexpr std::size_t depth = key_type::depth; + out.live_levels = depth; + out.mask = k0.key.cmp().mask; + out.cw_last = k0.key.cw_last(); + out.addend0 = k0.key.cmp_addend().raw(); + out.addend1 = k1.key.cmp_addend().raw(); + out.correction_words.resize(depth); + out.correction_advice.resize(depth); + out.value_cw.resize(depth); + for (std::size_t level = 0; level < depth; ++level) + { + out.correction_words[level] = k0.key.correction_word(level); + out.correction_advice[level] = + static_cast(k0.key.correction_advice(level)); + out.value_cw[level] = k0.key.value_cw(level); + } + for (auto it = begin; it != end; ++it) + { + out.party0.push_back(detail::ic_impl::opened_u64( + eval_point(ic, k0, *it), out.mask)); + out.party1.push_back(detail::ic_impl::opened_u64( + eval_point(ic, k1, *it), out.mask)); + } + return out; +} + +/// Additive-share geneval_ic. `r0 + r1` is the secret mask. +template +HEDLEY_WARN_UNUSED_RESULT +geneval_cmp_result geneval_ic(arith_input_t, InputT r0, InputT r1, Iter begin, + Iter end, ds_randomness rng, const ic_pack & spec) +{ + static_assert(!is_wildcard_v, + "geneval_ic: payload must be concrete (assign_cmp is a separate step)"); + geneval_cmp_result out; + if (begin == end) + return out; + + auto keys = make_dpf_doerner_shelat(arith_input, std::move(r0), std::move(r1), + std::move(rng), spec); + const auto & k0 = keys.first; + const auto & k1 = keys.second; + using key_type = unwrap_party_key_t::key_type>; + constexpr std::size_t depth = key_type::depth; + out.live_levels = depth; + out.mask = k0.key.cmp().mask; + out.cw_last = k0.key.cw_last(); + out.addend0 = k0.key.cmp_addend().raw(); + out.addend1 = k1.key.cmp_addend().raw(); + out.correction_words.resize(depth); + out.correction_advice.resize(depth); + out.value_cw.resize(depth); + for (std::size_t level = 0; level < depth; ++level) + { + out.correction_words[level] = k0.key.correction_word(level); + out.correction_advice[level] = + static_cast(k0.key.correction_advice(level)); + out.value_cw[level] = k0.key.value_cw(level); + } + for (auto it = begin; it != end; ++it) + { + out.party0.push_back(detail::ic_impl::opened_u64( + eval_point(ic, k0, *it), out.mask)); + out.party1.push_back(detail::ic_impl::opened_u64( + eval_point(ic, k1, *it), out.mask)); + } + return out; +} + +} // namespace dpf + +#endif // LIBDPF_INCLUDE_DPF_INTERVAL_HPP__ diff --git a/include/dpf/interval_memoizer.hpp b/include/dpf/interval_memoizer.hpp index da3b691..061af06 100644 --- a/include/dpf/interval_memoizer.hpp +++ b/include/dpf/interval_memoizer.hpp @@ -1,6 +1,13 @@ /// @file dpf/interval_memoizer.hpp -/// @brief -/// @details +/// @brief Workspaces for an inclusive interval of DPF leaves. +/// @details `basic_interval_memoizer` keeps two levels of the interval. +/// `full_tree_interval_memoizer` keeps every level. Size either one +/// for the widest interval you will evaluate; a wider interval +/// throws `std::length_error`. The same key and the same endpoints +/// leave the final interior level in place. +/// +/// Factories unwrap `party_key`. Pass the memoizer as a mutable +/// lvalue to `eval_interval` or `eval_full`. /// @author Ryan Henry /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) @@ -55,10 +62,13 @@ struct interval_memoizer_base // level 0 should access the root // level goes up to (and including) depth + HEDLEY_NO_THROW virtual return_type operator[](std::size_t) const noexcept = 0; // iterators should access most recently completed level + HEDLEY_NO_THROW virtual return_type begin() const noexcept = 0; + HEDLEY_NO_THROW virtual return_type end() const noexcept = 0; virtual std::size_t assign_interval(const dpf_type & dpf, integral_type new_from, integral_type new_to) @@ -151,6 +161,8 @@ struct interval_memoizer_base std::optional to_; }; +/// Two-level workspace for one interval. This is what +/// `eval_interval(key, from, to)` allocates when you omit the memoizer. template ::interior_node>> @@ -229,6 +241,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) unique_ptr buf; }; +/// Every level of the interval. `retains_all_levels` is true. template ::interior_node>> @@ -392,6 +405,7 @@ struct basic_interval_memoizer_at - utils::shift_right(from_node, offset) + 1; } + HEDLEY_NO_THROW return_type operator[](std::size_t level) const noexcept { bool b = (depth ^ level) & 1; @@ -420,9 +434,6 @@ auto make_interval_memoizer(InputT from, InputT to) { using dpf_type = DpfKey; - utils::flip_msb_if_signed_integral(from); - utils::flip_msb_if_signed_integral(to); - std::size_t nodes_in_interval = utils::get_leafnodes_in_output_interval(from, to); return MemoizerT(nodes_in_interval); @@ -430,6 +441,10 @@ auto make_interval_memoizer(InputT from, InputT to) } // namespace detail +/// Two-level workspace sized for the closed interval `[from, to]`. +/// @param from Inclusive start, in the key's input domain. +/// @param to Inclusive end. `to` is at least `from` in that domain. +/// @snippet evaluation/memoizers.cpp interval-memoizer template inline auto make_basic_interval_memoizer(InputT from, InputT to) @@ -448,6 +463,7 @@ inline auto make_basic_interval_memoizer(const DpfKey &, InputT from, InputT to) return make_basic_interval_memoizer(from, to); } +/// `make_basic_interval_memoizer` sized for the whole input domain. template inline auto make_basic_full_memoizer() { @@ -464,6 +480,7 @@ inline auto make_basic_full_memoizer(const DpfKey &) return make_basic_full_memoizer(); } +/// Full-tree workspace sized for the closed interval `[from, to]`. template inline auto make_full_tree_interval_memoizer(InputT from, InputT to) @@ -482,6 +499,7 @@ inline auto make_full_tree_interval_memoizer(const DpfKey &, InputT from, InputT return make_full_tree_interval_memoizer(from, to); } +/// `make_full_tree_interval_memoizer` sized for the whole input domain. template inline auto make_full_tree_full_memoizer() { @@ -521,11 +539,13 @@ inline auto make_basic_interval_memoizer(InputT from, InputT to) utils::flip_msb_if_signed_integral(from); utils::flip_msb_if_signed_integral(to); - const integral_type from_node = utils::leaf_node_floor( - static_cast(to_int(from)), lg); - const integral_type to_node = utils::leaf_node_ceil_exclusive( - static_cast(to_int(to)), lg); - const auto segs = utils::split_leaf_nodes(from_node, to_node, stop); + const auto from_i = static_cast(to_int(from)); + const auto to_i = static_cast(to_int(to)); + const integral_type from_node = utils::leaf_node_floor(from_i, lg); + const integral_type to_node = utils::leaf_node_ceil_exclusive(to_i, lg); + const bool wraps = utils::interval_wraps(from_i, to_i, + utils::bitlength_of_v); + const auto segs = utils::split_leaf_nodes(from_node, to_node, stop, wraps); return basic_interval_memoizer_at(segs.total); } diff --git a/include/dpf/json.hpp b/include/dpf/json.hpp index 805073a..f52adba 100644 --- a/include/dpf/json.hpp +++ b/include/dpf/json.hpp @@ -1,6 +1,7 @@ /// @file dpf/json.hpp -/// @brief -/// @details +/// @brief nlohmann::json serializers for DPF keys and beaver triples. +/// @details ADL `adl_serializer` specializations so `nlohmann::json` can +/// convert the library's key and triple types. /// @author Ryan Henry /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license; @@ -92,6 +93,9 @@ struct adl_serializer j.at("eval_as_ge").get_to(c.eval_as_ge); j.at("include_eq").get_to(c.include_eq); j.at("active").get_to(c.active); + c.incremental = j.value("incremental", false); + c.block_width = j.value("block_width", 0); + c.tail_bits = j.value("tail_bits", 0); } static void to_json(nlohmann::json & j, const dpf::detail::cmp_meta & c) // NOLINT(runtime/references) @@ -105,6 +109,13 @@ struct adl_serializer {"include_eq", c.include_eq}, {"active", c.active} }; + if (c.incremental) + j["incremental"] = true; + if (c.block_width != 0) + { + j["block_width"] = c.block_width; + j["tail_bits"] = c.tail_bits; + } } }; @@ -197,6 +208,12 @@ struct adl_serializer(); uint64_t cmp_addend = j.at("cmp_addend").template get(); + typename dpf_type::tail_array tail{}; + if constexpr (dpf_type::cmp_block > 0) + j.at("tail_cw").get_to(tail); + typename dpf_type::prefix_cw_array prefix{}; + if constexpr (dpf_type::cmp_idcf) + j.at("prefix_cw").get_to(prefix); typename dpf_type::leaf_wrapper_tuple leaves{}; input_type offset_share{}; @@ -204,7 +221,7 @@ struct adl_serializer(dpf.cw_last())}, {"cmp_addend", static_cast(dpf.cmp_addend())} }; + if constexpr (dpf_type::cmp_block > 0) + j["tail_cw"] = dpf.tail_cw(); + if constexpr (dpf_type::cmp_idcf) + j["prefix_cw"] = dpf.prefix_cws(); } }; diff --git a/include/dpf/keyword.hpp b/include/dpf/keyword.hpp index ebd9bc3..a45db43 100644 --- a/include/dpf/keyword.hpp +++ b/include/dpf/keyword.hpp @@ -174,11 +174,13 @@ class basic_fixed_length_string : public dpf::modint(st /// @brief default constructor /// @details Constructs the `basic_fixed_length_string` with a value /// corresponding to the empty string. + HEDLEY_NO_THROW constexpr basic_fixed_length_string() noexcept = default; /// @brief copy constructor /// @details Constructs the `basic_fixed_length_string` with a value /// copied from another `basic_fixed_length_string`. + HEDLEY_NO_THROW constexpr basic_fixed_length_string(const basic_fixed_length_string &) noexcept = default; @@ -186,6 +188,7 @@ class basic_fixed_length_string : public dpf::modint(st /// @brief move constructor /// @details Constructs the `basic_fixed_length_string` from another /// `basic_fixed_length_string` using move semantics. + HEDLEY_NO_THROW constexpr basic_fixed_length_string(basic_fixed_length_string &&) noexcept = default; @@ -232,6 +235,7 @@ class basic_fixed_length_string : public dpf::modint(st /// @brief move assignment /// @details Assigns the `basic_fixed_length_string` from another /// `basic_fixed_length_string` using move semantics. + HEDLEY_NO_THROW constexpr basic_fixed_length_string & operator=(basic_fixed_length_string &&) noexcept = default; @@ -268,10 +272,12 @@ class basic_fixed_length_string : public dpf::modint(st private: constexpr // cppcheck-suppress noExplicitConstructor + HEDLEY_NO_THROW basic_fixed_length_string(integral_type val) // NOLINT(runtime/explicit) noexcept : parent::modint(val) { } + HEDLEY_NO_THROW constexpr basic_fixed_length_string(parent val) noexcept : parent::modint(val) { } @@ -448,6 +454,7 @@ struct make_from_integral_value; using integral_type = integral_type_from_bitlength_t>; + HEDLEY_NO_THROW constexpr T operator()(integral_type val) const noexcept { return T{val}; @@ -499,14 +506,23 @@ class numeric_limits::traps; static constexpr bool tinyness_before = false; + HEDLEY_NO_THROW static constexpr keyword_type min() noexcept { return keyword_type{""}; } + HEDLEY_NO_THROW static constexpr keyword_type lowest() noexcept { return keyword_type{""}; } + HEDLEY_NO_THROW static constexpr keyword_type max() noexcept { return ~keyword_type{""}; } + HEDLEY_NO_THROW static constexpr keyword_type epsilon() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr keyword_type round_error() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr keyword_type infinity() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr keyword_type quiet_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr keyword_type signaling_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr keyword_type denorm_min() noexcept { return 0; } }; diff --git a/include/dpf/keyword2.hpp b/include/dpf/keyword2.hpp index 2e0adca..a794289 100644 --- a/include/dpf/keyword2.hpp +++ b/include/dpf/keyword2.hpp @@ -45,6 +45,8 @@ #include #include +#include "hedley/hedley.h" + #include "dpf/modint.hpp" #include "dpf/utils.hpp" @@ -1492,6 +1494,7 @@ constexpr hit match_full(const program & p, const char * s, std::size_t n) return run_id(c, root, 0, u256{}, nullptr, 0, 0); } +HEDLEY_NON_NULL(4) constexpr int unrank_node(const program & p, std::uint16_t id, u256 rank, char * out, int n); constexpr int unrank_node(const program & p, std::uint16_t id, u256 rank, char * out, int n) @@ -1590,12 +1593,14 @@ constexpr int unrank_node(const program & p, std::uint16_t id, u256 rank, char * return n; } +HEDLEY_NON_NULL(3) constexpr int unrank_root(const program & p, u256 rank, char * out) { if (!p.lang.all && cmp_card_u(rank, p.lang) >= 0) return -1; return unrank_node(p, p.tree.root, rank, out, 0); } +HEDLEY_NON_NULL(1) constexpr program compile_pattern(const char * pattern) { program p{}; @@ -1719,6 +1724,7 @@ class keyword2 constexpr keyword2(std::string_view str) : parent(encode_(str)) { } + HEDLEY_NON_NULL(1) constexpr keyword2(const char * str) : keyword2(std::string_view(str)) { } @@ -1731,7 +1737,7 @@ class keyword2 constexpr keyword2 & operator=(const keyword2 &) noexcept = default; constexpr keyword2 & operator=(keyword2 &&) noexcept = default; - ~keyword2() = default; + ~keyword2() noexcept = default; /// @brief Rank, including values outside the language that fill the bit width. static constexpr keyword2 from_rank(integral_type rank) noexcept diff --git a/include/dpf/leaf_node.hpp b/include/dpf/leaf_node.hpp index cf66b12..474029d 100644 --- a/include/dpf/leaf_node.hpp +++ b/include/dpf/leaf_node.hpp @@ -81,6 +81,7 @@ static constexpr std::size_t block_length_of_leaf_v template +HEDLEY_NO_THROW constexpr std::size_t offset_within_block(InputT x) noexcept { constexpr auto mod = utils::mod_pow_2{}; diff --git a/include/dpf/literals.hpp b/include/dpf/literals.hpp index 94ff484..7eee820 100644 --- a/include/dpf/literals.hpp +++ b/include/dpf/literals.hpp @@ -1,3 +1,8 @@ +/// @file dpf/literals.hpp +/// @brief Re-exports the user-defined literal namespaces. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + #ifndef LIBDPF_INCLUDE_DPF_LITERALS_HPP__ #define LIBDPF_INCLUDE_DPF_LITERALS_HPP__ diff --git a/include/dpf/modint.hpp b/include/dpf/modint.hpp index 3504f86..35154c8 100644 --- a/include/dpf/modint.hpp +++ b/include/dpf/modint.hpp @@ -57,18 +57,21 @@ class modint /// @brief default constructor /// @details Constructs a `modint` whose value is initialized to `0`. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr modint() noexcept = default; /// @brief copy constructor /// @details Constructs the `modint` with a value copied from another /// `modint`. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr modint(const modint &) noexcept = default; /// @brief move constructor /// @details Constructs the `modint` from another `modint` using move /// semantics. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr modint(modint &&) noexcept = default; @@ -91,12 +94,14 @@ class modint /// @brief copy assignment /// @details Assigns the `modint` with a value copied from another /// `modint`. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr modint & operator=(const modint &) noexcept = default; /// @brief move assignment /// @details Assigns the `modint` from another `modint` using move /// semantics. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr modint & operator=(modint &&) noexcept = default; @@ -278,7 +283,11 @@ class modint HEDLEY_ALWAYS_INLINE constexpr modint operator<<(std::size_t shift_amount) const noexcept { - return modint{static_cast(this->val << shift_amount)}; + if (shift_amount >= Nbits) + return modint{integral_type{0}}; + // Reduce first: an unreduced limb shifted by less than Nbits can + // still leave the modulus, and a shift of the limb width is UB. + return modint{static_cast(this->reduced_value() << shift_amount)}; } /// @brief bitwise-left-shift-assignment operator @@ -290,7 +299,10 @@ class modint HEDLEY_ALWAYS_INLINE constexpr modint & operator<<=(std::size_t shift_amount) noexcept { - this->val <<= shift_amount; + if (shift_amount >= Nbits) + this->val = integral_type{0}; + else + this->val = static_cast(this->reduced_value() << shift_amount); return *this; } @@ -304,6 +316,8 @@ class modint HEDLEY_ALWAYS_INLINE constexpr modint operator>>(std::size_t shift_amount) const noexcept { + if (shift_amount >= Nbits) + return modint{integral_type{0}}; return modint{static_cast(this->reduced_value() >> shift_amount)}; } @@ -316,7 +330,10 @@ class modint HEDLEY_ALWAYS_INLINE constexpr modint & operator>>=(std::size_t shift_amount) noexcept { - this->val = this->reduced_value() >> shift_amount; + if (shift_amount >= Nbits) + this->val = integral_type{0}; + else + this->val = this->reduced_value() >> shift_amount; return *this; } @@ -603,6 +620,7 @@ class modint template HEDLEY_CONST HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr modint operator*(typename modint::integral_type lhs, modint rhs) noexcept { @@ -616,6 +634,7 @@ constexpr modint operator*(typename modint::integral_type lhs, template HEDLEY_CONST HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr bool operator<(modint lhs, modint rhs) noexcept { return static_cast::integral_type>(lhs) @@ -626,6 +645,7 @@ constexpr bool operator<(modint lhs, modint rhs) noexcept template HEDLEY_CONST HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr bool operator<=(modint lhs, modint rhs) noexcept { return static_cast::integral_type>(lhs) @@ -636,6 +656,7 @@ constexpr bool operator<=(modint lhs, modint rhs) noexcept template HEDLEY_CONST HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr bool operator>(modint lhs, modint rhs) noexcept { return static_cast::integral_type>(lhs) @@ -646,6 +667,7 @@ constexpr bool operator>(modint lhs, modint rhs) noexcept template HEDLEY_CONST HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr bool operator>=(modint lhs, modint rhs) noexcept { return static_cast::integral_type>(lhs) @@ -656,6 +678,7 @@ constexpr bool operator>=(modint lhs, modint rhs) noexcept template HEDLEY_CONST HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr bool operator==(modint lhs, modint rhs) noexcept { return static_cast::integral_type>(lhs) @@ -666,6 +689,7 @@ constexpr bool operator==(modint lhs, modint rhs) noexcept template HEDLEY_CONST HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr bool operator!=(modint lhs, modint rhs) noexcept { return static_cast::integral_type>(lhs) @@ -695,6 +719,7 @@ template struct countl_zero_symmetric_difference> { using T = dpf::modint; + HEDLEY_NO_THROW HEDLEY_CONST HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(T lhs, T rhs) const noexcept @@ -731,6 +756,7 @@ struct mod_pow_2> { using T = dpf::modint; static constexpr auto mod = mod_pow_2{}; + HEDLEY_NO_THROW std::size_t operator()(T val, std::size_t n) const noexcept { return mod(val.val, n); @@ -1106,218 +1132,218 @@ constexpr static auto operator "" _u61(unsigned long long int x) { return dpf::m constexpr static auto operator "" _u62(unsigned long long int x) { return dpf::modints::modint62_t{static_cast(x)}; } constexpr static auto operator "" _u63(unsigned long long int x) { return dpf::modints::modint63_t{static_cast(x)}; } constexpr static auto operator "" _u64(unsigned long long int x) { return dpf::modints::modint64_t{static_cast(x)}; } -template constexpr static auto operator "" _u65() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint65_t{x}; } -template constexpr static auto operator "" _u66() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint66_t{x}; } -template constexpr static auto operator "" _u67() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint67_t{x}; } -template constexpr static auto operator "" _u68() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint68_t{x}; } -template constexpr static auto operator "" _u69() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint69_t{x}; } +template constexpr static auto operator "" _u65() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint65_t{x}; } +template constexpr static auto operator "" _u66() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint66_t{x}; } +template constexpr static auto operator "" _u67() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint67_t{x}; } +template constexpr static auto operator "" _u68() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint68_t{x}; } +template constexpr static auto operator "" _u69() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint69_t{x}; } // 70--79 -template constexpr static auto operator "" _u70() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint70_t{x}; } -template constexpr static auto operator "" _u71() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint71_t{x}; } -template constexpr static auto operator "" _u72() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint72_t{x}; } -template constexpr static auto operator "" _u73() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint73_t{x}; } -template constexpr static auto operator "" _u74() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint74_t{x}; } -template constexpr static auto operator "" _u75() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint75_t{x}; } -template constexpr static auto operator "" _u76() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint76_t{x}; } -template constexpr static auto operator "" _u77() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint77_t{x}; } -template constexpr static auto operator "" _u78() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint78_t{x}; } -template constexpr static auto operator "" _u79() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint79_t{x}; } +template constexpr static auto operator "" _u70() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint70_t{x}; } +template constexpr static auto operator "" _u71() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint71_t{x}; } +template constexpr static auto operator "" _u72() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint72_t{x}; } +template constexpr static auto operator "" _u73() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint73_t{x}; } +template constexpr static auto operator "" _u74() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint74_t{x}; } +template constexpr static auto operator "" _u75() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint75_t{x}; } +template constexpr static auto operator "" _u76() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint76_t{x}; } +template constexpr static auto operator "" _u77() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint77_t{x}; } +template constexpr static auto operator "" _u78() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint78_t{x}; } +template constexpr static auto operator "" _u79() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint79_t{x}; } // 80--89 -template constexpr static auto operator "" _u80() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint80_t{x}; } -template constexpr static auto operator "" _u81() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint81_t{x}; } -template constexpr static auto operator "" _u82() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint82_t{x}; } -template constexpr static auto operator "" _u83() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint83_t{x}; } -template constexpr static auto operator "" _u84() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint84_t{x}; } -template constexpr static auto operator "" _u85() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint85_t{x}; } -template constexpr static auto operator "" _u86() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint86_t{x}; } -template constexpr static auto operator "" _u87() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint87_t{x}; } -template constexpr static auto operator "" _u88() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint88_t{x}; } -template constexpr static auto operator "" _u89() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint89_t{x}; } +template constexpr static auto operator "" _u80() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint80_t{x}; } +template constexpr static auto operator "" _u81() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint81_t{x}; } +template constexpr static auto operator "" _u82() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint82_t{x}; } +template constexpr static auto operator "" _u83() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint83_t{x}; } +template constexpr static auto operator "" _u84() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint84_t{x}; } +template constexpr static auto operator "" _u85() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint85_t{x}; } +template constexpr static auto operator "" _u86() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint86_t{x}; } +template constexpr static auto operator "" _u87() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint87_t{x}; } +template constexpr static auto operator "" _u88() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint88_t{x}; } +template constexpr static auto operator "" _u89() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint89_t{x}; } // 90--99 -template constexpr static auto operator "" _u90() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint90_t{x}; } -template constexpr static auto operator "" _u91() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint91_t{x}; } -template constexpr static auto operator "" _u92() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint92_t{x}; } -template constexpr static auto operator "" _u93() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint93_t{x}; } -template constexpr static auto operator "" _u94() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint94_t{x}; } -template constexpr static auto operator "" _u95() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint95_t{x}; } -template constexpr static auto operator "" _u96() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint96_t{x}; } -template constexpr static auto operator "" _u97() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint97_t{x}; } -template constexpr static auto operator "" _u98() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint98_t{x}; } -template constexpr static auto operator "" _u99() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint99_t{x}; } +template constexpr static auto operator "" _u90() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint90_t{x}; } +template constexpr static auto operator "" _u91() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint91_t{x}; } +template constexpr static auto operator "" _u92() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint92_t{x}; } +template constexpr static auto operator "" _u93() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint93_t{x}; } +template constexpr static auto operator "" _u94() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint94_t{x}; } +template constexpr static auto operator "" _u95() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint95_t{x}; } +template constexpr static auto operator "" _u96() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint96_t{x}; } +template constexpr static auto operator "" _u97() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint97_t{x}; } +template constexpr static auto operator "" _u98() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint98_t{x}; } +template constexpr static auto operator "" _u99() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint99_t{x}; } // 100--109 -template constexpr static auto operator "" _u100() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint100_t{x}; } -template constexpr static auto operator "" _u101() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint101_t{x}; } -template constexpr static auto operator "" _u102() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint102_t{x}; } -template constexpr static auto operator "" _u103() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint103_t{x}; } -template constexpr static auto operator "" _u104() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint104_t{x}; } -template constexpr static auto operator "" _u105() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint105_t{x}; } -template constexpr static auto operator "" _u106() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint106_t{x}; } -template constexpr static auto operator "" _u107() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint107_t{x}; } -template constexpr static auto operator "" _u108() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint108_t{x}; } -template constexpr static auto operator "" _u109() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint109_t{x}; } +template constexpr static auto operator "" _u100() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint100_t{x}; } +template constexpr static auto operator "" _u101() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint101_t{x}; } +template constexpr static auto operator "" _u102() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint102_t{x}; } +template constexpr static auto operator "" _u103() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint103_t{x}; } +template constexpr static auto operator "" _u104() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint104_t{x}; } +template constexpr static auto operator "" _u105() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint105_t{x}; } +template constexpr static auto operator "" _u106() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint106_t{x}; } +template constexpr static auto operator "" _u107() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint107_t{x}; } +template constexpr static auto operator "" _u108() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint108_t{x}; } +template constexpr static auto operator "" _u109() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint109_t{x}; } // 110--119 -template constexpr static auto operator "" _u110() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint110_t{x}; } -template constexpr static auto operator "" _u111() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint111_t{x}; } -template constexpr static auto operator "" _u112() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint112_t{x}; } -template constexpr static auto operator "" _u113() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint113_t{x}; } -template constexpr static auto operator "" _u114() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint114_t{x}; } -template constexpr static auto operator "" _u115() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint115_t{x}; } -template constexpr static auto operator "" _u116() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint116_t{x}; } -template constexpr static auto operator "" _u117() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint117_t{x}; } -template constexpr static auto operator "" _u118() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint118_t{x}; } -template constexpr static auto operator "" _u119() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint119_t{x}; } +template constexpr static auto operator "" _u110() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint110_t{x}; } +template constexpr static auto operator "" _u111() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint111_t{x}; } +template constexpr static auto operator "" _u112() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint112_t{x}; } +template constexpr static auto operator "" _u113() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint113_t{x}; } +template constexpr static auto operator "" _u114() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint114_t{x}; } +template constexpr static auto operator "" _u115() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint115_t{x}; } +template constexpr static auto operator "" _u116() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint116_t{x}; } +template constexpr static auto operator "" _u117() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint117_t{x}; } +template constexpr static auto operator "" _u118() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint118_t{x}; } +template constexpr static auto operator "" _u119() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint119_t{x}; } // 120--128 -template constexpr static auto operator "" _u120() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint120_t{x}; } -template constexpr static auto operator "" _u121() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint121_t{x}; } -template constexpr static auto operator "" _u122() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint122_t{x}; } -template constexpr static auto operator "" _u123() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint123_t{x}; } -template constexpr static auto operator "" _u124() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint124_t{x}; } -template constexpr static auto operator "" _u125() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint125_t{x}; } -template constexpr static auto operator "" _u126() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint126_t{x}; } -template constexpr static auto operator "" _u127() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint127_t{x}; } -template constexpr static auto operator "" _u128() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint128_t{x}; } -template constexpr static auto operator "" _u129() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint129_t{x}; } +template constexpr static auto operator "" _u120() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint120_t{x}; } +template constexpr static auto operator "" _u121() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint121_t{x}; } +template constexpr static auto operator "" _u122() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint122_t{x}; } +template constexpr static auto operator "" _u123() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint123_t{x}; } +template constexpr static auto operator "" _u124() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint124_t{x}; } +template constexpr static auto operator "" _u125() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint125_t{x}; } +template constexpr static auto operator "" _u126() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint126_t{x}; } +template constexpr static auto operator "" _u127() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint127_t{x}; } +template constexpr static auto operator "" _u128() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint128_t{x}; } +template constexpr static auto operator "" _u129() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint129_t{x}; } // 120--139 -template constexpr static auto operator "" _u130() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint130_t{x}; } -template constexpr static auto operator "" _u131() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint131_t{x}; } -template constexpr static auto operator "" _u132() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint132_t{x}; } -template constexpr static auto operator "" _u133() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint133_t{x}; } -template constexpr static auto operator "" _u134() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint134_t{x}; } -template constexpr static auto operator "" _u135() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint135_t{x}; } -template constexpr static auto operator "" _u136() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint136_t{x}; } -template constexpr static auto operator "" _u137() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint137_t{x}; } -template constexpr static auto operator "" _u138() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint138_t{x}; } -template constexpr static auto operator "" _u139() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint139_t{x}; } +template constexpr static auto operator "" _u130() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint130_t{x}; } +template constexpr static auto operator "" _u131() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint131_t{x}; } +template constexpr static auto operator "" _u132() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint132_t{x}; } +template constexpr static auto operator "" _u133() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint133_t{x}; } +template constexpr static auto operator "" _u134() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint134_t{x}; } +template constexpr static auto operator "" _u135() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint135_t{x}; } +template constexpr static auto operator "" _u136() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint136_t{x}; } +template constexpr static auto operator "" _u137() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint137_t{x}; } +template constexpr static auto operator "" _u138() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint138_t{x}; } +template constexpr static auto operator "" _u139() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint139_t{x}; } // 140--149 -template constexpr static auto operator "" _u140() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint140_t{x}; } -template constexpr static auto operator "" _u141() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint141_t{x}; } -template constexpr static auto operator "" _u142() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint142_t{x}; } -template constexpr static auto operator "" _u143() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint143_t{x}; } -template constexpr static auto operator "" _u144() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint144_t{x}; } -template constexpr static auto operator "" _u145() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint145_t{x}; } -template constexpr static auto operator "" _u146() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint146_t{x}; } -template constexpr static auto operator "" _u147() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint147_t{x}; } -template constexpr static auto operator "" _u148() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint148_t{x}; } -template constexpr static auto operator "" _u149() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint149_t{x}; } +template constexpr static auto operator "" _u140() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint140_t{x}; } +template constexpr static auto operator "" _u141() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint141_t{x}; } +template constexpr static auto operator "" _u142() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint142_t{x}; } +template constexpr static auto operator "" _u143() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint143_t{x}; } +template constexpr static auto operator "" _u144() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint144_t{x}; } +template constexpr static auto operator "" _u145() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint145_t{x}; } +template constexpr static auto operator "" _u146() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint146_t{x}; } +template constexpr static auto operator "" _u147() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint147_t{x}; } +template constexpr static auto operator "" _u148() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint148_t{x}; } +template constexpr static auto operator "" _u149() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint149_t{x}; } // 150--159 -template constexpr static auto operator "" _u150() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint150_t{x}; } -template constexpr static auto operator "" _u151() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint151_t{x}; } -template constexpr static auto operator "" _u152() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint152_t{x}; } -template constexpr static auto operator "" _u153() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint153_t{x}; } -template constexpr static auto operator "" _u154() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint154_t{x}; } -template constexpr static auto operator "" _u155() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint155_t{x}; } -template constexpr static auto operator "" _u156() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint156_t{x}; } -template constexpr static auto operator "" _u157() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint157_t{x}; } -template constexpr static auto operator "" _u158() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint158_t{x}; } -template constexpr static auto operator "" _u159() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint159_t{x}; } +template constexpr static auto operator "" _u150() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint150_t{x}; } +template constexpr static auto operator "" _u151() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint151_t{x}; } +template constexpr static auto operator "" _u152() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint152_t{x}; } +template constexpr static auto operator "" _u153() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint153_t{x}; } +template constexpr static auto operator "" _u154() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint154_t{x}; } +template constexpr static auto operator "" _u155() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint155_t{x}; } +template constexpr static auto operator "" _u156() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint156_t{x}; } +template constexpr static auto operator "" _u157() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint157_t{x}; } +template constexpr static auto operator "" _u158() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint158_t{x}; } +template constexpr static auto operator "" _u159() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint159_t{x}; } // 160--169 -template constexpr static auto operator "" _u160() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint160_t{x}; } -template constexpr static auto operator "" _u161() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint161_t{x}; } -template constexpr static auto operator "" _u162() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint162_t{x}; } -template constexpr static auto operator "" _u163() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint163_t{x}; } -template constexpr static auto operator "" _u164() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint164_t{x}; } -template constexpr static auto operator "" _u165() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint165_t{x}; } -template constexpr static auto operator "" _u166() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint166_t{x}; } -template constexpr static auto operator "" _u167() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint167_t{x}; } -template constexpr static auto operator "" _u168() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint168_t{x}; } -template constexpr static auto operator "" _u169() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint169_t{x}; } +template constexpr static auto operator "" _u160() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint160_t{x}; } +template constexpr static auto operator "" _u161() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint161_t{x}; } +template constexpr static auto operator "" _u162() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint162_t{x}; } +template constexpr static auto operator "" _u163() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint163_t{x}; } +template constexpr static auto operator "" _u164() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint164_t{x}; } +template constexpr static auto operator "" _u165() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint165_t{x}; } +template constexpr static auto operator "" _u166() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint166_t{x}; } +template constexpr static auto operator "" _u167() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint167_t{x}; } +template constexpr static auto operator "" _u168() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint168_t{x}; } +template constexpr static auto operator "" _u169() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint169_t{x}; } // 170-179 -template constexpr static auto operator "" _u170() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint170_t{x}; } -template constexpr static auto operator "" _u171() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint171_t{x}; } -template constexpr static auto operator "" _u172() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint172_t{x}; } -template constexpr static auto operator "" _u173() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint173_t{x}; } -template constexpr static auto operator "" _u174() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint174_t{x}; } -template constexpr static auto operator "" _u175() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint175_t{x}; } -template constexpr static auto operator "" _u176() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint176_t{x}; } -template constexpr static auto operator "" _u177() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint177_t{x}; } -template constexpr static auto operator "" _u178() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint178_t{x}; } -template constexpr static auto operator "" _u179() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint179_t{x}; } +template constexpr static auto operator "" _u170() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint170_t{x}; } +template constexpr static auto operator "" _u171() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint171_t{x}; } +template constexpr static auto operator "" _u172() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint172_t{x}; } +template constexpr static auto operator "" _u173() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint173_t{x}; } +template constexpr static auto operator "" _u174() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint174_t{x}; } +template constexpr static auto operator "" _u175() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint175_t{x}; } +template constexpr static auto operator "" _u176() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint176_t{x}; } +template constexpr static auto operator "" _u177() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint177_t{x}; } +template constexpr static auto operator "" _u178() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint178_t{x}; } +template constexpr static auto operator "" _u179() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint179_t{x}; } // 180--189 -template constexpr static auto operator "" _u180() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint180_t{x}; } -template constexpr static auto operator "" _u181() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint181_t{x}; } -template constexpr static auto operator "" _u182() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint182_t{x}; } -template constexpr static auto operator "" _u183() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint183_t{x}; } -template constexpr static auto operator "" _u184() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint184_t{x}; } -template constexpr static auto operator "" _u185() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint185_t{x}; } -template constexpr static auto operator "" _u186() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint186_t{x}; } -template constexpr static auto operator "" _u187() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint187_t{x}; } -template constexpr static auto operator "" _u188() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint188_t{x}; } -template constexpr static auto operator "" _u189() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint189_t{x}; } +template constexpr static auto operator "" _u180() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint180_t{x}; } +template constexpr static auto operator "" _u181() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint181_t{x}; } +template constexpr static auto operator "" _u182() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint182_t{x}; } +template constexpr static auto operator "" _u183() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint183_t{x}; } +template constexpr static auto operator "" _u184() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint184_t{x}; } +template constexpr static auto operator "" _u185() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint185_t{x}; } +template constexpr static auto operator "" _u186() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint186_t{x}; } +template constexpr static auto operator "" _u187() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint187_t{x}; } +template constexpr static auto operator "" _u188() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint188_t{x}; } +template constexpr static auto operator "" _u189() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint189_t{x}; } // 190--199 -template constexpr static auto operator "" _u190() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint190_t{x}; } -template constexpr static auto operator "" _u191() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint191_t{x}; } -template constexpr static auto operator "" _u192() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint192_t{x}; } -template constexpr static auto operator "" _u193() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint193_t{x}; } -template constexpr static auto operator "" _u194() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint194_t{x}; } -template constexpr static auto operator "" _u195() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint195_t{x}; } -template constexpr static auto operator "" _u196() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint196_t{x}; } -template constexpr static auto operator "" _u197() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint197_t{x}; } -template constexpr static auto operator "" _u198() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint198_t{x}; } -template constexpr static auto operator "" _u199() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint199_t{x}; } +template constexpr static auto operator "" _u190() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint190_t{x}; } +template constexpr static auto operator "" _u191() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint191_t{x}; } +template constexpr static auto operator "" _u192() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint192_t{x}; } +template constexpr static auto operator "" _u193() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint193_t{x}; } +template constexpr static auto operator "" _u194() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint194_t{x}; } +template constexpr static auto operator "" _u195() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint195_t{x}; } +template constexpr static auto operator "" _u196() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint196_t{x}; } +template constexpr static auto operator "" _u197() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint197_t{x}; } +template constexpr static auto operator "" _u198() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint198_t{x}; } +template constexpr static auto operator "" _u199() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint199_t{x}; } // 200--209 -template constexpr static auto operator "" _u200() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint200_t{x}; } -template constexpr static auto operator "" _u201() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint201_t{x}; } -template constexpr static auto operator "" _u202() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint202_t{x}; } -template constexpr static auto operator "" _u203() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint203_t{x}; } -template constexpr static auto operator "" _u204() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint204_t{x}; } -template constexpr static auto operator "" _u205() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint205_t{x}; } -template constexpr static auto operator "" _u206() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint206_t{x}; } -template constexpr static auto operator "" _u207() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint207_t{x}; } -template constexpr static auto operator "" _u208() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint208_t{x}; } -template constexpr static auto operator "" _u209() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint209_t{x}; } +template constexpr static auto operator "" _u200() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint200_t{x}; } +template constexpr static auto operator "" _u201() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint201_t{x}; } +template constexpr static auto operator "" _u202() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint202_t{x}; } +template constexpr static auto operator "" _u203() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint203_t{x}; } +template constexpr static auto operator "" _u204() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint204_t{x}; } +template constexpr static auto operator "" _u205() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint205_t{x}; } +template constexpr static auto operator "" _u206() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint206_t{x}; } +template constexpr static auto operator "" _u207() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint207_t{x}; } +template constexpr static auto operator "" _u208() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint208_t{x}; } +template constexpr static auto operator "" _u209() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint209_t{x}; } // 210--219 -template constexpr static auto operator "" _u210() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint210_t{x}; } -template constexpr static auto operator "" _u211() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint211_t{x}; } -template constexpr static auto operator "" _u212() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint212_t{x}; } -template constexpr static auto operator "" _u213() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint213_t{x}; } -template constexpr static auto operator "" _u214() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint214_t{x}; } -template constexpr static auto operator "" _u215() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint215_t{x}; } -template constexpr static auto operator "" _u216() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint216_t{x}; } -template constexpr static auto operator "" _u217() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint217_t{x}; } -template constexpr static auto operator "" _u218() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint218_t{x}; } -template constexpr static auto operator "" _u219() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint219_t{x}; } +template constexpr static auto operator "" _u210() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint210_t{x}; } +template constexpr static auto operator "" _u211() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint211_t{x}; } +template constexpr static auto operator "" _u212() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint212_t{x}; } +template constexpr static auto operator "" _u213() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint213_t{x}; } +template constexpr static auto operator "" _u214() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint214_t{x}; } +template constexpr static auto operator "" _u215() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint215_t{x}; } +template constexpr static auto operator "" _u216() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint216_t{x}; } +template constexpr static auto operator "" _u217() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint217_t{x}; } +template constexpr static auto operator "" _u218() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint218_t{x}; } +template constexpr static auto operator "" _u219() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint219_t{x}; } // 220--229 -template constexpr static auto operator "" _u220() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint220_t{x}; } -template constexpr static auto operator "" _u221() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint221_t{x}; } -template constexpr static auto operator "" _u222() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint222_t{x}; } -template constexpr static auto operator "" _u223() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint223_t{x}; } -template constexpr static auto operator "" _u224() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint224_t{x}; } -template constexpr static auto operator "" _u225() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint225_t{x}; } -template constexpr static auto operator "" _u226() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint226_t{x}; } -template constexpr static auto operator "" _u227() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint227_t{x}; } -template constexpr static auto operator "" _u228() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint228_t{x}; } -template constexpr static auto operator "" _u229() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint229_t{x}; } +template constexpr static auto operator "" _u220() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint220_t{x}; } +template constexpr static auto operator "" _u221() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint221_t{x}; } +template constexpr static auto operator "" _u222() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint222_t{x}; } +template constexpr static auto operator "" _u223() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint223_t{x}; } +template constexpr static auto operator "" _u224() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint224_t{x}; } +template constexpr static auto operator "" _u225() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint225_t{x}; } +template constexpr static auto operator "" _u226() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint226_t{x}; } +template constexpr static auto operator "" _u227() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint227_t{x}; } +template constexpr static auto operator "" _u228() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint228_t{x}; } +template constexpr static auto operator "" _u229() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint229_t{x}; } // 230--239 -template constexpr static auto operator "" _u230() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint230_t{x}; } -template constexpr static auto operator "" _u231() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint231_t{x}; } -template constexpr static auto operator "" _u232() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint232_t{x}; } -template constexpr static auto operator "" _u233() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint233_t{x}; } -template constexpr static auto operator "" _u234() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint234_t{x}; } -template constexpr static auto operator "" _u235() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint235_t{x}; } -template constexpr static auto operator "" _u236() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint236_t{x}; } -template constexpr static auto operator "" _u237() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint237_t{x}; } -template constexpr static auto operator "" _u238() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint238_t{x}; } -template constexpr static auto operator "" _u239() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint239_t{x}; } +template constexpr static auto operator "" _u230() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint230_t{x}; } +template constexpr static auto operator "" _u231() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint231_t{x}; } +template constexpr static auto operator "" _u232() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint232_t{x}; } +template constexpr static auto operator "" _u233() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint233_t{x}; } +template constexpr static auto operator "" _u234() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint234_t{x}; } +template constexpr static auto operator "" _u235() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint235_t{x}; } +template constexpr static auto operator "" _u236() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint236_t{x}; } +template constexpr static auto operator "" _u237() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint237_t{x}; } +template constexpr static auto operator "" _u238() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint238_t{x}; } +template constexpr static auto operator "" _u239() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint239_t{x}; } // 240--249 -template constexpr static auto operator "" _u240() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint240_t{x}; } -template constexpr static auto operator "" _u241() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint241_t{x}; } -template constexpr static auto operator "" _u242() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint242_t{x}; } -template constexpr static auto operator "" _u243() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint243_t{x}; } -template constexpr static auto operator "" _u244() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint244_t{x}; } -template constexpr static auto operator "" _u245() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint245_t{x}; } -template constexpr static auto operator "" _u246() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint246_t{x}; } -template constexpr static auto operator "" _u247() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint247_t{x}; } -template constexpr static auto operator "" _u248() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint248_t{x}; } -template constexpr static auto operator "" _u249() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint249_t{x}; } +template constexpr static auto operator "" _u240() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint240_t{x}; } +template constexpr static auto operator "" _u241() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint241_t{x}; } +template constexpr static auto operator "" _u242() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint242_t{x}; } +template constexpr static auto operator "" _u243() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint243_t{x}; } +template constexpr static auto operator "" _u244() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint244_t{x}; } +template constexpr static auto operator "" _u245() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint245_t{x}; } +template constexpr static auto operator "" _u246() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint246_t{x}; } +template constexpr static auto operator "" _u247() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint247_t{x}; } +template constexpr static auto operator "" _u248() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint248_t{x}; } +template constexpr static auto operator "" _u249() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint249_t{x}; } // 250--256 -template constexpr static auto operator "" _u250() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint250_t{x}; } -template constexpr static auto operator "" _u251() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint251_t{x}; } -template constexpr static auto operator "" _u252() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint252_t{x}; } -template constexpr static auto operator "" _u253() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint253_t{x}; } -template constexpr static auto operator "" _u254() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint254_t{x}; } -template constexpr static auto operator "" _u255() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint255_t{x}; } -template constexpr static auto operator "" _u256() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::modints::modint256_t{x}; } +template constexpr static auto operator "" _u250() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint250_t{x}; } +template constexpr static auto operator "" _u251() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint251_t{x}; } +template constexpr static auto operator "" _u252() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint252_t{x}; } +template constexpr static auto operator "" _u253() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint253_t{x}; } +template constexpr static auto operator "" _u254() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint254_t{x}; } +template constexpr static auto operator "" _u255() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint255_t{x}; } +template constexpr static auto operator "" _u256() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::modints::modint256_t{x}; } } // namespace modints @@ -1362,14 +1388,23 @@ class numeric_limits> = std::numeric_limits::integral_type>::traps; static constexpr bool tinyness_before = false; + HEDLEY_NO_THROW static constexpr dpf::modint min() noexcept { return dpf::modint{0}; } + HEDLEY_NO_THROW static constexpr dpf::modint lowest() noexcept { return dpf::modint{0}; } + HEDLEY_NO_THROW static constexpr dpf::modint max() noexcept { return ~dpf::modint{0}; } + HEDLEY_NO_THROW static constexpr dpf::modint epsilon() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::modint round_error() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::modint infinity() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::modint quiet_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::modint signaling_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr dpf::modint denorm_min() noexcept { return 0; } }; diff --git a/include/dpf/nyble.hpp b/include/dpf/nyble.hpp index 42bed9c..315ce00 100644 --- a/include/dpf/nyble.hpp +++ b/include/dpf/nyble.hpp @@ -157,6 +157,7 @@ template <> struct make_from_integral_value { using integral_type = std::uint8_t; + HEDLEY_NO_THROW constexpr dpf::nyble operator()(integral_type val) const noexcept { return dpf::to_nyble(val); @@ -189,8 +190,11 @@ class numeric_limits : public numeric_limits public: static constexpr int digits = 4; static constexpr int digits10 = 1; + HEDLEY_NO_THROW static constexpr dpf::nyble min() noexcept { return dpf::nyble::zero; } + HEDLEY_NO_THROW static constexpr dpf::nyble max() noexcept { return dpf::nyble{15}; } + HEDLEY_NO_THROW static constexpr dpf::nyble lowest() noexcept { return min(); } }; diff --git a/include/dpf/output_buffer.hpp b/include/dpf/output_buffer.hpp index ab796f2..764ba3c 100644 --- a/include/dpf/output_buffer.hpp +++ b/include/dpf/output_buffer.hpp @@ -1,6 +1,14 @@ /// @file dpf/output_buffer.hpp -/// @brief -/// @details +/// @brief Move-only storage for shares written by multi-point evaluation. +/// @details Slot type follows the key. A `party_key` leaf buffer holds +/// `subtractive_share`s; a comparison buffer holds +/// `additive_share`s. `bit`, `twobit`, and `nyble` slots are packed. +/// Trivially default-constructible slots are left uninitialized +/// because evaluation overwrites every slot it is responsible for. +/// +/// `eval_interval` and recipe `eval_sequence` take the buffer by +/// non-const reference. The returned iterable refers into it. +/// @snippet evaluation/output_buffers.cpp output-buffer /// @author Ryan Henry /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license; @@ -9,6 +17,8 @@ #ifndef LIBDPF_INCLUDE_DPF_OUTPUT_BUFFER_HPP__ #define LIBDPF_INCLUDE_DPF_OUTPUT_BUFFER_HPP__ +#include "hedley/hedley.h" + #include #include #include @@ -77,9 +87,12 @@ class output_buffer_allocator : public aligned_allocator using other = output_buffer_allocator; }; + HEDLEY_NO_THROW output_buffer_allocator() noexcept = default; + HEDLEY_NO_THROW output_buffer_allocator(const output_buffer_allocator &) noexcept = default; template + HEDLEY_NO_THROW output_buffer_allocator(const output_buffer_allocator &) noexcept {} template @@ -100,6 +113,7 @@ class output_buffer_allocator : public aligned_allocator } template + HEDLEY_NO_THROW void destroy(U * p) noexcept { if constexpr (!std::is_trivially_destructible_v) @@ -110,6 +124,7 @@ class output_buffer_allocator : public aligned_allocator }; template +HEDLEY_NO_THROW constexpr bool operator==(const output_buffer_allocator &, const output_buffer_allocator &) noexcept { @@ -117,12 +132,15 @@ constexpr bool operator==(const output_buffer_allocator &, } template +HEDLEY_NO_THROW constexpr bool operator!=(const output_buffer_allocator & lhs, const output_buffer_allocator & rhs) noexcept { return !(lhs == rhs); } +/// Move-only vector of `T`. Copy construction and copy assignment are +/// deleted. `at`, `operator[]`, `data`, iterators, and `size` are public. template class output_buffer final @@ -135,13 +153,17 @@ class output_buffer final using iterator = typename vector::iterator; using const_iterator = typename vector::const_iterator; using size_type = typename vector::size_type; + HEDLEY_NO_THROW output_buffer() noexcept = default; explicit output_buffer(size_type size) : vector(size) { } + HEDLEY_NO_THROW output_buffer(output_buffer &&) noexcept = default; output_buffer(const output_buffer &) = delete; + HEDLEY_NO_THROW output_buffer & operator=(output_buffer &&) noexcept = default; output_buffer & operator=(const output_buffer &) = delete; - ~output_buffer() = default; + HEDLEY_NO_THROW + ~output_buffer() noexcept = default; // "selectively public" inheritance using vector::at; @@ -161,11 +183,14 @@ class output_buffer : public dpf::dynamic_bit_array<> using size_type = typename dpf::dynamic_bit_array<>::size_type; public: explicit output_buffer(size_type size) : dynamic_bit_array(size) { } + HEDLEY_NO_THROW output_buffer(output_buffer &&) noexcept = default; output_buffer(const output_buffer &) = delete; + HEDLEY_NO_THROW output_buffer & operator=(output_buffer &&) noexcept = default; output_buffer & operator=(const output_buffer &) = delete; - ~output_buffer() = default; + HEDLEY_NO_THROW + ~output_buffer() noexcept = default; }; template <> @@ -175,11 +200,14 @@ class output_buffer : public dpf::dynamic_packed_array public: using size_type = typename base::size_type; explicit output_buffer(size_type size) : base(size) { } + HEDLEY_NO_THROW output_buffer(output_buffer &&) noexcept = default; output_buffer(const output_buffer &) = delete; + HEDLEY_NO_THROW output_buffer & operator=(output_buffer &&) noexcept = default; output_buffer & operator=(const output_buffer &) = delete; - ~output_buffer() = default; + HEDLEY_NO_THROW + ~output_buffer() noexcept = default; }; template <> @@ -189,11 +217,14 @@ class output_buffer : public dpf::dynamic_packed_array public: using size_type = typename base::size_type; explicit output_buffer(size_type size) : base(size) { } + HEDLEY_NO_THROW output_buffer(output_buffer &&) noexcept = default; output_buffer(const output_buffer &) = delete; + HEDLEY_NO_THROW output_buffer & operator=(output_buffer &&) noexcept = default; output_buffer & operator=(const output_buffer &) = delete; - ~output_buffer() = default; + HEDLEY_NO_THROW + ~output_buffer() noexcept = default; }; #define LIBDPF_PACKED_SHARE_BUFFER(LANE, PARTY) \ @@ -205,11 +236,14 @@ class output_buffer> public: \ using size_type = typename base::size_type; \ explicit output_buffer(size_type size) : base(size) {} \ + HEDLEY_NO_THROW \ output_buffer(output_buffer &&) noexcept = default; \ output_buffer(const output_buffer &) = delete; \ + HEDLEY_NO_THROW \ output_buffer & operator=(output_buffer &&) noexcept = default; \ output_buffer & operator=(const output_buffer &) = delete; \ - ~output_buffer() = default; \ + HEDLEY_NO_THROW \ + ~output_buffer() noexcept = default; \ }; LIBDPF_PACKED_SHARE_BUFFER(dpf::twobit, 0); @@ -232,12 +266,14 @@ class output_buffer> output_buffer(const output_buffer &) = delete; \ output_buffer & operator=(output_buffer &&) noexcept = default; \ output_buffer & operator=(const output_buffer &) = delete; \ - ~output_buffer() = default; \ + ~output_buffer() noexcept = default; \ }; LIBDPF_BIT_SHARE_BUFFER(0); LIBDPF_BIT_SHARE_BUFFER(1); #undef LIBDPF_BIT_SHARE_BUFFER +/// Buffer sized for the closed interval `[from, to]` of output `I`. +/// On a `party_key`, elements are subtractive shares of that output. template @@ -247,9 +283,6 @@ auto make_output_buffer_for_interval(InputT from, InputT to) using output_type = typename DpfKey::concrete_output_type; using buffer_elem = leaf_buffer_elem_t; - utils::flip_msb_if_signed_integral(from); - utils::flip_msb_if_signed_integral(to); - std::size_t nodes_in_interval = utils::get_leafnodes_in_output_interval(from, to); return dpf::output_buffer(nodes_in_interval*dpf_type::outputs_per_leaf); } @@ -285,6 +318,7 @@ inline auto make_output_buffer_for_interval(const DpfKey &, InputT from, InputT return make_output_buffer_for_interval(from, to); } +/// Buffer sized for every input of output `I`. template auto make_output_buffer_for_full() diff --git a/include/dpf/packed_array.hpp b/include/dpf/packed_array.hpp index 11ba2b5..52ac4b2 100644 --- a/include/dpf/packed_array.hpp +++ b/include/dpf/packed_array.hpp @@ -50,15 +50,18 @@ class dynamic_packed_array class lane_ref { public: + HEDLEY_NO_THROW lane_ref(word_type * word, unsigned shift) noexcept : word_{word}, shift_{shift} {} + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE operator LaneT() const noexcept { return static_cast((*word_ >> shift_) & lane_mask); } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE lane_ref & operator=(LaneT value) noexcept { @@ -69,23 +72,28 @@ class dynamic_packed_array return *this; } + HEDLEY_NO_THROW lane_ref & operator=(const lane_ref & other) noexcept { return (*this = static_cast(other)); } + HEDLEY_NO_THROW friend bool operator==(lane_ref lhs, LaneT rhs) noexcept { return static_cast(lhs) == rhs; } + HEDLEY_NO_THROW friend bool operator==(LaneT lhs, lane_ref rhs) noexcept { return rhs == lhs; } + HEDLEY_NO_THROW friend bool operator!=(lane_ref lhs, LaneT rhs) noexcept { return !(lhs == rhs); } + HEDLEY_NO_THROW friend bool operator!=(LaneT lhs, lane_ref rhs) noexcept { return !(rhs == lhs); @@ -108,25 +116,33 @@ class dynamic_packed_array using pointer = void; using reference = lane_ref; + HEDLEY_NO_THROW iterator() noexcept = default; + HEDLEY_NO_THROW iterator(word_type * data, size_type index) noexcept : data_{data}, index_{index} {} + HEDLEY_NO_THROW lane_ref operator*() const noexcept { return ref_at(index_); } + HEDLEY_NO_THROW lane_ref operator[](difference_type n) const noexcept { return ref_at(static_cast( static_cast(index_) + n)); } + HEDLEY_NO_THROW iterator & operator++() noexcept { ++index_; return *this; } + HEDLEY_NO_THROW iterator operator++(int) noexcept { iterator prev = *this; ++*this; return prev; } + HEDLEY_NO_THROW iterator & operator--() noexcept { --index_; return *this; } + HEDLEY_NO_THROW iterator operator--(int) noexcept { iterator prev = *this; @@ -134,53 +150,66 @@ class dynamic_packed_array return prev; } + HEDLEY_NO_THROW iterator & operator+=(difference_type n) noexcept { index_ = static_cast( static_cast(index_) + n); return *this; } + HEDLEY_NO_THROW iterator & operator-=(difference_type n) noexcept { return *this += -n; } + HEDLEY_NO_THROW friend iterator operator+(iterator it, difference_type n) noexcept { it += n; return it; } + HEDLEY_NO_THROW friend iterator operator+(difference_type n, iterator it) noexcept { return it + n; } + HEDLEY_NO_THROW friend iterator operator-(iterator it, difference_type n) noexcept { it -= n; return it; } + HEDLEY_NO_THROW friend difference_type operator-(iterator a, iterator b) noexcept { return static_cast(a.index_) - static_cast(b.index_); } + HEDLEY_NO_THROW friend bool operator==(iterator a, iterator b) noexcept { return a.index_ == b.index_; } + HEDLEY_NO_THROW friend bool operator!=(iterator a, iterator b) noexcept { return !(a == b); } + HEDLEY_NO_THROW friend bool operator<(iterator a, iterator b) noexcept { return a.index_ < b.index_; } + HEDLEY_NO_THROW friend bool operator>(iterator a, iterator b) noexcept { return b < a; } + HEDLEY_NO_THROW friend bool operator<=(iterator a, iterator b) noexcept { return !(b < a); } + HEDLEY_NO_THROW friend bool operator>=(iterator a, iterator b) noexcept { return !(a < b); } private: + HEDLEY_NO_THROW lane_ref ref_at(size_type index) const noexcept { const size_type bit = index * lane_bits; @@ -213,12 +242,14 @@ class dynamic_packed_array dynamic_packed_array(const dynamic_packed_array &) = delete; dynamic_packed_array & operator=(const dynamic_packed_array &) = delete; + HEDLEY_NO_THROW dynamic_packed_array(dynamic_packed_array && other) noexcept : nlanes_{std::exchange(other.nlanes_, 0)}, nwords_{std::exchange(other.nwords_, 0)}, data_{std::move(other.data_)} {} + HEDLEY_NO_THROW dynamic_packed_array & operator=(dynamic_packed_array && other) noexcept { if (this != &other) @@ -232,13 +263,19 @@ class dynamic_packed_array ~dynamic_packed_array() = default; + HEDLEY_NO_THROW size_type size() const noexcept { return nlanes_; } + HEDLEY_NO_THROW bool empty() const noexcept { return nlanes_ == 0; } + HEDLEY_NO_THROW size_type data_length() const noexcept { return nwords_; } + HEDLEY_NO_THROW word_type * data() noexcept { return data_.get(); } + HEDLEY_NO_THROW const word_type * data() const noexcept { return data_.get(); } + HEDLEY_NO_THROW LaneT operator[](size_type i) const noexcept { assert(i < nlanes_); @@ -247,6 +284,7 @@ class dynamic_packed_array return static_cast((data_[bit / 64u] >> shift) & lane_mask); } + HEDLEY_NO_THROW lane_ref operator[](size_type i) noexcept { assert(i < nlanes_); @@ -255,11 +293,17 @@ class dynamic_packed_array static_cast(bit % 64u)); } + HEDLEY_NO_THROW iterator begin() noexcept { return iterator{data(), 0}; } + HEDLEY_NO_THROW iterator end() noexcept { return iterator{data(), nlanes_}; } + HEDLEY_NO_THROW iterator begin() const noexcept { return iterator{data_.get(), 0}; } + HEDLEY_NO_THROW iterator end() const noexcept { return iterator{data_.get(), nlanes_}; } + HEDLEY_NO_THROW iterator cbegin() const noexcept { return begin(); } + HEDLEY_NO_THROW iterator cend() const noexcept { return end(); } private: @@ -284,25 +328,30 @@ class packed_share_output : public dynamic_packed_array class reference { public: + HEDLEY_NO_THROW explicit reference(typename lanes::reference lane) noexcept : lane_{lane} {} + HEDLEY_NO_THROW operator share_type() const noexcept { return share_type::from_raw(static_cast(lane_)); } + HEDLEY_NO_THROW reference & operator=(const share_type & share) noexcept { lane_ = share.raw(); return *this; } + HEDLEY_NO_THROW reference & operator=(LaneT value) noexcept { lane_ = value; return *this; } + HEDLEY_NO_THROW reference & operator=(const reference & other) noexcept { return (*this = static_cast(other)); @@ -321,48 +370,68 @@ class packed_share_output : public dynamic_packed_array using pointer = void; using reference = share_type; + HEDLEY_NO_THROW iterator() noexcept = default; + HEDLEY_NO_THROW explicit iterator(typename lanes::iterator it) noexcept : it_{it} {} + HEDLEY_NO_THROW share_type operator*() const noexcept { return share_type::from_raw(static_cast(*it_)); } + HEDLEY_NO_THROW share_type operator[](difference_type n) const noexcept { return share_type::from_raw(static_cast(it_[n])); } + HEDLEY_NO_THROW iterator & operator++() noexcept { ++it_; return *this; } + HEDLEY_NO_THROW iterator operator++(int) noexcept { iterator p = *this; ++*this; return p; } + HEDLEY_NO_THROW iterator & operator--() noexcept { --it_; return *this; } + HEDLEY_NO_THROW iterator operator--(int) noexcept { iterator p = *this; --*this; return p; } + HEDLEY_NO_THROW iterator & operator+=(difference_type n) noexcept { it_ += n; return *this; } + HEDLEY_NO_THROW iterator & operator-=(difference_type n) noexcept { it_ -= n; return *this; } + HEDLEY_NO_THROW friend iterator operator+(iterator it, difference_type n) noexcept { it += n; return it; } + HEDLEY_NO_THROW friend iterator operator+(difference_type n, iterator it) noexcept { return it + n; } + HEDLEY_NO_THROW friend iterator operator-(iterator it, difference_type n) noexcept { it -= n; return it; } + HEDLEY_NO_THROW friend difference_type operator-(iterator a, iterator b) noexcept { return a.it_ - b.it_; } + HEDLEY_NO_THROW friend bool operator==(iterator a, iterator b) noexcept { return a.it_ == b.it_; } + HEDLEY_NO_THROW friend bool operator!=(iterator a, iterator b) noexcept { return !(a == b); } + HEDLEY_NO_THROW friend bool operator<(iterator a, iterator b) noexcept { return a.it_ < b.it_; } + HEDLEY_NO_THROW friend bool operator>(iterator a, iterator b) noexcept { return b < a; } + HEDLEY_NO_THROW friend bool operator<=(iterator a, iterator b) noexcept { return !(b < a); } + HEDLEY_NO_THROW friend bool operator>=(iterator a, iterator b) noexcept { return !(a < b); } private: @@ -375,7 +444,9 @@ class packed_share_output : public dynamic_packed_array packed_share_output(const packed_share_output &) = delete; packed_share_output & operator=(const packed_share_output &) = delete; + HEDLEY_NO_THROW packed_share_output(packed_share_output &&) noexcept = default; + HEDLEY_NO_THROW packed_share_output & operator=(packed_share_output &&) noexcept = default; ~packed_share_output() = default; @@ -383,26 +454,34 @@ class packed_share_output : public dynamic_packed_array using lanes::empty; using lanes::size; + HEDLEY_NO_THROW reference operator[](size_type i) noexcept { return reference{lanes::operator[](i)}; } + HEDLEY_NO_THROW share_type operator[](size_type i) const noexcept { return share_type::from_raw(lanes::operator[](i)); } + HEDLEY_NO_THROW iterator begin() noexcept { return iterator{lanes::begin()}; } + HEDLEY_NO_THROW iterator end() noexcept { return iterator{lanes::end()}; } + HEDLEY_NO_THROW iterator begin() const noexcept { return iterator{typename lanes::iterator{this->data(), 0}}; } + HEDLEY_NO_THROW iterator end() const noexcept { return iterator{typename lanes::iterator{this->data(), this->size()}}; } + HEDLEY_NO_THROW iterator cbegin() const noexcept { return begin(); } + HEDLEY_NO_THROW iterator cend() const noexcept { return end(); } }; diff --git a/include/dpf/packed_lane_arithmetic.hpp b/include/dpf/packed_lane_arithmetic.hpp index e80f0aa..63e9611 100644 --- a/include/dpf/packed_lane_arithmetic.hpp +++ b/include/dpf/packed_lane_arithmetic.hpp @@ -34,6 +34,7 @@ namespace detail template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW unsigned lane_at(unsigned byte, unsigned shift) noexcept { constexpr unsigned mask = (1u << Bits) - 1u; @@ -67,6 +68,7 @@ NodeT apply_bytes(const NodeT & a, const NodeT & b, Op op) noexcept return out; } +HEDLEY_NO_THROW inline simde__m128i nibble_lut(const unsigned char lut[16]) noexcept { simde__m128i table; @@ -74,6 +76,7 @@ inline simde__m128i nibble_lut(const unsigned char lut[16]) noexcept return table; } +HEDLEY_NO_THROW inline simde__m256i nibble_lut256(const unsigned char lut[16]) noexcept { alignas(32) unsigned char both[32]; @@ -84,6 +87,7 @@ inline simde__m256i nibble_lut256(const unsigned char lut[16]) noexcept return table; } +HEDLEY_NO_THROW inline void fill_epi2_mul_lut(unsigned k, unsigned char lut[16]) noexcept { k &= 3u; @@ -95,6 +99,7 @@ inline void fill_epi2_mul_lut(unsigned k, unsigned char lut[16]) noexcept } } +HEDLEY_NO_THROW inline void fill_epi4_mul_lut(unsigned k, unsigned char lut[16]) noexcept { k &= 0x0fu; @@ -104,6 +109,7 @@ inline void fill_epi4_mul_lut(unsigned k, unsigned char lut[16]) noexcept } } +HEDLEY_NO_THROW inline simde__m128i shuffle_nibbles(simde__m128i table, simde__m128i a) noexcept { const auto m = simde_mm_set1_epi8(0x0f); @@ -114,6 +120,7 @@ inline simde__m128i shuffle_nibbles(simde__m128i table, simde__m128i a) noexcept simde_mm_slli_epi16(simde_mm_and_si128(hi, m), 4)); } +HEDLEY_NO_THROW inline simde__m256i shuffle_nibbles(simde__m256i table, simde__m256i a) noexcept { const auto m = simde_mm256_set1_epi8(0x0f); @@ -126,6 +133,7 @@ inline simde__m256i shuffle_nibbles(simde__m256i table, simde__m256i a) noexcept /// Low nibble of every byte, product mod 16. Even and odd bytes are split /// so a product in one byte cannot land in the next. +HEDLEY_NO_THROW inline simde__m128i mul_low_nibbles(simde__m128i a, simde__m128i b) noexcept { const auto lane = simde_mm_set1_epi16(0x000f); @@ -138,6 +146,7 @@ inline simde__m128i mul_low_nibbles(simde__m128i a, simde__m128i b) noexcept return simde_mm_or_si128(pe, simde_mm_slli_epi16(po, 8)); } +HEDLEY_NO_THROW inline simde__m256i mul_low_nibbles(simde__m256i a, simde__m256i b) noexcept { const auto lane = simde_mm256_set1_epi16(0x000f); diff --git a/include/dpf/parallel_bit_iterable.hpp b/include/dpf/parallel_bit_iterable.hpp index 9d81ff3..9d3db09 100644 --- a/include/dpf/parallel_bit_iterable.hpp +++ b/include/dpf/parallel_bit_iterable.hpp @@ -143,9 +143,11 @@ class parallel_const_bit_iterator using const_reference = const value_type &; using pointer = std::add_pointer_t; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr parallel_const_bit_iterator(parallel_const_bit_iterator &&) noexcept = default; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr parallel_const_bit_iterator(const parallel_const_bit_iterator &) noexcept = default; @@ -273,6 +275,7 @@ class parallel_const_bit_iterator std::make_index_sequence()); } + HEDLEY_NO_THROW explicit constexpr parallel_const_bit_iterator( const word_pointer_array & arr) noexcept : iter_{arr}, @@ -293,7 +296,9 @@ class parallel_const_bit_iterator simde_type vec_mask_; simde_array all_vecs_; + HEDLEY_NO_THROW friend parallel_const_bit_iterator parallel_bit_iterable::begin() const noexcept; + HEDLEY_NO_THROW friend parallel_const_bit_iterator parallel_bit_iterable::end() const noexcept; }; // class dpf::parallel_const_bit_iterator @@ -302,6 +307,7 @@ template HEDLEY_PURE HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW auto batch_of(Iter it) noexcept { return dpf::parallel_bit_iterable{it}; @@ -311,6 +317,7 @@ template HEDLEY_PURE HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW auto batch_of(const dpf::bit_array_base & t, const Ts & ...ts) noexcept { return dpf::parallel_bit_iterable<1+sizeof...(Ts), ChildT>{t, ts...}; diff --git a/include/dpf/parallel_bit_iterable_helpers.hpp b/include/dpf/parallel_bit_iterable_helpers.hpp index 1ad74bd..4653486 100644 --- a/include/dpf/parallel_bit_iterable_helpers.hpp +++ b/include/dpf/parallel_bit_iterable_helpers.hpp @@ -75,10 +75,12 @@ HEDLEY_PRAGMA(GCC diagnostic pop) static constexpr auto left_shift = simde_mm256_slli_epi64; static constexpr auto right_shift = simde_mm256_srli_epi64; static constexpr auto bit_and = simde_mm256_and_si256; + HEDLEY_NO_THROW static auto get_mask() noexcept { return simde_mm256_set1_epi64x(1); } + HEDLEY_NO_THROW static simde_array build_vecs(const word_type * cur_word, std::size_t nwords) noexcept { return { loadu_word_vec(cur_word, nwords, 0) }; @@ -104,10 +106,12 @@ HEDLEY_PRAGMA(GCC diagnostic pop) static constexpr auto left_shift = simde_mm256_slli_epi64; static constexpr auto right_shift = simde_mm256_srli_epi64; static constexpr auto bit_and = simde_mm256_and_si256; + HEDLEY_NO_THROW static auto get_mask() noexcept { return simde_mm256_set1_epi32(1); } + HEDLEY_NO_THROW static simde_array build_vecs(const typename dpf::bit_array_base::word_type * cur_word, std::size_t nwords) noexcept { @@ -149,10 +153,12 @@ HEDLEY_PRAGMA(GCC diagnostic pop) static constexpr auto left_shift = simde_mm256_slli_epi64; static constexpr auto right_shift = simde_mm256_srli_epi64; static constexpr auto bit_and = simde_mm256_and_si256; + HEDLEY_NO_THROW static auto get_mask() noexcept { return simde_mm256_set1_epi16(1); } + HEDLEY_NO_THROW static simde_array build_vecs(const typename dpf::bit_array_base::word_type * cur_word, std::size_t nwords) noexcept { @@ -215,10 +221,12 @@ HEDLEY_PRAGMA(GCC diagnostic pop) static constexpr auto left_shift = simde_mm256_slli_epi64; static constexpr auto right_shift = simde_mm256_srli_epi64; static constexpr auto bit_and = simde_mm256_and_si256; + HEDLEY_NO_THROW static auto get_mask() noexcept { return simde_mm256_set1_epi8(1); } + HEDLEY_NO_THROW static simde_array build_vecs(const typename dpf::bit_array_base::word_type * cur_word, std::size_t nwords) noexcept { diff --git a/include/dpf/path_memoizer.hpp b/include/dpf/path_memoizer.hpp index 9cf00bb..5a4c7e0 100644 --- a/include/dpf/path_memoizer.hpp +++ b/include/dpf/path_memoizer.hpp @@ -1,6 +1,13 @@ /// @file dpf/path_memoizer.hpp -/// @brief -/// @details +/// @brief Workspaces that resume a root-to-leaf DPF walk. +/// @details `basic_path_memoizer` keeps one interior node per level. The next +/// `eval_point` recomputes the suffix after the common prefix with +/// the previous input. `nonmemoizing_path_memoizer` keeps a single +/// node and starts at the root on every call. +/// +/// Pass a mutable lvalue to `eval_point`. The factories unwrap +/// `party_key`, so a memoizer built from either party's type +/// accepts both parties. A different root restarts the path. /// @author Ryan Henry /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) @@ -44,13 +51,21 @@ struct path_memoizer_base using return_type = ReturnT; using iterator_type = return_type; + HEDLEY_NO_THROW virtual std::size_t assign_x(const dpf_type &, input_type) noexcept = 0; + HEDLEY_NO_THROW virtual node_type & operator[](std::size_t) noexcept = 0; + HEDLEY_NO_THROW virtual return_type begin() const noexcept = 0; + HEDLEY_NO_THROW virtual return_type end() const noexcept = 0; }; +/// One interior node per level. `assign_x` returns the first level that the +/// next walk must recompute. `filled_to` is the deepest level already +/// written for the current input. Callers pass this object to `eval_point`; +/// they do not call `assign_x` themselves. template struct alignas(alignof(typename path_memoizer_key_t::interior_node)) basic_path_memoizer final @@ -69,12 +84,15 @@ HEDLEY_PRAGMA(GCC diagnostic pop) basic_path_memoizer() : dpf_{std::nullopt}, x_{std::nullopt}, filled_to_{0} { } + HEDLEY_NO_THROW basic_path_memoizer(basic_path_memoizer &&) noexcept = default; basic_path_memoizer(const basic_path_memoizer &) = default; + HEDLEY_NO_THROW basic_path_memoizer & operator=(basic_path_memoizer &&) noexcept = default; basic_path_memoizer & operator=(const basic_path_memoizer &) = default; ~basic_path_memoizer() = default; + HEDLEY_NO_THROW std::size_t assign_x(const dpf_type & dpf, input_type new_x) noexcept override { static constexpr auto clz_xor = utils::countl_zero_symmetric_difference{}; @@ -100,11 +118,13 @@ HEDLEY_PRAGMA(GCC diagnostic pop) return 1; } + HEDLEY_NO_THROW node_type & operator[](std::size_t i) noexcept override { return arr_[i]; } + HEDLEY_NO_THROW return_type begin() const noexcept override { if (x_.has_value() == true) @@ -117,14 +137,17 @@ HEDLEY_PRAGMA(GCC diagnostic pop) } } + HEDLEY_NO_THROW return_type end() const noexcept override { return std::addressof(arr_[depth+1]); } /// Inclusive high-water: `arr_[0..filled_to_]` are valid for the current x. + HEDLEY_NO_THROW std::size_t filled_to() const noexcept { return filled_to_; } + HEDLEY_NO_THROW void note_filled(std::size_t level) noexcept { if (level > filled_to_) @@ -143,6 +166,7 @@ HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") HEDLEY_PRAGMA(GCC diagnostic pop) }; +/// A single interior node. Every `assign_x` restarts at the root. template struct nonmemoizing_path_memoizer final : public path_memoizer_base> @@ -159,8 +183,10 @@ HEDLEY_PRAGMA(GCC diagnostic pop) nonmemoizing_path_memoizer() : dpf_{std::nullopt} { } + HEDLEY_NO_THROW nonmemoizing_path_memoizer(nonmemoizing_path_memoizer &&) noexcept = default; nonmemoizing_path_memoizer(const nonmemoizing_path_memoizer &) = default; + HEDLEY_NO_THROW nonmemoizing_path_memoizer & operator=(nonmemoizing_path_memoizer &&) noexcept = default; nonmemoizing_path_memoizer & operator=(const nonmemoizing_path_memoizer &) = default; ~nonmemoizing_path_memoizer() = default; @@ -186,6 +212,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) return std::addressof(v); } + HEDLEY_NO_THROW return_type end() const noexcept override { return std::addressof(v) + 1; @@ -260,6 +287,9 @@ void ensure_level(const DpfKey & dpf, typename DpfKey::input_type x, } // namespace detail +/// Path workspace for `DpfKey`. A `party_key` argument is unwrapped, and the +/// result accepts both parties. +/// @snippet evaluation/memoizers.cpp path-memoizer template auto make_basic_path_memoizer() { @@ -272,6 +302,7 @@ auto make_basic_path_memoizer(const DpfKey &) return make_basic_path_memoizer(); } +/// Single-node path workspace. Suitable for one query. template auto make_nonmemoizing_path_memoizer() { diff --git a/include/dpf/placement.hpp b/include/dpf/placement.hpp index 370c87e..0150703 100644 --- a/include/dpf/placement.hpp +++ b/include/dpf/placement.hpp @@ -64,7 +64,8 @@ template inline constexpr bool is_at_v = is_at::value; /// `OutBits` is the comparison output group width (bits of the β payload), /// so the value CWs / addend can be stored at group width instead of a full /// padded `uint64_t` per level. -template +template struct cmp_channel_tag { static constexpr std::size_t depth = Depth; @@ -73,11 +74,17 @@ struct cmp_channel_tag /// after keygen. Concrete (non-wildcard) cmp keys keep `Wild == false` /// so their layout / type name is unchanged. static constexpr bool wild = Wild; + /// 0 keeps the per-level path-sum. `B >= 1` selects blocked checkpoints + /// of target width `B`. + static constexpr std::size_t block_width = BlockWidth; + /// Save a final correction at every depth (`idcf`). + static constexpr bool incremental = Incremental; }; template struct is_cmp_channel_tag : std::false_type {}; -template -struct is_cmp_channel_tag> +template +struct is_cmp_channel_tag> : std::true_type {}; template inline constexpr bool is_cmp_channel_tag_v = @@ -107,6 +114,28 @@ struct is_placed> : std::true_type {}; template inline constexpr bool is_placed_v = is_placed>::value; +/// Heavy-hitters incremental point function: one payload per prefix length. +/// `levels[i]` is the bit length of slot `i`. +template +struct idpf_pack; + +template +struct idpf_pack, Betas...> +{ + static constexpr bool is_idpf = true; + static constexpr std::size_t n = sizeof...(Levels); + static constexpr std::array levels{ + Levels...}; + std::tuple values; +}; + +template struct is_idpf : std::false_type {}; +template +struct is_idpf, Betas...>> + : std::true_type {}; +template +inline constexpr bool is_idpf_v = is_idpf>::value; + template inline constexpr std::size_t out_bits_v = utils::bitlength_of_output_v, NodeT>; @@ -385,6 +414,8 @@ struct normalize_one static constexpr std::size_t cmp_depth = 0; static constexpr std::size_t cmp_out_bits = 0; static constexpr bool cmp_wild = false; + static constexpr std::size_t cmp_block = 0; + static constexpr bool cmp_idcf = false; }; template struct normalize_one> @@ -393,14 +424,20 @@ struct normalize_one> static constexpr std::size_t cmp_depth = 0; static constexpr std::size_t cmp_out_bits = 0; static constexpr bool cmp_wild = false; + static constexpr std::size_t cmp_block = 0; + static constexpr bool cmp_idcf = false; }; -template -struct normalize_one> +template +struct normalize_one> { using placed_tuple = std::tuple<>; static constexpr std::size_t cmp_depth = Depth; static constexpr std::size_t cmp_out_bits = OutBits; static constexpr bool cmp_wild = Wild; + static constexpr std::size_t cmp_block = Block; + static constexpr bool cmp_idcf = Incremental; }; template @@ -419,6 +456,10 @@ struct normalize_pack // wildcard flag (false when there is no cmp channel). static constexpr bool cmp_wild = (false || ... || normalize_one::cmp_wild); + static constexpr std::size_t cmp_block = + (std::size_t{0} + ... + normalize_one::cmp_block); + static constexpr bool cmp_idcf = + (false || ... || normalize_one::cmp_idcf); }; template @@ -428,6 +469,8 @@ struct normalize_pack static constexpr std::size_t cmp_depth = 0; static constexpr std::size_t cmp_out_bits = 0; static constexpr bool cmp_wild = false; + static constexpr std::size_t cmp_block = 0; + static constexpr bool cmp_idcf = false; }; /// True iff the pack is "classic-shaped": every element is a bare output (no @@ -439,6 +482,32 @@ inline constexpr bool is_classic_pack_v = } // namespace incr } // namespace detail +/// Sparse heavy-hitters IDPF. Slot `i` is the point function on prefix +/// `Levels[i]`, evaluated with `out`. +template +auto idpf_at(Betas ...betas) +{ + static_assert(sizeof...(Levels) == sizeof...(Betas), + "idpf_at: one payload per prefix length"); + return detail::incr::idpf_pack, + std::decay_t...>{ + std::tuple...>{std::move(betas)...}}; +} + +template +auto idpf_from_seq(std::index_sequence, Betas ...betas) +{ + return idpf_at<(I + 1)...>(std::move(betas)...); +} + +/// Consecutive prefixes of length 1, 2, …, `sizeof...(Betas)`. +template +auto idpf(Betas ...betas) +{ + return idpf_from_seq(std::make_index_sequence{}, + std::move(betas)...); +} + } // namespace dpf #endif // LIBDPF_INCLUDE_DPF_PLACEMENT_HPP__ diff --git a/include/dpf/prg.hpp b/include/dpf/prg.hpp index 7455c9f..a8209d4 100644 --- a/include/dpf/prg.hpp +++ b/include/dpf/prg.hpp @@ -13,9 +13,11 @@ #include #include +#include #include #include "dpf/prg_aes.hpp" +#include "dpf/prg_chacha.hpp" #include "dpf/prg_dummy.hpp" #include "dpf/prg_lowmc.hpp" #include "dpf/secret_share.hpp" @@ -63,23 +65,34 @@ auto expand_as_share(typename PRG::block_type seed, template template +HEDLEY_NO_THROW auto aes::expand(block_type seed, psnip_uint32_t pos) noexcept { return detail::expand_as_share, T, Party>(seed, pos); } template +HEDLEY_NO_THROW auto dummy::expand(block_type seed, psnip_uint32_t pos) noexcept { return detail::expand_as_share(seed, pos); } template +HEDLEY_NO_THROW auto lowmc128::expand(block_type seed, psnip_uint32_t pos) noexcept { return detail::expand_as_share(seed, pos); } +template +template +HEDLEY_NO_THROW +auto chacha::expand(block_type seed, psnip_uint32_t pos) noexcept +{ + return detail::expand_as_share, T, Party>(seed, pos); +} + template struct counter_wrapper final { @@ -101,17 +114,22 @@ struct counter_wrapper final return PRG::eval01(seed); } - HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE static void eval(block_type seed, block_type * HEDLEY_RESTRICT output, - psnip_uint32_t count, psnip_uint32_t pos = 0) noexcept + psnip_uint32_t count, psnip_uint32_t pos = 0) { + if (count > 1 && + pos > static_cast(~static_cast(0)) - (count - 1u)) + { + throw std::invalid_argument("prg lane index is out of range"); + } count_.fetch_add(count, std::memory_order::memory_order_relaxed); PRG::eval(seed, output, count, pos); } HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2, 3) static void eval01_x4(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT left, block_type * HEDLEY_RESTRICT right) noexcept @@ -122,6 +140,7 @@ struct counter_wrapper final HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void eval_x4(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT output, psnip_uint32_t pos = 0) noexcept { @@ -131,6 +150,7 @@ struct counter_wrapper final HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void eval_x8(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT output, psnip_uint32_t pos = 0) noexcept { diff --git a/include/dpf/prg_aes.hpp b/include/dpf/prg_aes.hpp index 67e4b69..4533987 100644 --- a/include/dpf/prg_aes.hpp +++ b/include/dpf/prg_aes.hpp @@ -13,6 +13,7 @@ #include #include #include +#include #include "hedley/hedley.h" #include "simde/simde/x86/avx2.h" @@ -103,15 +104,21 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// `eval` / `eval01` (`set_epi64x(0, pos)`). The first AddRoundKey /// includes `rd_key[0]` so this matches the one-block `eval` for any /// key, not only the all-zero key this PRG currently installs. - HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE static void eval(block_type seed, block_type * HEDLEY_RESTRICT output, - psnip_uint32_t count, psnip_uint32_t pos = 0) noexcept + psnip_uint32_t count, psnip_uint32_t pos = 0) { if (HEDLEY_UNLIKELY(count == 0)) { return; } + // `pos + i` is a uint32 add. A span that passes UINT32_MAX must + // fail the same way buffered_prg does, rather than wrap to 0. + if (count > 1 && + pos > static_cast(~static_cast(0)) - (count - 1u)) + { + throw std::invalid_argument("prg lane index is out of range"); + } require_block_aligned(output); block_type * HEDLEY_RESTRICT out = @@ -161,6 +168,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// `left[i] == eval(seeds[i], 0)`, `right[i] == eval(seeds[i], 1)`. HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2, 3) static void eval01_x4(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT left, block_type * HEDLEY_RESTRICT right) noexcept @@ -193,6 +201,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// Four independent `eval(seed, pos)` as one 4-block round-major AES. HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void eval_x4(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT output, psnip_uint32_t pos) noexcept { @@ -223,6 +232,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// Eight independent `eval(seed, pos)` as one 8-block round-major AES. HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void eval_x8(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT output, psnip_uint32_t pos) noexcept { @@ -252,6 +262,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// Raw-bit subtractive share of `T` for party `Party` (see `prg.hpp`). template + HEDLEY_NO_THROW static auto expand(block_type seed, psnip_uint32_t pos = 0) noexcept; private: @@ -261,6 +272,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// rounds 1..last and the MMO feed-forward `XOR seed[i]`. HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void aes_mmo_rounds_x4(block_type * HEDLEY_RESTRICT blk, const block_type * HEDLEY_RESTRICT seed) noexcept { @@ -283,6 +295,7 @@ HEDLEY_PRAGMA(GCC unroll(14)) HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void aes_mmo_rounds_x8(block_type * HEDLEY_RESTRICT blk, const block_type * HEDLEY_RESTRICT seed) noexcept { diff --git a/include/dpf/prg_chacha.hpp b/include/dpf/prg_chacha.hpp new file mode 100644 index 0000000..f3a38c9 --- /dev/null +++ b/include/dpf/prg_chacha.hpp @@ -0,0 +1,458 @@ +/// @file dpf/prg_chacha.hpp +/// @brief ChaCha stream PRG. Same 128-bit block interface as `aes128`. +/// @details `eval(seed, pos)` is the `pos`-th 16-byte chunk of ChaCha +/// keystream (RFC 8439). The 256-bit key is the 128-bit seed +/// followed by the fixed domain separator `"dpf-chacha-prg\0\0"`. +/// The nonce is zero. The ChaCha block counter is `pos / 4`, and +/// the chunk inside that block is `pos % 4`. +/// +/// `chacha20` is the RFC round count. `chacha12` and `chacha8` are +/// the same construction with fewer rounds. `chacha` accepts any +/// positive even round count. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license; +/// see [LICENSE.md](@ref license) for details. + +#ifndef LIBDPF_INCLUDE_DPF_PRG_CHACHA_HPP__ +#define LIBDPF_INCLUDE_DPF_PRG_CHACHA_HPP__ + +#include +#include +#include +#include +#include + +#include "hedley/hedley.h" +#include "simde/simde/x86/avx2.h" +#include "portable-snippets/exact-int/exact-int.h" + +namespace dpf +{ + +namespace prg +{ + +namespace chacha_detail +{ + +inline constexpr std::uint32_t zero_nonce[3] = {0, 0, 0}; + +/// ASCII `"dpf-chacha-prg"` plus two zero bytes. Public second half of the key. +inline constexpr std::uint8_t domain[16] = { + 'd', 'p', 'f', '-', 'c', 'h', 'a', 'c', + 'h', 'a', '-', 'p', 'r', 'g', 0, 0 +}; + +HEDLEY_PURE +HEDLEY_NON_NULL(1) +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr std::uint32_t load_le32(const std::uint8_t * p) noexcept +{ + return static_cast(p[0]) + | (static_cast(p[1]) << 8) + | (static_cast(p[2]) << 16) + | (static_cast(p[3]) << 24); +} + +HEDLEY_NON_NULL(1) +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +void store_le32(std::uint8_t * p, std::uint32_t w) noexcept +{ + p[0] = static_cast(w); + p[1] = static_cast(w >> 8); + p[2] = static_cast(w >> 16); + p[3] = static_cast(w >> 24); +} + +HEDLEY_PURE +HEDLEY_NON_NULL(1) +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +simde__m128i load_block(const std::uint8_t * p) noexcept +{ + simde__m128i out; + std::memcpy(&out, p, sizeof(out)); + return out; +} + +/// 128-bit seed in the low half, `domain` in the high half, both little-endian. +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +void seed_key(simde__m128i seed, std::uint32_t key[8]) noexcept +{ + std::uint8_t raw[16]; + std::memcpy(raw, &seed, sizeof(raw)); + for (int i = 0; i < 4; ++i) + { + key[i] = load_le32(raw + 4 * i); + key[4 + i] = load_le32(domain + 4 * i); + } +} + +template +HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW +constexpr std::uint32_t rotl(std::uint32_t x) noexcept +{ + static_assert(N > 0 && N < 32, "ChaCha rotation is between 1 and 31"); + return (x << N) | (x >> (32 - N)); +} + +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +void quarter(std::uint32_t & a, std::uint32_t & b, + std::uint32_t & c, std::uint32_t & d) noexcept +{ + a += b; d ^= a; d = rotl<16>(d); + c += d; b ^= c; b = rotl<12>(b); + a += b; d ^= a; d = rotl<8>(d); + c += d; b ^= c; b = rotl<7>(b); +} + +template +HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW +simde__m128i rotl_epi32(simde__m128i v) noexcept +{ + static_assert(N > 0 && N < 32, "ChaCha rotation is between 1 and 31"); + return simde_mm_or_si128(simde_mm_slli_epi32(v, N), + simde_mm_srli_epi32(v, 32 - N)); +} + +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +void quarter(simde__m128i x[], int a, int b, int c, int d) noexcept +{ + x[a] = simde_mm_add_epi32(x[a], x[b]); + x[d] = rotl_epi32<16>(simde_mm_xor_si128(x[d], x[a])); + x[c] = simde_mm_add_epi32(x[c], x[d]); + x[b] = rotl_epi32<12>(simde_mm_xor_si128(x[b], x[c])); + x[a] = simde_mm_add_epi32(x[a], x[b]); + x[d] = rotl_epi32<8>(simde_mm_xor_si128(x[d], x[a])); + x[c] = simde_mm_add_epi32(x[c], x[d]); + x[b] = rotl_epi32<7>(simde_mm_xor_si128(x[b], x[c])); +} + +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +std::uint32_t epi32_lane(simde__m128i v, int lane) noexcept +{ + // Shuffle control is an immediate, so each lane is its own case. + switch (lane) + { + case 1: v = simde_mm_shuffle_epi32(v, 0x01); break; + case 2: v = simde_mm_shuffle_epi32(v, 0x02); break; + case 3: v = simde_mm_shuffle_epi32(v, 0x03); break; + default: break; + } + return static_cast(simde_mm_cvtsi128_si32(v)); +} + +/// One ChaCha block. `key` is 8 little-endian words. `nonce` is 3 words. +template +HEDLEY_NO_THROW +void block(const std::uint32_t key[8], std::uint32_t counter, + const std::uint32_t nonce[3], std::uint8_t out[64]) noexcept +{ + static_assert(Rounds >= 2 && Rounds % 2 == 0, + "ChaCha rounds must be a positive even number"); + + std::uint32_t s[16] = { + 0x61707865u, 0x3320646eu, 0x79622d32u, 0x6b206574u, + key[0], key[1], key[2], key[3], + key[4], key[5], key[6], key[7], + counter, nonce[0], nonce[1], nonce[2] + }; + std::uint32_t orig[16]; + std::memcpy(orig, s, sizeof(orig)); + +HEDLEY_PRAGMA(GCC unroll 16) + for (unsigned r = 0; r < Rounds; r += 2) + { + quarter(s[0], s[4], s[8], s[12]); + quarter(s[1], s[5], s[9], s[13]); + quarter(s[2], s[6], s[10], s[14]); + quarter(s[3], s[7], s[11], s[15]); + quarter(s[0], s[5], s[10], s[15]); + quarter(s[1], s[6], s[11], s[12]); + quarter(s[2], s[7], s[8], s[13]); + quarter(s[3], s[4], s[9], s[14]); + } + + for (int i = 0; i < 16; ++i) + { + store_le32(out + 4 * i, s[i] + orig[i]); + } +} + +/// Four independent ChaCha blocks. Lane `i` uses `key[i]` and `counter[i]`. +/// Nonce is zero. Each `out[i]` receives 64 bytes. +template +HEDLEY_NO_THROW +void block4(const std::uint32_t key[][8], const std::uint32_t counter[4], + std::uint8_t out[][64]) noexcept +{ + static_assert(Rounds >= 2 && Rounds % 2 == 0, + "ChaCha rounds must be a positive even number"); + + simde__m128i x[16]; + x[0] = simde_mm_set1_epi32(static_cast(0x61707865u)); + x[1] = simde_mm_set1_epi32(static_cast(0x3320646eu)); + x[2] = simde_mm_set1_epi32(static_cast(0x79622d32u)); + x[3] = simde_mm_set1_epi32(static_cast(0x6b206574u)); + for (int w = 0; w < 8; ++w) + { + x[4 + w] = simde_mm_set_epi32( + static_cast(key[3][w]), + static_cast(key[2][w]), + static_cast(key[1][w]), + static_cast(key[0][w])); + } + x[12] = simde_mm_set_epi32( + static_cast(counter[3]), + static_cast(counter[2]), + static_cast(counter[1]), + static_cast(counter[0])); + x[13] = simde_mm_setzero_si128(); + x[14] = simde_mm_setzero_si128(); + x[15] = simde_mm_setzero_si128(); + + simde__m128i orig[16]; + for (int i = 0; i < 16; ++i) + { + orig[i] = x[i]; + } + +HEDLEY_PRAGMA(GCC unroll 16) + for (unsigned r = 0; r < Rounds; r += 2) + { + quarter(x, 0, 4, 8, 12); + quarter(x, 1, 5, 9, 13); + quarter(x, 2, 6, 10, 14); + quarter(x, 3, 7, 11, 15); + quarter(x, 0, 5, 10, 15); + quarter(x, 1, 6, 11, 12); + quarter(x, 2, 7, 8, 13); + quarter(x, 3, 4, 9, 14); + } + + for (int lane = 0; lane < 4; ++lane) + { + for (int w = 0; w < 16; ++w) + { + std::uint32_t sum = epi32_lane(x[w], lane) + epi32_lane(orig[w], lane); + store_le32(out[lane] + 4 * w, sum); + } + } +} + +} // namespace chacha_detail + +/// ChaCha stream PRG with `Rounds` rounds (20 is RFC 8439). +template +struct chacha final +{ + static_assert(Rounds >= 2 && Rounds % 2 == 0, + "ChaCha rounds must be a positive even number"); + + using block_type = simde__m128i; + static constexpr unsigned rounds = Rounds; + + HEDLEY_NO_THROW + HEDLEY_ALWAYS_INLINE + static block_type eval(block_type seed, psnip_uint32_t pos) noexcept + { + std::uint32_t key[8]; + chacha_detail::seed_key(seed, key); + std::uint8_t buf[64]; + chacha_detail::block(key, pos >> 2, + chacha_detail::zero_nonce, buf); + return chacha_detail::load_block(buf + 16 * (pos & 3u)); + } + + /// Positions 0 and 1, one ChaCha block (the first 32 keystream bytes). + HEDLEY_NO_THROW + HEDLEY_ALWAYS_INLINE + static auto eval01(block_type seed) noexcept + { + std::uint32_t key[8]; + chacha_detail::seed_key(seed, key); + std::uint8_t buf[64]; + chacha_detail::block(key, 0, chacha_detail::zero_nonce, buf); +HEDLEY_PRAGMA(GCC diagnostic push) +HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") + return std::array{ + chacha_detail::load_block(buf), + chacha_detail::load_block(buf + 16) + }; +HEDLEY_PRAGMA(GCC diagnostic pop) + } + + HEDLEY_ALWAYS_INLINE + static void eval(block_type seed, block_type * HEDLEY_RESTRICT output, + psnip_uint32_t count, psnip_uint32_t pos = 0) + { + if (HEDLEY_UNLIKELY(count == 0)) + { + return; + } + if (count > 1 && + pos > static_cast(~static_cast(0)) - (count - 1u)) + { + throw std::invalid_argument("prg lane index is out of range"); + } + if (count == 1) + { + output[0] = eval(seed, pos); + return; + } + if (count == 2 && pos == 0) + { + auto kids = eval01(seed); + output[0] = kids[0]; + output[1] = kids[1]; + return; + } + + std::uint32_t key[8]; + chacha_detail::seed_key(seed, key); + psnip_uint32_t i = 0; + + // `pos` may begin mid-block. Those chunks share one ChaCha block. + if ((pos & 3u) != 0u) + { + std::uint8_t buf[64]; + chacha_detail::block(key, pos >> 2, + chacha_detail::zero_nonce, buf); + while (i < count && ((pos + i) & 3u) != 0u) + { + output[i] = chacha_detail::load_block( + buf + 16 * ((pos + i) & 3u)); + ++i; + } + } + + // Four consecutive counters cover 16 output blocks. + while (i + 16u <= count) + { + std::uint32_t base = (pos + i) >> 2; + std::uint32_t keys[4][8]; + std::uint32_t counters[4]; + for (int lane = 0; lane < 4; ++lane) + { + std::memcpy(keys[lane], key, sizeof(key)); + counters[lane] = base + static_cast(lane); + } + std::uint8_t buf[4][64]; + chacha_detail::block4(keys, counters, buf); + for (int lane = 0; lane < 4; ++lane) + { + for (int chunk = 0; chunk < 4; ++chunk) + { + output[i++] = chacha_detail::load_block( + buf[lane] + 16 * chunk); + } + } + } + + while (i + 4u <= count) + { + std::uint8_t buf[64]; + chacha_detail::block(key, (pos + i) >> 2, + chacha_detail::zero_nonce, buf); + for (int chunk = 0; chunk < 4; ++chunk) + { + output[i++] = chacha_detail::load_block(buf + 16 * chunk); + } + } + + if (i < count) + { + std::uint8_t buf[64]; + chacha_detail::block(key, (pos + i) >> 2, + chacha_detail::zero_nonce, buf); + unsigned chunk = 0; + while (i < count) + { + output[i++] = chacha_detail::load_block(buf + 16 * chunk); + ++chunk; + } + } + } + + HEDLEY_NO_THROW + HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2, 3) + static void eval01_x4(const block_type * HEDLEY_RESTRICT seeds, + block_type * HEDLEY_RESTRICT left, + block_type * HEDLEY_RESTRICT right) noexcept + { + std::uint32_t keys[4][8]; + std::uint32_t counters[4] = {0, 0, 0, 0}; + for (int lane = 0; lane < 4; ++lane) + { + chacha_detail::seed_key(seeds[lane], keys[lane]); + } + std::uint8_t buf[4][64]; + chacha_detail::block4(keys, counters, buf); + for (int lane = 0; lane < 4; ++lane) + { + left[lane] = chacha_detail::load_block(buf[lane]); + right[lane] = chacha_detail::load_block(buf[lane] + 16); + } + } + + HEDLEY_NO_THROW + HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) + static void eval_x4(const block_type * HEDLEY_RESTRICT seeds, + block_type * HEDLEY_RESTRICT output, psnip_uint32_t pos = 0) noexcept + { + std::uint32_t keys[4][8]; + std::uint32_t ctr = pos >> 2; + std::uint32_t counters[4] = {ctr, ctr, ctr, ctr}; + for (int lane = 0; lane < 4; ++lane) + { + chacha_detail::seed_key(seeds[lane], keys[lane]); + } + std::uint8_t buf[4][64]; + chacha_detail::block4(keys, counters, buf); + unsigned chunk = pos & 3u; + for (int lane = 0; lane < 4; ++lane) + { + output[lane] = chacha_detail::load_block(buf[lane] + 16 * chunk); + } + } + + HEDLEY_NO_THROW + HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) + static void eval_x8(const block_type * HEDLEY_RESTRICT seeds, + block_type * HEDLEY_RESTRICT output, psnip_uint32_t pos = 0) noexcept + { + eval_x4(seeds, output, pos); + eval_x4(seeds + 4, output + 4, pos); + } + + /// Raw-bit subtractive share of `T` for party `Party` (see `prg.hpp`). + template + HEDLEY_NO_THROW + static auto expand(block_type seed, psnip_uint32_t pos = 0) noexcept; +}; // struct chacha + +/// RFC 8439 ChaCha20. +using chacha20 = chacha<20>; + +/// ChaCha12. Same keying as `chacha20`, 12 rounds. +using chacha12 = chacha<12>; + +/// ChaCha8. Same keying as `chacha20`, 8 rounds. +using chacha8 = chacha<8>; + +} // namespace prg + +} // namespace dpf + +#endif // LIBDPF_INCLUDE_DPF_PRG_CHACHA_HPP__ diff --git a/include/dpf/prg_dummy.hpp b/include/dpf/prg_dummy.hpp index ef197fe..8e4c6d4 100644 --- a/include/dpf/prg_dummy.hpp +++ b/include/dpf/prg_dummy.hpp @@ -1,6 +1,8 @@ /// @file dpf/prg_dummy.hpp -/// @brief -/// @details +/// @brief Identity PRG. `eval` returns the seed and ignores the lane. +/// @details Used where a PRG-shaped type is required and the block must stay +/// equal to the seed. The batched entry points write that seed into +/// every output lane. /// @author Ryan Henry /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license; @@ -57,6 +59,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2, 3) static void eval01_x4(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT left, block_type * HEDLEY_RESTRICT right) noexcept @@ -69,6 +72,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void eval_x4(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT output, psnip_uint32_t = 0) noexcept { @@ -77,6 +81,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void eval_x8(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT output, psnip_uint32_t = 0) noexcept { @@ -85,6 +90,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// Raw-bit subtractive share of `T` for party `Party` (see `prg.hpp`). template + HEDLEY_NO_THROW static auto expand(block_type seed, psnip_uint32_t pos = 0) noexcept; }; // struct dummy diff --git a/include/dpf/prg_lowmc.hpp b/include/dpf/prg_lowmc.hpp index 25743e9..1ab39dd 100644 --- a/include/dpf/prg_lowmc.hpp +++ b/include/dpf/prg_lowmc.hpp @@ -11,6 +11,7 @@ #include #include #include +#include #include "hedley/hedley.h" #include "simde/simde/x86/avx2.h" @@ -49,11 +50,15 @@ HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") HEDLEY_PRAGMA(GCC diagnostic pop) } - HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE static void eval(block_type seed, block_type * HEDLEY_RESTRICT output, - psnip_uint32_t count, psnip_uint32_t pos = 0) noexcept + psnip_uint32_t count, psnip_uint32_t pos = 0) { + if (count > 1 && + pos > static_cast(~static_cast(0)) - (count - 1u)) + { + throw std::invalid_argument("prg lane index is out of range"); + } for (psnip_uint32_t i = 0; i < count; ++i) { output[i] = eval(seed, pos + i); @@ -62,6 +67,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2, 3) static void eval01_x4(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT left, block_type * HEDLEY_RESTRICT right) noexcept @@ -76,6 +82,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void eval_x4(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT output, psnip_uint32_t pos = 0) noexcept { @@ -87,6 +94,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE + HEDLEY_NON_NULL(1, 2) static void eval_x8(const block_type * HEDLEY_RESTRICT seeds, block_type * HEDLEY_RESTRICT output, psnip_uint32_t pos = 0) noexcept { @@ -98,9 +106,11 @@ HEDLEY_PRAGMA(GCC diagnostic pop) /// Raw-bit subtractive share of `T` for party `Party` (see `prg.hpp`). template + HEDLEY_NO_THROW static auto expand(block_type seed, psnip_uint32_t pos = 0) noexcept; private: + HEDLEY_NO_THROW static lowmc::block to_block(block_type x) noexcept { std::uint64_t lane[2]; @@ -114,6 +124,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) return b; } + HEDLEY_NO_THROW static block_type from_block(const lowmc::block & b) noexcept { std::uint64_t lane[2] = {0, 0}; @@ -127,6 +138,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) return x; } + HEDLEY_NO_THROW static block_type permute(block_type x) noexcept { static lowmc::LowMC cipher; diff --git a/include/dpf/random.hpp b/include/dpf/random.hpp index 5ac68ce..a6ad5ce 100644 --- a/include/dpf/random.hpp +++ b/include/dpf/random.hpp @@ -39,6 +39,7 @@ inline thread_local void (*uniform_bytes_hook)(void *, std::size_t) = nullptr; template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW bool fill_from_hook(T & buf) noexcept { if (uniform_bytes_hook == nullptr) @@ -52,6 +53,7 @@ bool fill_from_hook(T & buf) noexcept /// `bool` and `enum : bool` (including `dpf::bit`) have only two valid /// representations. Filling them with a raw entropy byte is undefined. template +HEDLEY_NO_THROW constexpr bool is_boolean_representation() noexcept { using U = std::remove_cv_t; @@ -94,7 +96,8 @@ struct entropy_source entropy_source(entropy_source &&) = delete; entropy_source & operator=(entropy_source &&) = delete; - ~entropy_source() + HEDLEY_NO_THROW + ~entropy_source() noexcept { if (fp != nullptr) { diff --git a/include/dpf/rotated_iterable.hpp b/include/dpf/rotated_iterable.hpp index 19e07b7..d0f81d0 100644 --- a/include/dpf/rotated_iterable.hpp +++ b/include/dpf/rotated_iterable.hpp @@ -1,3 +1,8 @@ +/// @file dpf/rotated_iterable.hpp +/// @brief Retired container rotation view. The live type is `rotation_iterable`. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + // /// @file dpf/rotated_iterable.hpp // /// @author Ryan Henry // /// @brief defines `dpf::rotated_iterable` and associated helpers diff --git a/include/dpf/rotation_iterable.hpp b/include/dpf/rotation_iterable.hpp index dbdf562..31bcc3e 100644 --- a/include/dpf/rotation_iterable.hpp +++ b/include/dpf/rotation_iterable.hpp @@ -1,3 +1,11 @@ +/// @file dpf/rotation_iterable.hpp +/// @brief A rotated view of an iterator range. +/// @details `begin()` starts `distance` elements into the wrapped range and +/// wraps back to the original begin. Indexing is O(1) when the +/// wrapped iterator is random-access. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + #ifndef LIBDPF_INCLUDE_DPF_ROTATED_VIEW_HPP__ #define LIBDPF_INCLUDE_DPF_ROTATED_VIEW_HPP__ @@ -30,20 +38,14 @@ struct rotation_iterable constexpr rotation_iterable(wrapped_iterator begin, wrapped_iterator end, difference_type distance) : size_{std::distance(begin, end)}, - distance_{ - [this, &distance]() - { - distance %= this->size_; - if (distance < 0) - { - distance += this->size_; - } - return distance; - }()}, - begin_{std::next(begin, static_cast(distance_))}, + distance_{normalize_distance(distance, size_)}, + begin_{size_ == 0 ? begin : std::next(begin, distance_)}, wrap_to_{begin}, - wrap_after_{std::next(end, -difference_type(distance_ > 0))}, - end_after_{std::next(wrap_to_, static_cast(distance_-1))}, + wrap_after_{size_ == 0 || distance_ == 0 + ? end : std::next(end, difference_type{-1})}, + end_after_{size_ == 0 ? begin + : (distance_ == 0 ? std::next(end, difference_type{-1}) + : std::next(begin, distance_ - 1))}, end_{end} { } @@ -52,11 +54,11 @@ struct rotation_iterable constexpr rotation_iterable(wrapped_iterator begin, wrapped_iterator end, wrapped_iterator middle) : size_{std::distance(begin, end)}, - distance_{std::distance(begin, middle)}, + distance_{size_ == 0 ? difference_type{0} : std::distance(begin, middle)}, begin_{middle}, wrap_to_{begin}, - wrap_after_{std::next(end, difference_type(-1))}, - end_after_{std::next(middle, difference_type(-1))}, + wrap_after_{size_ == 0 ? end : std::next(end, difference_type{-1})}, + end_after_{size_ == 0 ? begin : std::next(middle, difference_type{-1})}, end_{end} { } @@ -136,6 +138,19 @@ struct rotation_iterable } private: + HEDLEY_CONST + HEDLEY_NO_THROW + static constexpr difference_type normalize_distance(difference_type distance, + difference_type size) noexcept + { + if (size == 0) + return difference_type{0}; + distance %= size; + if (distance < 0) + distance += size; + return distance; + } + difference_type size_; difference_type distance_; wrapped_iterator begin_; @@ -163,6 +178,7 @@ struct rotation_iterator_base std::add_const_t>; using pointer = typename std::iterator_traits::pointer; + HEDLEY_NO_THROW rotation_iterator_base(const rotation_iterable & iterable_, wrapped_iterator iterator) noexcept : iterable{iterable_}, it{iterator} { } @@ -256,6 +272,7 @@ struct rotation_iterator final using wrapped_iterator = WrappedIterator; using reference = typename base::reference; + HEDLEY_NO_THROW rotation_iterator(const rotation_iterable & iterable, wrapped_iterator iterator) noexcept : base{iterable, iterator} {} @@ -277,6 +294,7 @@ struct rotation_const_iterator final using wrapped_iterator = WrappedIterator; using const_reference = typename base::const_reference; + HEDLEY_NO_THROW rotation_const_iterator(const rotation_iterable & iterable, wrapped_iterator iterator) noexcept : base{iterable, iterator} {} diff --git a/include/dpf/secret_share.hpp b/include/dpf/secret_share.hpp index 7c030ba..4d51b9d 100644 --- a/include/dpf/secret_share.hpp +++ b/include/dpf/secret_share.hpp @@ -99,10 +99,19 @@ namespace detail template HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr T party_coeff_times(const T & v) noexcept { if constexpr (Scheme == sharing::additive || Party == 0) return v; + else if constexpr (std::is_integral_v && std::is_signed_v) + { + // Signed negation of the minimum is undefined. The two's-complement + // negation is well-defined on the unsigned width. + using unsigned_type = std::make_unsigned_t; + return static_cast(static_cast(0) + - static_cast(v)); + } else return static_cast(-v); } @@ -122,13 +131,18 @@ struct secret_share T value{}; secret_share() = default; + HEDLEY_NO_THROW secret_share(const secret_share &) noexcept = default; + HEDLEY_NO_THROW secret_share(secret_share &&) noexcept = default; + HEDLEY_NO_THROW secret_share & operator=(const secret_share &) noexcept = default; + HEDLEY_NO_THROW secret_share & operator=(secret_share &&) noexcept = default; ~secret_share() = default; /// Bit-preserving construction. Does not apply a party coefficient. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_CONST static constexpr secret_share from_raw(T v) noexcept @@ -138,15 +152,18 @@ struct secret_share return s; } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_CONST constexpr const T & raw() const noexcept { return value; } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_PURE constexpr T & raw() noexcept { return value; } /// Secret-preserving conversion to an additive share of the same party. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_PURE constexpr additive_share as_additive() const noexcept @@ -159,6 +176,7 @@ struct secret_share } /// Secret-preserving conversion to a subtractive share of the same party. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_PURE constexpr subtractive_share as_subtractive() const noexcept @@ -174,18 +192,28 @@ struct secret_share template HEDLEY_ALWAYS_INLINE HEDLEY_PURE + HEDLEY_NO_THROW constexpr secret_share retag() const noexcept { return secret_share::from_raw(value); } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_PURE constexpr secret_share operator-() const noexcept { - return from_raw(static_cast(-value)); + if constexpr (std::is_integral_v && std::is_signed_v) + { + using unsigned_type = std::make_unsigned_t; + return from_raw(static_cast(static_cast(0) + - static_cast(value))); + } + else + return from_raw(static_cast(-value)); } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr secret_share & operator+=(const secret_share & rhs) noexcept { @@ -193,6 +221,7 @@ struct secret_share return *this; } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr secret_share & operator-=(const secret_share & rhs) noexcept { @@ -203,6 +232,7 @@ struct secret_share template , int> = 0> HEDLEY_ALWAYS_INLINE + HEDLEY_NO_THROW constexpr secret_share & operator*=(const Scalar & c) noexcept { value = static_cast(value * static_cast(c)); @@ -214,6 +244,7 @@ struct secret_share std::enable_if_t && std::is_convertible_v, int> = 0> HEDLEY_ALWAYS_INLINE + HEDLEY_NO_THROW constexpr secret_share & operator+=(const Plain & c) noexcept { if constexpr (Party == 0) @@ -225,6 +256,7 @@ struct secret_share std::enable_if_t && std::is_convertible_v, int> = 0> HEDLEY_ALWAYS_INLINE + HEDLEY_NO_THROW constexpr secret_share & operator-=(const Plain & c) noexcept { if constexpr (Party == 0) @@ -240,6 +272,7 @@ struct secret_share template HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator+( secret_share lhs, const secret_share & rhs) noexcept @@ -251,6 +284,7 @@ constexpr secret_share operator+( template HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator-( secret_share lhs, const secret_share & rhs) noexcept @@ -263,6 +297,7 @@ template , int> = 0> HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator*( secret_share lhs, const Scalar & c) noexcept { @@ -274,6 +309,7 @@ template , int> = 0> HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator*( const Scalar & c, secret_share rhs) noexcept { @@ -290,6 +326,7 @@ template = 0> HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator+( const secret_share & lhs, const secret_share & rhs) noexcept @@ -306,6 +343,7 @@ template = 0> HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator-( const secret_share & lhs, const secret_share & rhs) noexcept @@ -327,6 +365,7 @@ template , int> = 0> HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator+( secret_share lhs, const Plain & c) noexcept { @@ -339,6 +378,7 @@ template , int> = 0> HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator+( const Plain & c, secret_share rhs) noexcept { @@ -351,6 +391,7 @@ template , int> = 0> HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr secret_share operator-( secret_share lhs, const Plain & c) noexcept { @@ -365,6 +406,7 @@ constexpr secret_share operator-( template HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr bool operator==(const secret_share & lhs, const secret_share & rhs) noexcept { @@ -374,6 +416,7 @@ constexpr bool operator==(const secret_share & lhs, template HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr bool operator!=(const secret_share & lhs, const secret_share & rhs) noexcept { @@ -387,10 +430,21 @@ constexpr bool operator!=(const secret_share & lhs, template HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr T reconstruct(const secret_share & s0, const secret_share & s1) noexcept { - if constexpr (Scheme == sharing::additive) + if constexpr (std::is_integral_v && std::is_signed_v) + { + using unsigned_type = std::make_unsigned_t; + if constexpr (Scheme == sharing::additive) + return static_cast(static_cast(s0.raw()) + + static_cast(s1.raw())); + else + return static_cast(static_cast(s0.raw()) + - static_cast(s1.raw())); + } + else if constexpr (Scheme == sharing::additive) return static_cast(s0.raw() + s1.raw()); else return static_cast(s0.raw() - s1.raw()); @@ -399,6 +453,7 @@ constexpr T reconstruct(const secret_share & s0, template HEDLEY_ALWAYS_INLINE HEDLEY_PURE +HEDLEY_NO_THROW constexpr T reconstruct(const secret_share & s1, const secret_share & s0) noexcept { @@ -412,6 +467,7 @@ constexpr T reconstruct(const secret_share & s1, template HEDLEY_ALWAYS_INLINE HEDLEY_CONST +HEDLEY_NO_THROW constexpr auto make_additive_shares(T secret) noexcept { using T_ = std::remove_cv_t>; @@ -423,6 +479,7 @@ constexpr auto make_additive_shares(T secret) noexcept template HEDLEY_ALWAYS_INLINE HEDLEY_CONST +HEDLEY_NO_THROW constexpr auto make_subtractive_shares(T secret) noexcept { using T_ = std::remove_cv_t>; @@ -460,13 +517,16 @@ struct party_key : Key #endif } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE Key & key() noexcept { return static_cast(*this); } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE const Key & key() const noexcept { return static_cast(*this); } /// Party-tagged additive share of the comparison absorb addend. + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE auto cmp_addend() const noexcept { diff --git a/include/dpf/sequence_memoizer.hpp b/include/dpf/sequence_memoizer.hpp index d3562e6..54b1b0c 100644 --- a/include/dpf/sequence_memoizer.hpp +++ b/include/dpf/sequence_memoizer.hpp @@ -1,6 +1,13 @@ /// @file dpf/sequence_memoizer.hpp -/// @brief -/// @details +/// @brief Workspaces for a `sequence_recipe` traversal. +/// @details The memoizer stores a reference to the recipe it was built from +/// and later calls must pass that same object (`std::logic_error` +/// otherwise). The factories unwrap `party_key`. +/// +/// `inplace_reversing_sequence_memoizer` keeps one level. +/// `double_space_sequence_memoizer` keeps two and is the default +/// inside `eval_sequence(key, recipe, buffer)`. +/// `full_tree_sequence_memoizer` keeps every level. /// @author Ryan Henry /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) @@ -22,6 +29,7 @@ #include #include +#include "dpf/secret_share.hpp" #include "dpf/sequence_recipe.hpp" namespace dpf @@ -42,10 +50,13 @@ struct sequence_recipe_memoizer_base : public sequence_memoizer_tag_ // level 0 should access the root // level goes up to (and including) depth + HEDLEY_NO_THROW virtual return_type operator[](std::size_t) const noexcept = 0; // iterators should access most recently completed level + HEDLEY_NO_THROW virtual return_type begin() const noexcept = 0; + HEDLEY_NO_THROW virtual return_type end() const noexcept = 0; virtual std::size_t assign_dpf(const dpf_type & dpf, const sequence_recipe & r) @@ -210,11 +221,13 @@ struct pointer_facade return *this; } + HEDLEY_NO_THROW pointer_facade operator+(std::size_t n) const noexcept { return pointer_facade(flip_, it_ + n, rit_ + n); } + HEDLEY_NO_THROW pointer_facade & operator-=(std::size_t n) noexcept { it_ -= n; @@ -222,11 +235,13 @@ struct pointer_facade return *this; } + HEDLEY_NO_THROW pointer_facade operator-(std::size_t n) const noexcept { return pointer_facade(flip_, it_ - n, rit_ - n); } + HEDLEY_NO_THROW difference_type operator-(pointer_facade rhs) const noexcept { return std::make_pair(it_ - rhs.it_, rit_ - rhs.rit_); @@ -235,7 +250,7 @@ struct pointer_facade HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW HEDLEY_PURE - reference operator[](std::size_t i) + reference operator[](std::size_t i) noexcept { return flip_ ? rit_[i] : it_[i]; } @@ -243,7 +258,7 @@ struct pointer_facade HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW HEDLEY_PURE - const_reference operator[](std::size_t i) const + const_reference operator[](std::size_t i) const noexcept { return flip_ ? rit_[i] : it_[i]; } @@ -270,6 +285,8 @@ struct pointer_facade } // namespace detail +/// One level. The buffer is traversed in the opposite direction on +/// alternate levels. template > struct inplace_reversing_sequence_memoizer final @@ -364,6 +381,8 @@ HEDLEY_PRAGMA(GCC diagnostic pop) unique_ptr buf; }; +/// Two levels, so a level can be built while the previous level is still +/// intact. Default workspace for `eval_sequence` on a recipe. template > struct double_space_sequence_memoizer final @@ -416,6 +435,7 @@ HEDLEY_PRAGMA(GCC diagnostic pop) unique_ptr buf; }; +/// Every level of the recipe's traversal. template > struct full_tree_sequence_memoizer final @@ -482,10 +502,15 @@ auto make_sequence_memoizer(const sequence_recipe & recipe) HEDLEY_PRAGMA(GCC diagnostic push) HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes") +/// One-level sequence workspace bound to `recipe`. +/// @param recipe The object later passed to `eval_sequence`. The memoizer +/// holds a reference to it. +/// @snippet evaluation/memoizers.cpp sequence-memoizer template inline auto make_inplace_reversing_sequence_memoizer(const sequence_recipe & recipe) { - return detail::make_sequence_memoizer>(recipe); + using key_t = unwrap_party_key_t; + return detail::make_sequence_memoizer>(recipe); } template @@ -494,10 +519,13 @@ inline auto make_inplace_reversing_sequence_memoizer(const DpfKey &, const seque return make_inplace_reversing_sequence_memoizer(recipe); } +/// Two-level sequence workspace bound to `recipe`. +/// @snippet evaluation/eval_sequence.cpp eval-sequence-recipe template inline auto make_double_space_sequence_memoizer(const sequence_recipe & recipe) { - return detail::make_sequence_memoizer>(recipe); + using key_t = unwrap_party_key_t; + return detail::make_sequence_memoizer>(recipe); } template @@ -506,10 +534,12 @@ inline auto make_double_space_sequence_memoizer(const DpfKey &, const sequence_r return make_double_space_sequence_memoizer(recipe); } +/// Full-tree sequence workspace bound to `recipe`. template inline auto make_full_tree_sequence_memoizer(const sequence_recipe & recipe) { - return detail::make_sequence_memoizer>(recipe); + using key_t = unwrap_party_key_t; + return detail::make_sequence_memoizer>(recipe); } template diff --git a/include/dpf/sequence_recipe.hpp b/include/dpf/sequence_recipe.hpp index 7da48af..6efa515 100644 --- a/include/dpf/sequence_recipe.hpp +++ b/include/dpf/sequence_recipe.hpp @@ -1,6 +1,9 @@ /// @file dpf/sequence_recipe.hpp -/// @brief -/// @details +/// @brief Compiled traversal of a sorted DPF point list. +/// @details `make_sequence_recipe` requires a nondecreasing range and throws +/// `std::runtime_error` otherwise. The recipe is independent of +/// correction words, so one recipe serves every key of that input +/// type. A sequence memoizer is bound to a particular recipe object. /// @author Ryan Henry /// @author Christopher Jiang /// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors) @@ -10,6 +13,8 @@ #ifndef LIBDPF_INCLUDE_DPF_SEQUENCE_RECIPE_HPP__ #define LIBDPF_INCLUDE_DPF_SEQUENCE_RECIPE_HPP__ +#include "hedley/hedley.h" + #include #include #include @@ -21,6 +26,7 @@ namespace dpf { +/// Steps, leaf count, and per-level endpoints for one sorted point list. struct sequence_recipe { public: @@ -34,10 +40,21 @@ struct sequence_recipe level_endpoints_{level_endpoints} { } + HEDLEY_PURE + HEDLEY_NO_THROW constexpr const std::vector & recipe_steps() const noexcept { return recipe_steps_; } + HEDLEY_PURE + HEDLEY_NO_THROW constexpr const std::vector & output_indices() const noexcept { return output_indices_; } + HEDLEY_PURE + HEDLEY_NO_THROW constexpr std::size_t num_leaf_nodes() const noexcept { return num_leaf_nodes_; } + HEDLEY_PURE + HEDLEY_NO_THROW constexpr const std::vector & level_endpoints() const noexcept { return level_endpoints_; } + /// `level_endpoints().size() - 1`. Not `constexpr`: `std::vector::size` is not a constant expression in C++17. + HEDLEY_PURE + HEDLEY_NO_THROW std::size_t depth() const noexcept { return level_endpoints_.size()-1; } private: @@ -122,6 +139,10 @@ auto make_sequence_recipe(ForwardIterator begin, ForwardIterator end) } // namespace detail +/// Compile `[begin, end)` into a recipe for `DpfKey`'s input type. +/// @tparam DpfKey Key type, or a `party_key` of that key. Only the input +/// type and depth are used. +/// @throws std::runtime_error if the range is not sorted nondecreasing. template auto make_sequence_recipe(ForwardIterator begin, ForwardIterator end) diff --git a/include/dpf/sequence_utils.hpp b/include/dpf/sequence_utils.hpp index 4f9392b..d853f0c 100644 --- a/include/dpf/sequence_utils.hpp +++ b/include/dpf/sequence_utils.hpp @@ -1,14 +1,23 @@ +/// @file dpf/sequence_utils.hpp +/// @brief Tags that select how `eval_sequence` stores each visited leaf. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + #ifndef LIBDPF_INCLUDE_DPF_SEQUENCE_UTILS_HPP__ #define LIBDPF_INCLUDE_DPF_SEQUENCE_UTILS_HPP__ namespace dpf { +/// Tag base for `eval_sequence` storage layout. struct return_type_tag_{}; +/// Store whole leaves. Default for `eval_sequence`. The iterable still +/// yields one share per listed point. struct return_entire_node_tag_ final : public return_type_tag_ {}; // static constexpr auto return_entire_node_tag = return_entire_node_tag_{}; +/// Store one share per listed point. struct return_output_only_tag_ final : public return_type_tag_ {}; // static constexpr auto return_output_only_tag = return_output_only_tag_{}; diff --git a/include/dpf/setbit_index_iterable.hpp b/include/dpf/setbit_index_iterable.hpp index d5f86cf..696fb16 100644 --- a/include/dpf/setbit_index_iterable.hpp +++ b/include/dpf/setbit_index_iterable.hpp @@ -43,7 +43,10 @@ class setbit_index_iterable explicit setbit_index_iterable(iter it) noexcept : it_{std::move(it)}, begin_{it_.it_.word_ptr_}, - end_{begin_ + it_.buf_size_}, + // `buf_size_` is a bit count (`bit_array::size()`). The sentinel word + // sits one past the last data word, and the end iterator has to land + // on it rather than `buf_size_` words later. + end_{begin_ + utils::quotient_ceiling(it_.buf_size_, bits_per_word)}, base_index_{calc_base_index(it_)}, length_{calc_length(it_)} { @@ -89,6 +92,7 @@ class setbit_index_iterable return it.outputs_ == 0 ? it.from_ : utils::quotient_floor(it.from_, it.outputs_) * it.outputs_; } + HEDLEY_NO_THROW static constexpr size_type calc_length(const iter & it) noexcept { return it.outputs_ == 0 ? utils::quotient_ceiling(it.to_, bits_per_word) * bits_per_word : utils::quotient_ceiling(it.to_, it.outputs_) * it.outputs_; @@ -102,7 +106,7 @@ class setbit_index_iterable // while other words just need some bits to be masked out constexpr void update_bit_array() { - if (it_.outputs_ == 0) + if (it_.outputs_ == 0 || begin_ == end_) { return; } @@ -124,11 +128,16 @@ class setbit_index_iterable cur = end_-1; loc = it_.to_ % it_.outputs_; - while (loc < it_.outputs_ - bits_per_word) + // `outputs_ - bits_per_word` underflows when a leaf is narrower than + // a word, and the loop then walks off the front of the buffer. + if (it_.outputs_ > bits_per_word) { - *cur = zero; - --cur; - loc += bits_per_word; + while (loc < it_.outputs_ - bits_per_word) + { + *cur = zero; + --cur; + loc += bits_per_word; + } } *cur &= mask; } @@ -151,12 +160,16 @@ class const_setbit_iterator using difference_type = std::ptrdiff_t; static constexpr auto bits_per_word = array_type::bits_per_word; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr const_setbit_iterator(const_setbit_iterator &&) noexcept = default; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr const_setbit_iterator(const const_setbit_iterator &) noexcept = default; + HEDLEY_NO_THROW ~const_setbit_iterator() noexcept = default; + HEDLEY_NO_THROW const_setbit_iterator & operator=(const_setbit_iterator &&) noexcept = default; const_setbit_iterator & operator=(const const_setbit_iterator &) = default; @@ -279,6 +292,7 @@ class const_setbit_iterator struct const_iterator_end_tag final {}; struct const_iterator_begin_tag final {}; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_NON_NULL() explicit constexpr const_setbit_iterator(word_pointer word_ptr, @@ -290,6 +304,7 @@ class const_setbit_iterator seek_to_next_bit(); } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_NON_NULL() explicit constexpr const_setbit_iterator(word_pointer word_ptr, @@ -303,13 +318,16 @@ class const_setbit_iterator word_type current_word_; size_type base_index_; + HEDLEY_NO_THROW friend const_setbit_iterator setbit_index_iterable::begin() const noexcept; + HEDLEY_NO_THROW friend const_setbit_iterator setbit_index_iterable::end() const noexcept; }; // class dpf::const_setbit_iterator template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW dpf::setbit_index_iterable indices_set_in(const subinterval_iterable> & iter) noexcept { return dpf::setbit_index_iterable{iter}; @@ -318,6 +336,7 @@ dpf::setbit_index_iterable indices_set_in(const subinterval_itera template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW dpf::setbit_index_iterable indices_set_in(subinterval_iterable> && iter) noexcept { return dpf::setbit_index_iterable{std::forward>>(iter)}; diff --git a/include/dpf/subsequence_iterable.hpp b/include/dpf/subsequence_iterable.hpp index 46bba95..48ac7d5 100644 --- a/include/dpf/subsequence_iterable.hpp +++ b/include/dpf/subsequence_iterable.hpp @@ -80,16 +80,20 @@ class subsequence_iterable using subsequence_iterator_type = points_iterator; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr const_iterator(output_iterator out_it, subsequence_iterator_type it) noexcept : out_it_{out_it}, it_{it} { } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr const_iterator(const_iterator &&) noexcept = default; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr const_iterator(const const_iterator &) noexcept = default; + HEDLEY_NO_THROW const_iterator & operator=(const_iterator &&) noexcept = default; const_iterator & operator=(const const_iterator &) = default; ~const_iterator() = default; @@ -143,11 +147,13 @@ class subsequence_iterable return *this; } + HEDLEY_NO_THROW const_iterator operator+(std::size_t n) const noexcept { return const_iterator(out_it_ + outputs_per_leaf*n, it_ + n); } + HEDLEY_NO_THROW const_iterator & operator-=(std::size_t n) noexcept { it_ -= n; @@ -155,16 +161,19 @@ class subsequence_iterable return *this; } + HEDLEY_NO_THROW const_iterator operator-(std::size_t n) const noexcept { return const_iterator(out_it_ - outputs_per_leaf*n, it_ - n); } + HEDLEY_NO_THROW difference_type operator-(const_iterator rhs) const noexcept { return it_ - rhs.it_; } + HEDLEY_NO_THROW reference operator[](std::size_t i) const noexcept { return out_it_[i*outputs_per_leaf + mod(it_[i], lg_outputs_per_leaf)]; @@ -279,17 +288,21 @@ class recipe_subsequence_iterable using subsequence_iterator_type = typename std::vector::const_iterator; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr const_iterator(output_iterator out_it, subsequence_iterator_type it) noexcept : out_it_{out_it}, it_{it} { } + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr const_iterator(const_iterator &&) noexcept = default; + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE constexpr const_iterator(const const_iterator &) noexcept = default; const_iterator & operator=(const const_iterator &) = default; + HEDLEY_NO_THROW const_iterator & operator=(const_iterator &&) noexcept = default; ~const_iterator() = default; @@ -339,27 +352,32 @@ class recipe_subsequence_iterable return *this; } + HEDLEY_NO_THROW const_iterator operator+(std::size_t n) const noexcept { return const_iterator(out_it_, it_ + n); } + HEDLEY_NO_THROW const_iterator & operator-=(std::size_t n) noexcept { it_ -= n; return *this; } + HEDLEY_NO_THROW const_iterator operator-(std::size_t n) const noexcept { return const_iterator(out_it_, it_ - n); } + HEDLEY_NO_THROW difference_type operator-(const_iterator rhs) const noexcept { return it_ - rhs.it_; } + HEDLEY_NO_THROW reference operator[](std::size_t i) const noexcept { return out_it_[it_[i]]; diff --git a/include/dpf/twobit.hpp b/include/dpf/twobit.hpp index 5238e2f..dec5828 100644 --- a/include/dpf/twobit.hpp +++ b/include/dpf/twobit.hpp @@ -152,6 +152,7 @@ template <> struct make_from_integral_value { using integral_type = std::uint8_t; + HEDLEY_NO_THROW constexpr dpf::twobit operator()(integral_type val) const noexcept { return dpf::to_twobit(val); @@ -184,8 +185,11 @@ class numeric_limits : public numeric_limits public: static constexpr int digits = 2; static constexpr int digits10 = 0; + HEDLEY_NO_THROW static constexpr dpf::twobit min() noexcept { return dpf::twobit::zero; } + HEDLEY_NO_THROW static constexpr dpf::twobit max() noexcept { return dpf::twobit::three; } + HEDLEY_NO_THROW static constexpr dpf::twobit lowest() noexcept { return min(); } }; diff --git a/include/dpf/uint256_t.hpp b/include/dpf/uint256_t.hpp index 3eda1c2..931cc79 100644 --- a/include/dpf/uint256_t.hpp +++ b/include/dpf/uint256_t.hpp @@ -1,3 +1,8 @@ +/// @file dpf/uint256_t.hpp +/// @brief Leaf arithmetic for `uint256_t` and the 128-bit SIMD lanes. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + #ifndef LIBDPF_INCLUDE_DPF_UINT256_T_HPP__ #define LIBDPF_INCLUDE_DPF_UINT256_T_HPP__ @@ -205,6 +210,7 @@ struct msb_of template <> struct mod_pow_2 { + HEDLEY_NO_THROW std::size_t operator()(uint128_t val, std::size_t n) const noexcept { return mod_pow_2{}(static_cast(val.lower()), n); @@ -214,6 +220,7 @@ struct mod_pow_2 template <> struct mod_pow_2 { + HEDLEY_NO_THROW std::size_t operator()(uint256_t val, std::size_t n) const noexcept { return mod_pow_2{}(val.lower(), n); @@ -227,6 +234,7 @@ struct to_integral_type using parent = to_integral_type_base; using typename parent::integral_type; + HEDLEY_NO_THROW constexpr integral_type operator()(uint128_t val) const noexcept { return (simde_uint128(val.upper()) << 64) | simde_uint128(val.lower()); @@ -240,6 +248,7 @@ struct to_integral_type using parent = to_integral_type_base; using typename parent::integral_type; + HEDLEY_NO_THROW constexpr integral_type operator()(uint256_t val) const noexcept { return val; diff --git a/include/dpf/utils.hpp b/include/dpf/utils.hpp index b0b3ed5..e8ebbf4 100644 --- a/include/dpf/utils.hpp +++ b/include/dpf/utils.hpp @@ -69,14 +69,23 @@ class numeric_limits<::uint128_t> static constexpr bool traps = false; static constexpr bool tinyness_before = false; + HEDLEY_NO_THROW static constexpr uint128_t min() noexcept { return uint128_t{0ul, 0ul}; } + HEDLEY_NO_THROW static constexpr uint128_t lowest() noexcept { return uint128_t{0ul, 0ul}; } + HEDLEY_NO_THROW static constexpr uint128_t max() noexcept { return uint128_t{-1ul, -1ul}; } + HEDLEY_NO_THROW static constexpr uint128_t epsilon() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint128_t round_error() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint128_t infinity() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint128_t quiet_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint128_t signaling_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint128_t denorm_min() noexcept { return 0; } }; @@ -127,14 +136,23 @@ class numeric_limits<::uint256_t> static constexpr bool traps = false; static constexpr bool tinyness_before = false; + HEDLEY_NO_THROW static constexpr uint256_t min() noexcept { return uint256_t{uint128_t{0ul, 0ul}, uint128_t{0ul, 0ul}}; } + HEDLEY_NO_THROW static constexpr uint256_t lowest() noexcept { return uint256_t{uint128_t{0ul, 0ul}, uint128_t{0ul, 0ul}}; } + HEDLEY_NO_THROW static constexpr uint256_t max() noexcept { return uint256_t{uint128_t{-1ul, -1ul}, uint128_t{-1ul, -1ul}}; } + HEDLEY_NO_THROW static constexpr uint256_t epsilon() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint256_t round_error() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint256_t infinity() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint256_t quiet_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint256_t signaling_NaN() noexcept { return 0; } + HEDLEY_NO_THROW static constexpr uint256_t denorm_min() noexcept { return 0; } }; @@ -266,6 +284,7 @@ auto make_bitset(Bools ...bs) } template +HEDLEY_NO_THROW static NodeT single_bit_mask(std::size_t i) noexcept; template <> @@ -449,6 +468,7 @@ struct make_from_integral_value // for `unsigned __int128`. using integral_type = typename make_signed_if && sizeof(S_integral_type) <= 8>::type; + HEDLEY_NO_THROW constexpr T operator()(integral_type val) const noexcept { return static_cast(val); @@ -459,6 +479,7 @@ struct make_from_integral_value /// `static_cast(x0 ^ x1)`: for `keyword`, `operator^` yields the parent /// `modint`, which cannot convert back through the private keyword ctor. template +HEDLEY_NO_THROW constexpr T xor_input_shares(T x0, T x1) noexcept { constexpr auto to_int = to_integral_type{}; @@ -490,6 +511,7 @@ static constexpr IntegralT get_node_mask(InputT mask, std::size_t level_index) /// width is undefined for the native unsigned types). template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr IntegralT shift_right(IntegralT value, std::size_t offset) noexcept { if (offset >= bitlength_of_v) @@ -500,6 +522,7 @@ constexpr IntegralT shift_right(IntegralT value, std::size_t offset) noexcept /// Floor of `from_inclusive / 2^lg_opl`. `lg_opl` is `log2(outputs_per_leaf)`. template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr IntegralT leaf_node_floor(IntegralT from_inclusive, std::size_t lg_opl) noexcept { if (lg_opl == 0) @@ -514,6 +537,7 @@ constexpr IntegralT leaf_node_floor(IntegralT from_inclusive, std::size_t lg_opl /// end (`[from, 2^width)`), which `split_leaf_nodes` interprets. template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW constexpr IntegralT leaf_node_ceil_exclusive(IntegralT to_inclusive, std::size_t lg_opl) noexcept { constexpr std::size_t width = bitlength_of_v; @@ -585,25 +609,87 @@ struct node_segments std::size_t total = 0; }; +/// True when the inclusive walk `[from, to]` wraps the low `bits` of the +/// domain. Comparison is on the post-MSB-flip bit pattern. Leaf ids alone +/// cannot carry this: packing can put a wrapping pair into `from_node <= to_node`. +template +inline bool interval_wraps(IntegralT from, IntegralT to, std::size_t bits) +{ + if (bits == 0) + return false; + if (bits < bitlength_of_v) + { + const IntegralT mask = static_cast( + (IntegralT{1} << bits) - IntegralT{1}); + return (from & mask) > (to & mask); + } + return from > to; +} + /// Split an inclusive output interval, already reduced to leaf ids, into one /// or two half-open walks. A linearized `from_node > to_node` wraps the node /// id space `[0, 2^depth)`. A saturated `to_node == 0` means the exclusive end /// is `2^{bitwidth(IntegralT)}`, which is the whole id space when `depth` is /// that width. +/// +/// `input_wraps` is the order of the original inputs, before leaf coarsening. +/// The buffer is still two runs, `[from_node, 2^depth)` then `[0, to_node)`, +/// even when packing makes `from_node <= to_node`. In that case the runs +/// overlap on the shared leaf: the iterable's preclip consumes the start of +/// the first copy and its length stops inside the second. Collapsing the +/// overlap into one forward segment writes the wrong leaves. template inline node_segments split_leaf_nodes(IntegralT from_node, - IntegralT to_node, std::size_t depth) + IntegralT to_node, std::size_t depth, bool input_wraps = false) { node_segments out; + constexpr std::size_t width = bitlength_of_v; + constexpr std::size_t size_digits = bitlength_of_v; auto push = [&](IntegralT lo, IntegralT hi) { - const std::size_t count = static_cast(hi - lo); + std::size_t count = 0; + if (hi == IntegralT{0} && lo != IntegralT{0}) + { + // Exclusive end is 2^width. The count fits in size_t only when + // that power is one past size_t's maximum and lo is nonzero. + if (width > size_digits) + throw std::length_error("DPF leaf domain does not fit in size_t"); + count = static_cast(0) - static_cast(lo); + } + else + { + const auto wide = hi - lo; + if (wide > IntegralT(std::numeric_limits::max())) + throw std::length_error("DPF leaf domain does not fit in size_t"); + count = static_cast(wide); + } if (count == 0) return; + if (out.total > std::numeric_limits::max() - count) + throw std::length_error("DPF leaf domain does not fit in size_t"); out.seg[out.n++] = node_segment{lo, hi, count}; out.total += count; }; + if (input_wraps) + { + if (depth >= width) + { + if (from_node == IntegralT{0}) + throw std::length_error("DPF leaf domain does not fit in size_t"); + push(from_node, IntegralT{0}); + if (to_node != IntegralT{0}) + push(IntegralT{0}, to_node); + return out; + } + const IntegralT domain_end = static_cast(IntegralT{1} << depth); + if (from_node < domain_end) + push(from_node, domain_end); + if (to_node != IntegralT{0}) + push(IntegralT{0}, to_node); + return out; + } + if (to_node != IntegralT{0} && from_node < to_node) { push(from_node, to_node); @@ -612,7 +698,6 @@ inline node_segments split_leaf_nodes(IntegralT from_node, if (to_node != IntegralT{0} && from_node == to_node) return out; - constexpr std::size_t width = bitlength_of_v; if (from_node == IntegralT{0} && to_node == IntegralT{0}) throw std::length_error("DPF leaf domain does not fit in size_t"); @@ -638,14 +723,26 @@ static constexpr std::size_t get_leafnodes_in_node_interval(IntegralT from_node, return static_cast(to_node - from_node); } +template +inline void flip_msb_if_signed_integral(T & x); + template static std::size_t get_leafnodes_in_output_interval(InputT from, InputT to) { - return split_leaf_nodes(get_from_node(from), - get_to_node(to), - static_cast(DpfKey::depth)).total; + // Match eval: the walk order is the bit pattern after the sign flip. + InputT flipped_from = from; + InputT flipped_to = to; + flip_msb_if_signed_integral(flipped_from); + flip_msb_if_signed_integral(flipped_to); + constexpr auto to_int = to_integral_type{}; + const auto from_i = static_cast(to_int(flipped_from)); + const auto to_i = static_cast(to_int(flipped_to)); + const bool wraps = interval_wraps(from_i, to_i, bitlength_of_v); + return split_leaf_nodes(get_from_node(flipped_from), + get_to_node(flipped_to), + static_cast(DpfKey::depth), wraps).total; } /// Historical name used by the test suite. @@ -660,6 +757,7 @@ static std::size_t get_nodes_in_interval(InputT from, InputT to) template struct mod_pow_2 { + HEDLEY_NO_THROW std::size_t operator()(T val, std::size_t n) const noexcept { if (n == 0) @@ -696,6 +794,7 @@ static constexpr auto msb_of_v = msb_of::value; template struct countl_zero { + HEDLEY_NO_THROW HEDLEY_CONST HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(T val) const noexcept @@ -724,6 +823,7 @@ struct countl_zero template struct countr_zero { + HEDLEY_NO_THROW HEDLEY_CONST HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(T val) const noexcept @@ -756,6 +856,7 @@ struct countr_zero template struct countl_zero_symmetric_difference { + HEDLEY_NO_THROW HEDLEY_CONST HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(T lhs, T rhs) const noexcept @@ -773,6 +874,7 @@ struct countl_zero { using T = simde_int128; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -795,6 +897,7 @@ struct countl_zero { using T = simde_uint128; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -812,6 +915,7 @@ struct countl_zero { using T = uint128_t; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -828,6 +932,7 @@ struct countl_zero { using T = uint256_t; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -844,6 +949,7 @@ struct countl_zero { using T = simde__m128i; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE std::size_t operator()(const T & val) const noexcept @@ -860,6 +966,7 @@ template <> struct countl_zero { using T = simde__m256i; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -882,6 +989,7 @@ struct countr_zero { using T = simde_int128; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -898,6 +1006,7 @@ struct countr_zero { using T = simde_uint128; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -914,6 +1023,7 @@ struct countr_zero { using T = uint128_t; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -930,6 +1040,7 @@ struct countr_zero { using T = uint256_t; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -946,6 +1057,7 @@ struct countr_zero { using T = simde__m128i; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE std::size_t operator()(const T & val) const noexcept @@ -963,6 +1075,7 @@ struct countr_zero { using T = simde__m256i; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & val) const noexcept @@ -1026,13 +1139,19 @@ static constexpr std::size_t packed_lane_bits_v = packed_lane_bits>::value; template -auto data(T & bar) // NOLINT(runtime/references) +HEDLEY_PURE +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr auto data(T & bar) noexcept // NOLINT(runtime/references) { return std::data(bar); } +/// Pointer overload. Constness of `bar` is the constness of `T`. template -auto data(T * bar) +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr T * data(T * bar) noexcept { return bar; } @@ -1063,7 +1182,10 @@ template static constexpr bool is_tuple_v = is_tuple::value; template -auto & get(T & t) // NOLINT(runtime/references) +HEDLEY_PURE +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr auto & get(T & t) noexcept // NOLINT(runtime/references) { if constexpr(I == 0 && is_tuple_v == false) { @@ -1085,7 +1207,10 @@ template static constexpr bool is_bit_array_v = is_bit_array::value; template -auto size(const T & t) +HEDLEY_PURE +HEDLEY_NO_THROW +HEDLEY_ALWAYS_INLINE +constexpr auto size(const T & t) noexcept { if constexpr(is_bit_array_v == false) { diff --git a/include/dpf/wildcard.hpp b/include/dpf/wildcard.hpp index 5d72c25..857c0c7 100644 --- a/include/dpf/wildcard.hpp +++ b/include/dpf/wildcard.hpp @@ -44,8 +44,11 @@ struct wildcard_value static_assert(std::numeric_limits::is_iec559 || !(std::is_same_v || std::is_same_v), "floating point types only supported for iec559"); + HEDLEY_NO_THROW inline constexpr wildcard_value() noexcept : val{std::nullopt} { } + HEDLEY_NO_THROW inline constexpr wildcard_value(const T & t) noexcept : val{t} { } + HEDLEY_NO_THROW inline constexpr wildcard_value(T && t) noexcept : val{std::move(t)} { } HEDLEY_ALWAYS_INLINE @@ -93,6 +96,7 @@ template using concrete_type_t = typename concrete_type::type; template struct concrete_value { + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_CONST constexpr std::optional operator()(T y) const noexcept @@ -102,6 +106,7 @@ template struct concrete_value }; template struct concrete_value> { + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_CONST constexpr std::optional operator()(wildcard_value) const noexcept diff --git a/include/dpf/xor_wrapper.hpp b/include/dpf/xor_wrapper.hpp index 4206a18..5bd4299 100644 --- a/include/dpf/xor_wrapper.hpp +++ b/include/dpf/xor_wrapper.hpp @@ -49,13 +49,16 @@ struct xor_wrapper constexpr xor_wrapper() = default; /// @brief Copy c'tor + HEDLEY_NO_THROW constexpr xor_wrapper(const xor_wrapper &) noexcept = default; /// @brief Move c'tor + HEDLEY_NO_THROW constexpr xor_wrapper(xor_wrapper &&) noexcept = default; /// @brief Value c'tor // cppcheck-suppress noExplicitConstructor + HEDLEY_NO_THROW constexpr xor_wrapper(value_type v) noexcept : value{v} { } // NOLINT(runtime/explicit) /// @} @@ -163,7 +166,7 @@ struct xor_wrapper HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW - constexpr xor_wrapper & operator<<=(std::size_t amount) + constexpr xor_wrapper & operator<<=(std::size_t amount) noexcept { this->value <<= amount; return *this; @@ -171,7 +174,7 @@ struct xor_wrapper HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW - constexpr xor_wrapper & operator>>=(std::size_t amount) + constexpr xor_wrapper & operator>>=(std::size_t amount) noexcept { this->value >>= amount; return *this; @@ -274,14 +277,14 @@ struct xor_wrapper HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW - friend constexpr xor_wrapper operator<<(const xor_wrapper & val, std::size_t amount) + friend constexpr xor_wrapper operator<<(const xor_wrapper & val, std::size_t amount) noexcept { return xor_wrapper(val.value << amount); } HEDLEY_ALWAYS_INLINE HEDLEY_NO_THROW - friend constexpr xor_wrapper operator>>(const xor_wrapper & val, std::size_t amount) + friend constexpr xor_wrapper operator>>(const xor_wrapper & val, std::size_t amount) noexcept { return xor_wrapper(val.value >> amount); } @@ -670,217 +673,217 @@ constexpr static auto operator "" _x61(unsigned long long int x) { return dpf::x constexpr static auto operator "" _x62(unsigned long long int x) { return dpf::xints::xint62_t{static_cast(x)}; } constexpr static auto operator "" _x63(unsigned long long int x) { return dpf::xints::xint63_t{static_cast(x)}; } constexpr static auto operator "" _x64(unsigned long long int x) { return dpf::xints::xint64_t{static_cast(x)}; } -template constexpr static auto operator "" _x65() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint65_t{x}; } -template constexpr static auto operator "" _x66() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint66_t{x}; } -template constexpr static auto operator "" _x67() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint67_t{x}; } -template constexpr static auto operator "" _x68() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint68_t{x}; } -template constexpr static auto operator "" _x69() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint69_t{x}; } +template constexpr static auto operator "" _x65() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint65_t{x}; } +template constexpr static auto operator "" _x66() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint66_t{x}; } +template constexpr static auto operator "" _x67() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint67_t{x}; } +template constexpr static auto operator "" _x68() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint68_t{x}; } +template constexpr static auto operator "" _x69() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint69_t{x}; } // 70--79 -template constexpr static auto operator "" _x70() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint70_t{x}; } -template constexpr static auto operator "" _x71() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint71_t{x}; } -template constexpr static auto operator "" _x72() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint72_t{x}; } -template constexpr static auto operator "" _x73() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint73_t{x}; } -template constexpr static auto operator "" _x74() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint74_t{x}; } -template constexpr static auto operator "" _x75() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint75_t{x}; } -template constexpr static auto operator "" _x76() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint76_t{x}; } -template constexpr static auto operator "" _x77() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint77_t{x}; } -template constexpr static auto operator "" _x78() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint78_t{x}; } -template constexpr static auto operator "" _x79() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint79_t{x}; } +template constexpr static auto operator "" _x70() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint70_t{x}; } +template constexpr static auto operator "" _x71() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint71_t{x}; } +template constexpr static auto operator "" _x72() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint72_t{x}; } +template constexpr static auto operator "" _x73() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint73_t{x}; } +template constexpr static auto operator "" _x74() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint74_t{x}; } +template constexpr static auto operator "" _x75() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint75_t{x}; } +template constexpr static auto operator "" _x76() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint76_t{x}; } +template constexpr static auto operator "" _x77() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint77_t{x}; } +template constexpr static auto operator "" _x78() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint78_t{x}; } +template constexpr static auto operator "" _x79() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint79_t{x}; } // 80--89 -template constexpr static auto operator "" _x80() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint80_t{x}; } -template constexpr static auto operator "" _x81() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint81_t{x}; } -template constexpr static auto operator "" _x82() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint82_t{x}; } -template constexpr static auto operator "" _x83() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint83_t{x}; } -template constexpr static auto operator "" _x84() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint84_t{x}; } -template constexpr static auto operator "" _x85() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint85_t{x}; } -template constexpr static auto operator "" _x86() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint86_t{x}; } -template constexpr static auto operator "" _x87() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint87_t{x}; } -template constexpr static auto operator "" _x88() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint88_t{x}; } -template constexpr static auto operator "" _x89() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint89_t{x}; } +template constexpr static auto operator "" _x80() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint80_t{x}; } +template constexpr static auto operator "" _x81() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint81_t{x}; } +template constexpr static auto operator "" _x82() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint82_t{x}; } +template constexpr static auto operator "" _x83() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint83_t{x}; } +template constexpr static auto operator "" _x84() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint84_t{x}; } +template constexpr static auto operator "" _x85() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint85_t{x}; } +template constexpr static auto operator "" _x86() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint86_t{x}; } +template constexpr static auto operator "" _x87() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint87_t{x}; } +template constexpr static auto operator "" _x88() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint88_t{x}; } +template constexpr static auto operator "" _x89() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint89_t{x}; } // 90--99 -template constexpr static auto operator "" _x90() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint90_t{x}; } -template constexpr static auto operator "" _x91() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint91_t{x}; } -template constexpr static auto operator "" _x92() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint92_t{x}; } -template constexpr static auto operator "" _x93() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint93_t{x}; } -template constexpr static auto operator "" _x94() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint94_t{x}; } -template constexpr static auto operator "" _x95() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint95_t{x}; } -template constexpr static auto operator "" _x96() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint96_t{x}; } -template constexpr static auto operator "" _x97() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint97_t{x}; } -template constexpr static auto operator "" _x98() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint98_t{x}; } -template constexpr static auto operator "" _x99() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint99_t{x}; } +template constexpr static auto operator "" _x90() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint90_t{x}; } +template constexpr static auto operator "" _x91() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint91_t{x}; } +template constexpr static auto operator "" _x92() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint92_t{x}; } +template constexpr static auto operator "" _x93() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint93_t{x}; } +template constexpr static auto operator "" _x94() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint94_t{x}; } +template constexpr static auto operator "" _x95() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint95_t{x}; } +template constexpr static auto operator "" _x96() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint96_t{x}; } +template constexpr static auto operator "" _x97() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint97_t{x}; } +template constexpr static auto operator "" _x98() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint98_t{x}; } +template constexpr static auto operator "" _x99() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint99_t{x}; } // 100--109 -template constexpr static auto operator "" _x100() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint100_t{x}; } -template constexpr static auto operator "" _x101() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint101_t{x}; } -template constexpr static auto operator "" _x102() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint102_t{x}; } -template constexpr static auto operator "" _x103() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint103_t{x}; } -template constexpr static auto operator "" _x104() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint104_t{x}; } -template constexpr static auto operator "" _x105() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint105_t{x}; } -template constexpr static auto operator "" _x106() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint106_t{x}; } -template constexpr static auto operator "" _x107() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint107_t{x}; } -template constexpr static auto operator "" _x108() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint108_t{x}; } -template constexpr static auto operator "" _x109() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint109_t{x}; } +template constexpr static auto operator "" _x100() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint100_t{x}; } +template constexpr static auto operator "" _x101() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint101_t{x}; } +template constexpr static auto operator "" _x102() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint102_t{x}; } +template constexpr static auto operator "" _x103() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint103_t{x}; } +template constexpr static auto operator "" _x104() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint104_t{x}; } +template constexpr static auto operator "" _x105() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint105_t{x}; } +template constexpr static auto operator "" _x106() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint106_t{x}; } +template constexpr static auto operator "" _x107() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint107_t{x}; } +template constexpr static auto operator "" _x108() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint108_t{x}; } +template constexpr static auto operator "" _x109() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint109_t{x}; } // 110--119 -template constexpr static auto operator "" _x110() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint110_t{x}; } -template constexpr static auto operator "" _x111() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint111_t{x}; } -template constexpr static auto operator "" _x112() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint112_t{x}; } -template constexpr static auto operator "" _x113() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint113_t{x}; } -template constexpr static auto operator "" _x114() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint114_t{x}; } -template constexpr static auto operator "" _x115() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint115_t{x}; } -template constexpr static auto operator "" _x116() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint116_t{x}; } -template constexpr static auto operator "" _x117() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint117_t{x}; } -template constexpr static auto operator "" _x118() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint118_t{x}; } -template constexpr static auto operator "" _x119() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint119_t{x}; } +template constexpr static auto operator "" _x110() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint110_t{x}; } +template constexpr static auto operator "" _x111() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint111_t{x}; } +template constexpr static auto operator "" _x112() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint112_t{x}; } +template constexpr static auto operator "" _x113() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint113_t{x}; } +template constexpr static auto operator "" _x114() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint114_t{x}; } +template constexpr static auto operator "" _x115() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint115_t{x}; } +template constexpr static auto operator "" _x116() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint116_t{x}; } +template constexpr static auto operator "" _x117() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint117_t{x}; } +template constexpr static auto operator "" _x118() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint118_t{x}; } +template constexpr static auto operator "" _x119() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint119_t{x}; } // 120--129 -template constexpr static auto operator "" _x120() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint120_t{x}; } -template constexpr static auto operator "" _x121() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint121_t{x}; } -template constexpr static auto operator "" _x122() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint122_t{x}; } -template constexpr static auto operator "" _x123() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint123_t{x}; } -template constexpr static auto operator "" _x124() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint124_t{x}; } -template constexpr static auto operator "" _x125() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint125_t{x}; } -template constexpr static auto operator "" _x126() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint126_t{x}; } -template constexpr static auto operator "" _x127() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint127_t{x}; } -template constexpr static auto operator "" _x128() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint128_t{x}; } -template constexpr static auto operator "" _x129() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint129_t{x}; } +template constexpr static auto operator "" _x120() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint120_t{x}; } +template constexpr static auto operator "" _x121() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint121_t{x}; } +template constexpr static auto operator "" _x122() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint122_t{x}; } +template constexpr static auto operator "" _x123() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint123_t{x}; } +template constexpr static auto operator "" _x124() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint124_t{x}; } +template constexpr static auto operator "" _x125() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint125_t{x}; } +template constexpr static auto operator "" _x126() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint126_t{x}; } +template constexpr static auto operator "" _x127() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint127_t{x}; } +template constexpr static auto operator "" _x128() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); simde_uint128 x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint128_t{x}; } +template constexpr static auto operator "" _x129() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint129_t{x}; } // 130--139 -template constexpr static auto operator "" _x130() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint130_t{x}; } -template constexpr static auto operator "" _x131() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint131_t{x}; } -template constexpr static auto operator "" _x132() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint132_t{x}; } -template constexpr static auto operator "" _x133() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint133_t{x}; } -template constexpr static auto operator "" _x134() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint134_t{x}; } -template constexpr static auto operator "" _x135() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint135_t{x}; } -template constexpr static auto operator "" _x136() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint136_t{x}; } -template constexpr static auto operator "" _x137() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint137_t{x}; } -template constexpr static auto operator "" _x138() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint138_t{x}; } -template constexpr static auto operator "" _x139() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint139_t{x}; } +template constexpr static auto operator "" _x130() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint130_t{x}; } +template constexpr static auto operator "" _x131() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint131_t{x}; } +template constexpr static auto operator "" _x132() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint132_t{x}; } +template constexpr static auto operator "" _x133() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint133_t{x}; } +template constexpr static auto operator "" _x134() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint134_t{x}; } +template constexpr static auto operator "" _x135() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint135_t{x}; } +template constexpr static auto operator "" _x136() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint136_t{x}; } +template constexpr static auto operator "" _x137() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint137_t{x}; } +template constexpr static auto operator "" _x138() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint138_t{x}; } +template constexpr static auto operator "" _x139() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint139_t{x}; } // 140--149 -template constexpr static auto operator "" _x140() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint140_t{x}; } -template constexpr static auto operator "" _x141() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint141_t{x}; } -template constexpr static auto operator "" _x142() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint142_t{x}; } -template constexpr static auto operator "" _x143() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint143_t{x}; } -template constexpr static auto operator "" _x144() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint144_t{x}; } -template constexpr static auto operator "" _x145() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint145_t{x}; } -template constexpr static auto operator "" _x146() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint146_t{x}; } -template constexpr static auto operator "" _x147() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint147_t{x}; } -template constexpr static auto operator "" _x148() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint148_t{x}; } -template constexpr static auto operator "" _x149() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint149_t{x}; } +template constexpr static auto operator "" _x140() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint140_t{x}; } +template constexpr static auto operator "" _x141() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint141_t{x}; } +template constexpr static auto operator "" _x142() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint142_t{x}; } +template constexpr static auto operator "" _x143() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint143_t{x}; } +template constexpr static auto operator "" _x144() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint144_t{x}; } +template constexpr static auto operator "" _x145() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint145_t{x}; } +template constexpr static auto operator "" _x146() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint146_t{x}; } +template constexpr static auto operator "" _x147() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint147_t{x}; } +template constexpr static auto operator "" _x148() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint148_t{x}; } +template constexpr static auto operator "" _x149() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint149_t{x}; } // 150--159 -template constexpr static auto operator "" _x150() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint150_t{x}; } -template constexpr static auto operator "" _x151() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint151_t{x}; } -template constexpr static auto operator "" _x152() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint152_t{x}; } -template constexpr static auto operator "" _x153() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint153_t{x}; } -template constexpr static auto operator "" _x154() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint154_t{x}; } -template constexpr static auto operator "" _x155() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint155_t{x}; } -template constexpr static auto operator "" _x156() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint156_t{x}; } -template constexpr static auto operator "" _x157() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint157_t{x}; } -template constexpr static auto operator "" _x158() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint158_t{x}; } -template constexpr static auto operator "" _x159() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint159_t{x}; } +template constexpr static auto operator "" _x150() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint150_t{x}; } +template constexpr static auto operator "" _x151() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint151_t{x}; } +template constexpr static auto operator "" _x152() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint152_t{x}; } +template constexpr static auto operator "" _x153() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint153_t{x}; } +template constexpr static auto operator "" _x154() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint154_t{x}; } +template constexpr static auto operator "" _x155() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint155_t{x}; } +template constexpr static auto operator "" _x156() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint156_t{x}; } +template constexpr static auto operator "" _x157() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint157_t{x}; } +template constexpr static auto operator "" _x158() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint158_t{x}; } +template constexpr static auto operator "" _x159() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint159_t{x}; } // 160--169 -template constexpr static auto operator "" _x160() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint160_t{x}; } -template constexpr static auto operator "" _x161() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint161_t{x}; } -template constexpr static auto operator "" _x162() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint162_t{x}; } -template constexpr static auto operator "" _x163() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint163_t{x}; } -template constexpr static auto operator "" _x164() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint164_t{x}; } -template constexpr static auto operator "" _x165() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint165_t{x}; } -template constexpr static auto operator "" _x166() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint166_t{x}; } -template constexpr static auto operator "" _x167() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint167_t{x}; } -template constexpr static auto operator "" _x168() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint168_t{x}; } -template constexpr static auto operator "" _x169() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint169_t{x}; } +template constexpr static auto operator "" _x160() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint160_t{x}; } +template constexpr static auto operator "" _x161() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint161_t{x}; } +template constexpr static auto operator "" _x162() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint162_t{x}; } +template constexpr static auto operator "" _x163() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint163_t{x}; } +template constexpr static auto operator "" _x164() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint164_t{x}; } +template constexpr static auto operator "" _x165() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint165_t{x}; } +template constexpr static auto operator "" _x166() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint166_t{x}; } +template constexpr static auto operator "" _x167() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint167_t{x}; } +template constexpr static auto operator "" _x168() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint168_t{x}; } +template constexpr static auto operator "" _x169() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint169_t{x}; } // 170--179 -template constexpr static auto operator "" _x170() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint170_t{x}; } -template constexpr static auto operator "" _x171() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint171_t{x}; } -template constexpr static auto operator "" _x172() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint172_t{x}; } -template constexpr static auto operator "" _x173() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint173_t{x}; } -template constexpr static auto operator "" _x174() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint174_t{x}; } -template constexpr static auto operator "" _x175() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint175_t{x}; } -template constexpr static auto operator "" _x176() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint176_t{x}; } -template constexpr static auto operator "" _x177() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint177_t{x}; } -template constexpr static auto operator "" _x178() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint178_t{x}; } -template constexpr static auto operator "" _x179() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint179_t{x}; } +template constexpr static auto operator "" _x170() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint170_t{x}; } +template constexpr static auto operator "" _x171() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint171_t{x}; } +template constexpr static auto operator "" _x172() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint172_t{x}; } +template constexpr static auto operator "" _x173() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint173_t{x}; } +template constexpr static auto operator "" _x174() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint174_t{x}; } +template constexpr static auto operator "" _x175() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint175_t{x}; } +template constexpr static auto operator "" _x176() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint176_t{x}; } +template constexpr static auto operator "" _x177() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint177_t{x}; } +template constexpr static auto operator "" _x178() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint178_t{x}; } +template constexpr static auto operator "" _x179() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint179_t{x}; } // 180--189 -template constexpr static auto operator "" _x180() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint180_t{x}; } -template constexpr static auto operator "" _x181() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint181_t{x}; } -template constexpr static auto operator "" _x182() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint182_t{x}; } -template constexpr static auto operator "" _x183() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint183_t{x}; } -template constexpr static auto operator "" _x184() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint184_t{x}; } -template constexpr static auto operator "" _x185() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint185_t{x}; } -template constexpr static auto operator "" _x186() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint186_t{x}; } -template constexpr static auto operator "" _x187() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint187_t{x}; } -template constexpr static auto operator "" _x188() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint188_t{x}; } -template constexpr static auto operator "" _x189() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint189_t{x}; } +template constexpr static auto operator "" _x180() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint180_t{x}; } +template constexpr static auto operator "" _x181() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint181_t{x}; } +template constexpr static auto operator "" _x182() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint182_t{x}; } +template constexpr static auto operator "" _x183() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint183_t{x}; } +template constexpr static auto operator "" _x184() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint184_t{x}; } +template constexpr static auto operator "" _x185() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint185_t{x}; } +template constexpr static auto operator "" _x186() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint186_t{x}; } +template constexpr static auto operator "" _x187() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint187_t{x}; } +template constexpr static auto operator "" _x188() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint188_t{x}; } +template constexpr static auto operator "" _x189() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint189_t{x}; } // 190--199 -template constexpr static auto operator "" _x190() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint190_t{x}; } -template constexpr static auto operator "" _x191() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint191_t{x}; } -template constexpr static auto operator "" _x192() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint192_t{x}; } -template constexpr static auto operator "" _x193() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint193_t{x}; } -template constexpr static auto operator "" _x194() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint194_t{x}; } -template constexpr static auto operator "" _x195() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint195_t{x}; } -template constexpr static auto operator "" _x196() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint196_t{x}; } -template constexpr static auto operator "" _x197() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint197_t{x}; } -template constexpr static auto operator "" _x198() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint198_t{x}; } -template constexpr static auto operator "" _x199() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint199_t{x}; } +template constexpr static auto operator "" _x190() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint190_t{x}; } +template constexpr static auto operator "" _x191() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint191_t{x}; } +template constexpr static auto operator "" _x192() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint192_t{x}; } +template constexpr static auto operator "" _x193() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint193_t{x}; } +template constexpr static auto operator "" _x194() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint194_t{x}; } +template constexpr static auto operator "" _x195() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint195_t{x}; } +template constexpr static auto operator "" _x196() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint196_t{x}; } +template constexpr static auto operator "" _x197() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint197_t{x}; } +template constexpr static auto operator "" _x198() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint198_t{x}; } +template constexpr static auto operator "" _x199() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint199_t{x}; } // 200--209 -template constexpr static auto operator "" _x200() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint200_t{x}; } -template constexpr static auto operator "" _x201() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint201_t{x}; } -template constexpr static auto operator "" _x202() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint202_t{x}; } -template constexpr static auto operator "" _x203() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint203_t{x}; } -template constexpr static auto operator "" _x204() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint204_t{x}; } -template constexpr static auto operator "" _x205() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint205_t{x}; } -template constexpr static auto operator "" _x206() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint206_t{x}; } -template constexpr static auto operator "" _x207() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint207_t{x}; } -template constexpr static auto operator "" _x208() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint208_t{x}; } -template constexpr static auto operator "" _x209() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint209_t{x}; } +template constexpr static auto operator "" _x200() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint200_t{x}; } +template constexpr static auto operator "" _x201() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint201_t{x}; } +template constexpr static auto operator "" _x202() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint202_t{x}; } +template constexpr static auto operator "" _x203() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint203_t{x}; } +template constexpr static auto operator "" _x204() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint204_t{x}; } +template constexpr static auto operator "" _x205() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint205_t{x}; } +template constexpr static auto operator "" _x206() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint206_t{x}; } +template constexpr static auto operator "" _x207() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint207_t{x}; } +template constexpr static auto operator "" _x208() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint208_t{x}; } +template constexpr static auto operator "" _x209() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint209_t{x}; } // 210--219 -template constexpr static auto operator "" _x210() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint210_t{x}; } -template constexpr static auto operator "" _x211() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint211_t{x}; } -template constexpr static auto operator "" _x212() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint212_t{x}; } -template constexpr static auto operator "" _x213() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint213_t{x}; } -template constexpr static auto operator "" _x214() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint214_t{x}; } -template constexpr static auto operator "" _x215() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint215_t{x}; } -template constexpr static auto operator "" _x216() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint216_t{x}; } -template constexpr static auto operator "" _x217() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint217_t{x}; } -template constexpr static auto operator "" _x218() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint218_t{x}; } -template constexpr static auto operator "" _x219() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint219_t{x}; } +template constexpr static auto operator "" _x210() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint210_t{x}; } +template constexpr static auto operator "" _x211() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint211_t{x}; } +template constexpr static auto operator "" _x212() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint212_t{x}; } +template constexpr static auto operator "" _x213() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint213_t{x}; } +template constexpr static auto operator "" _x214() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint214_t{x}; } +template constexpr static auto operator "" _x215() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint215_t{x}; } +template constexpr static auto operator "" _x216() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint216_t{x}; } +template constexpr static auto operator "" _x217() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint217_t{x}; } +template constexpr static auto operator "" _x218() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint218_t{x}; } +template constexpr static auto operator "" _x219() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint219_t{x}; } // 220--229 -template constexpr static auto operator "" _x220() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint220_t{x}; } -template constexpr static auto operator "" _x221() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint221_t{x}; } -template constexpr static auto operator "" _x222() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint222_t{x}; } -template constexpr static auto operator "" _x223() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint223_t{x}; } -template constexpr static auto operator "" _x224() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint224_t{x}; } -template constexpr static auto operator "" _x225() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint225_t{x}; } -template constexpr static auto operator "" _x226() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint226_t{x}; } -template constexpr static auto operator "" _x227() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint227_t{x}; } -template constexpr static auto operator "" _x228() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint228_t{x}; } -template constexpr static auto operator "" _x229() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint229_t{x}; } +template constexpr static auto operator "" _x220() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint220_t{x}; } +template constexpr static auto operator "" _x221() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint221_t{x}; } +template constexpr static auto operator "" _x222() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint222_t{x}; } +template constexpr static auto operator "" _x223() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint223_t{x}; } +template constexpr static auto operator "" _x224() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint224_t{x}; } +template constexpr static auto operator "" _x225() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint225_t{x}; } +template constexpr static auto operator "" _x226() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint226_t{x}; } +template constexpr static auto operator "" _x227() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint227_t{x}; } +template constexpr static auto operator "" _x228() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint228_t{x}; } +template constexpr static auto operator "" _x229() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint229_t{x}; } // 230--239 -template constexpr static auto operator "" _x230() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint230_t{x}; } -template constexpr static auto operator "" _x231() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint231_t{x}; } -template constexpr static auto operator "" _x232() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint232_t{x}; } -template constexpr static auto operator "" _x233() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint233_t{x}; } -template constexpr static auto operator "" _x234() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint234_t{x}; } -template constexpr static auto operator "" _x235() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint235_t{x}; } -template constexpr static auto operator "" _x236() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint236_t{x}; } -template constexpr static auto operator "" _x237() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint237_t{x}; } -template constexpr static auto operator "" _x238() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint238_t{x}; } -template constexpr static auto operator "" _x239() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint239_t{x}; } +template constexpr static auto operator "" _x230() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint230_t{x}; } +template constexpr static auto operator "" _x231() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint231_t{x}; } +template constexpr static auto operator "" _x232() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint232_t{x}; } +template constexpr static auto operator "" _x233() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint233_t{x}; } +template constexpr static auto operator "" _x234() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint234_t{x}; } +template constexpr static auto operator "" _x235() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint235_t{x}; } +template constexpr static auto operator "" _x236() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint236_t{x}; } +template constexpr static auto operator "" _x237() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint237_t{x}; } +template constexpr static auto operator "" _x238() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint238_t{x}; } +template constexpr static auto operator "" _x239() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint239_t{x}; } // 240--249 -template constexpr static auto operator "" _x240() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint240_t{x}; } -template constexpr static auto operator "" _x241() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint241_t{x}; } -template constexpr static auto operator "" _x242() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint242_t{x}; } -template constexpr static auto operator "" _x243() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint243_t{x}; } -template constexpr static auto operator "" _x244() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint244_t{x}; } -template constexpr static auto operator "" _x245() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint245_t{x}; } -template constexpr static auto operator "" _x246() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint246_t{x}; } -template constexpr static auto operator "" _x247() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint247_t{x}; } -template constexpr static auto operator "" _x248() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint248_t{x}; } -template constexpr static auto operator "" _x249() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint249_t{x}; } +template constexpr static auto operator "" _x240() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint240_t{x}; } +template constexpr static auto operator "" _x241() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint241_t{x}; } +template constexpr static auto operator "" _x242() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint242_t{x}; } +template constexpr static auto operator "" _x243() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint243_t{x}; } +template constexpr static auto operator "" _x244() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint244_t{x}; } +template constexpr static auto operator "" _x245() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint245_t{x}; } +template constexpr static auto operator "" _x246() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint246_t{x}; } +template constexpr static auto operator "" _x247() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint247_t{x}; } +template constexpr static auto operator "" _x248() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint248_t{x}; } +template constexpr static auto operator "" _x249() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint249_t{x}; } // 250--259 -template constexpr static auto operator "" _x250() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint250_t{x}; } -template constexpr static auto operator "" _x251() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint251_t{x}; } -template constexpr static auto operator "" _x252() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint252_t{x}; } -template constexpr static auto operator "" _x253() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint253_t{x}; } -template constexpr static auto operator "" _x254() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint254_t{x}; } -template constexpr static auto operator "" _x255() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint255_t{x}; } -template constexpr static auto operator "" _x256() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; (~((x *= 10) | (x += digits - '0')), ...); return dpf::xints::xint256_t{x}; } +template constexpr static auto operator "" _x250() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint250_t{x}; } +template constexpr static auto operator "" _x251() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint251_t{x}; } +template constexpr static auto operator "" _x252() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint252_t{x}; } +template constexpr static auto operator "" _x253() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint253_t{x}; } +template constexpr static auto operator "" _x254() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint254_t{x}; } +template constexpr static auto operator "" _x255() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint255_t{x}; } +template constexpr static auto operator "" _x256() { utils::constexpr_maybe_throw(!(std::isdigit(digits) && ...), "invalid char"); uint256_t x{0}; ((x = x * 10 + (digits - '0')), ...); return dpf::xints::xint256_t{x}; } } // namespace xints @@ -929,6 +932,7 @@ template struct mod_pow_2> { static constexpr auto mod = mod_pow_2{}; + HEDLEY_NO_THROW std::size_t operator()(xor_wrapper val, std::size_t n) const noexcept { return mod(val.value, n); diff --git a/include/dpf/zip_iterable.hpp b/include/dpf/zip_iterable.hpp index eb034a0..139aa42 100644 --- a/include/dpf/zip_iterable.hpp +++ b/include/dpf/zip_iterable.hpp @@ -100,6 +100,7 @@ struct zip_iterator template HEDLEY_ALWAYS_INLINE +HEDLEY_NO_THROW zip_iterable tuple_as_zip(std::tuple & tup) noexcept { return zip_iterable( diff --git a/include/grotto.hpp b/include/grotto.hpp index 002d98d..959f114 100644 --- a/include/grotto.hpp +++ b/include/grotto.hpp @@ -10,6 +10,8 @@ #include "grotto/fixedpoint.hpp" +#include "grotto/nmod.hpp" + #include "grotto/gadget_hints.hpp" #include "grotto/gadgets.hpp" @@ -24,6 +26,8 @@ #include "grotto/principal_lut.hpp" +#include "grotto/range_lut.hpp" + #include "grotto/window_lut.hpp" #include "grotto/prefix_parity.hpp" diff --git a/include/grotto/constant_lut.hpp b/include/grotto/constant_lut.hpp index 7df8158..eb896e6 100644 --- a/include/grotto/constant_lut.hpp +++ b/include/grotto/constant_lut.hpp @@ -10,6 +10,8 @@ #ifndef LIBDPF_INCLUDE_GROTTO_CONSTANT_LUT_HPP__ #define LIBDPF_INCLUDE_GROTTO_CONSTANT_LUT_HPP__ +#include "hedley/hedley.h" + #include #include #include @@ -57,11 +59,13 @@ struct constant_lut /// `bounds[i + 1]` or one past `numeric_limits::max()` for the last piece. std::vector values; + HEDLEY_NO_THROW std::size_t linear_parts() const noexcept { return values.size(); } /// Pieces after joining the first and last when they carry the same value. /// Those two meet across the signed wrap, which is how the paper counts /// parts for `zero` and `nonzero` (2, not 3). + HEDLEY_NO_THROW std::size_t wrapped_parts() const noexcept { if (values.size() >= 2 && values.front() == values.back()) @@ -69,6 +73,7 @@ struct constant_lut return values.size(); } + HEDLEY_NO_THROW std::int64_t operator()(Raw x) const noexcept { const auto it = std::upper_bound(bounds.begin(), bounds.end(), x); @@ -86,6 +91,7 @@ namespace detail using u128 = unsigned __int128; template +HEDLEY_NO_THROW constexpr u128 magnitude(Raw raw) noexcept { if (raw >= 0) @@ -95,6 +101,7 @@ constexpr u128 magnitude(Raw raw) noexcept return static_cast(-static_cast<__int128>(raw)); } +HEDLEY_NO_THROW constexpr int floor_log2(u128 mag) noexcept { if (mag <= std::uint64_t(-1)) @@ -102,6 +109,7 @@ constexpr int floor_log2(u128 mag) noexcept return 127 - __builtin_clzll(static_cast(mag >> 64)); } +HEDLEY_NO_THROW constexpr bool shift_fits(u128 value, unsigned shift) noexcept { return shift < 128 && value <= (~u128{0} >> shift); @@ -131,6 +139,7 @@ inline constexpr std::uint64_t pow10[] = { 10000000000000000000ull, }; +HEDLEY_NO_THROW inline bool magnitude_ge_pow10(u128 mag, int k, unsigned fractional_bits) noexcept { if (mag == 0 || fractional_bits >= 128) @@ -151,6 +160,7 @@ inline bool magnitude_ge_pow10(u128 mag, int k, unsigned fractional_bits) noexce } /// Smallest positive magnitude whose base-10 log is at least `k`. +HEDLEY_NO_THROW inline u128 first_magnitude_at_least_pow10(int k, unsigned fractional_bits) noexcept { if (fractional_bits >= 128) @@ -253,6 +263,7 @@ struct sign_program static constexpr std::int64_t canonical[] = { Neg, Zero, Pos }; template + HEDLEY_NO_THROW static std::int64_t eval(Raw raw, unsigned) noexcept { if (raw < 0) @@ -310,6 +321,7 @@ template <> struct exact_lut { template + HEDLEY_NO_THROW static std::int64_t eval(Raw raw, unsigned fractional_bits) noexcept { const detail::u128 mag = detail::magnitude(raw); @@ -341,6 +353,7 @@ template <> struct exact_lut { template + HEDLEY_NO_THROW static std::int64_t eval(Raw raw, unsigned fractional_bits) noexcept { const detail::u128 mag = detail::magnitude(raw); @@ -379,6 +392,7 @@ template <> struct exact_lut { template + HEDLEY_NO_THROW static std::int64_t eval(Raw raw, unsigned fractional_bits) noexcept { const detail::u128 mag = detail::magnitude(raw); @@ -416,6 +430,7 @@ template <> struct exact_lut { template + HEDLEY_NO_THROW static std::int64_t eval(Raw raw, unsigned fractional_bits) noexcept { if (raw < 0) @@ -452,6 +467,7 @@ template <> struct exact_lut { template + HEDLEY_NO_THROW static std::int64_t eval(Raw raw, unsigned fractional_bits) noexcept { const detail::u128 mag = detail::magnitude(raw); @@ -559,6 +575,7 @@ enum class threshold_cmp }; template +HEDLEY_NO_THROW std::int64_t evaluate_threshold(Raw raw, Raw bound, threshold_cmp kind) noexcept { switch (kind) diff --git a/include/grotto/dyadic_lut.hpp b/include/grotto/dyadic_lut.hpp new file mode 100644 index 0000000..ffa6906 --- /dev/null +++ b/include/grotto/dyadic_lut.hpp @@ -0,0 +1,433 @@ +/// @file grotto/dyadic_lut.hpp +/// @brief Exact dyadic step functions from the Grotto gadget list. +/// @details Integer results and boolean 1s are raw fixed-point values: +/// an integer n is stored as `n << fractional_bits`. `ilogb(0)` and +/// `ilog10(0)` return `ilog_of_zero`. +/// +/// `make_msb_lut(i)` is bit `i` counting from the most significant +/// bit. It is constant on `2^{i+1}` intervals, so `i` must be less +/// than `msb_bit_limit`. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + +#ifndef LIBDPF_INCLUDE_GROTTO_DYADIC_LUT_HPP__ +#define LIBDPF_INCLUDE_GROTTO_DYADIC_LUT_HPP__ + +#include "hedley/hedley.h" + +#include "grotto/easy_lut.hpp" + +#include +#include +#include +#include +#include + +namespace grotto +{ + +/// Sentinel raw value for `ilogb(0)` and `ilog10(0)`. +inline constexpr std::int64_t ilog_of_zero = + std::numeric_limits::min(); + +/// `make_msb_lut(i)` allows `i` in `[0, msb_bit_limit)`. +/// Bit 0 is two intervals; bit 7 is 256. +inline constexpr unsigned msb_bit_limit = 8; + +namespace detail +{ + +using u128 = unsigned __int128; + +template +HEDLEY_CONST +HEDLEY_NO_THROW +constexpr int raw_width() noexcept +{ + return std::numeric_limits::digits + 1; +} + +inline std::int64_t encode_units(std::int64_t units, unsigned fractional_bits) +{ + if (fractional_bits >= 63) + throw std::invalid_argument("dyadic lut: fractional width does not fit"); + const __int128 scaled = static_cast<__int128>(units) << fractional_bits; + if (scaled > std::numeric_limits::max() + || scaled < std::numeric_limits::min()) + throw std::overflow_error("dyadic lut: encoded value does not fit int64"); + return static_cast(scaled); +} + +inline easy_poly unit_poly(std::int64_t units, unsigned fractional_bits) +{ + return easy_poly{encode_units(units, fractional_bits), 0, 0, 1}; +} + +inline easy_poly indicator_poly(bool on, unsigned fractional_bits) +{ + return unit_poly(on ? 1 : 0, fractional_bits); +} + +template +HEDLEY_CONST +HEDLEY_NO_THROW +constexpr std::uint64_t raw_bits(std::int64_t raw) noexcept +{ + constexpr int width = raw_width(); + const auto masked = static_cast(raw); + if constexpr (width >= 64) + return masked; + else + return masked & ((std::uint64_t{1} << width) - 1); +} + +inline int countl_zero_width(std::uint64_t bits, int width) +{ + if (width <= 0 || width > 64) + throw std::invalid_argument("dyadic lut: width must be 1..64"); + if (width < 64) + bits &= (std::uint64_t{1} << width) - 1; + if (bits == 0) + return width; + return __builtin_clzll(bits) - (64 - width); +} + +inline int countl_one_width(std::uint64_t bits, int width) +{ + const std::uint64_t flipped = width >= 64 ? ~bits : (~bits & ((std::uint64_t{1} << width) - 1)); + return countl_zero_width(flipped, width); +} + +template +int clz_of(std::int64_t raw) +{ + return countl_zero_width(raw_bits(raw), raw_width()); +} + +template +int clrsb_of(std::int64_t raw) +{ + constexpr int width = raw_width(); + const std::uint64_t bits = raw_bits(raw); + const bool neg = ((bits >> (width - 1)) & 1u) != 0; + const int matched = neg ? countl_one_width(bits, width) + : countl_zero_width(bits, width); + return matched - 1; +} + +HEDLEY_CONST +HEDLEY_NO_THROW +constexpr int floor_log2_u128(u128 mag) noexcept +{ + if (mag == 0) + return -1; + int n = 0; + while (mag > 1) + { + mag >>= 1; + ++n; + } + return n; +} + +template +HEDLEY_CONST +HEDLEY_NO_THROW +constexpr u128 magnitude(std::int64_t raw) noexcept +{ + using lim = std::numeric_limits; + if (raw == static_cast(lim::min())) + return u128{1} << (raw_width() - 1); + const auto abs = raw < 0 ? -raw : raw; + return static_cast(abs); +} + +template +HEDLEY_CONST +HEDLEY_NO_THROW +constexpr int ilogb_units(std::int64_t raw, unsigned fractional_bits) noexcept +{ + if (raw == 0) + return 0; + return floor_log2_u128(magnitude(raw)) - static_cast(fractional_bits); +} + +inline u128 pow10_u128(int exponent) +{ + if (exponent < 0 || exponent > 38) + throw std::invalid_argument("dyadic lut: power of ten is out of range"); + u128 p = 1; + for (int i = 0; i < exponent; ++i) + p *= 10; + return p; +} + +/// `mag / 2^k >= 10^e`. +inline bool magnitude_ge_pow10(u128 mag, unsigned fractional_bits, int exponent) +{ + if (mag == 0) + return false; + if (exponent >= 0) + { + const u128 decade = pow10_u128(exponent); + if (fractional_bits > 0 && decade > (~u128{0} >> fractional_bits)) + return false; + return mag >= (decade << fractional_bits); + } + const u128 decade = pow10_u128(-exponent); + const u128 scale = u128{1} << fractional_bits; + u128 threshold = scale / decade; + if (scale % decade != 0) + ++threshold; + return mag >= threshold; +} + +template +int ilog10_units(std::int64_t raw, unsigned fractional_bits) +{ + if (raw == 0) + return 0; + const u128 mag = magnitude(raw); + int lo = -static_cast(fractional_bits) - 2; + int hi = raw_width(); + while (lo < hi) + { + const int mid = lo + (hi - lo + 1) / 2; + if (magnitude_ge_pow10(mag, fractional_bits, mid)) + lo = mid; + else + hi = mid - 1; + } + return lo; +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut steps_from_cuts(std::vector cuts, At && at) +{ + return assemble_easy(std::move(cuts), [&](std::int64_t raw) { + return at(raw); + }); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut predicate_lut(unsigned fractional_bits, bool at_or_below_zero, + bool at_zero, bool above_zero) +{ + std::vector cuts{0, 1}; + return steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { + const bool on = raw < 0 ? at_or_below_zero : (raw == 0 ? at_zero : above_zero); + return indicator_poly(on, fractional_bits); + }); +} + +} // namespace detail + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_positive_lut(unsigned fractional_bits = 0) +{ + return detail::predicate_lut(fractional_bits, false, false, true); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_negative_lut(unsigned fractional_bits = 0) +{ + return detail::predicate_lut(fractional_bits, true, false, false); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_nonnegative_lut(unsigned fractional_bits = 0) +{ + return detail::predicate_lut(fractional_bits, false, true, true); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_nonpositive_lut(unsigned fractional_bits = 0) +{ + return detail::predicate_lut(fractional_bits, true, true, false); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_zero_lut(unsigned fractional_bits = 0) +{ + return detail::predicate_lut(fractional_bits, false, true, false); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_nonzero_lut(unsigned fractional_bits = 0) +{ + return detail::predicate_lut(fractional_bits, true, false, true); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_signum_lut(unsigned fractional_bits = 0) +{ + const std::int64_t one = detail::encode_units(1, fractional_bits); + return detail::steps_from_cuts({0, 1}, [=](std::int64_t raw) { + if (raw < 0) + return detail::easy_poly{-one, 0, 0, 1}; + if (raw == 0) + return detail::kZero; + return detail::easy_poly{one, 0, 0, 1}; + }); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_clz_lut(unsigned fractional_bits = 0) +{ + constexpr int width = detail::raw_width(); + std::vector cuts{0, 1}; + for (int b = 1; b <= width - 2; ++b) + cuts.push_back(std::int64_t{1} << b); + return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { + return detail::unit_poly(detail::clz_of(raw), fractional_bits); + }); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_clrsb_lut(unsigned fractional_bits = 0) +{ + constexpr int width = detail::raw_width(); + std::vector cuts{0, -1, -2}; + for (int exp = 0; exp <= width - 2; ++exp) + cuts.push_back(std::int64_t{1} << exp); + for (int exp = 2; exp <= width - 1; ++exp) + { + if (exp >= 63) + cuts.push_back(static_cast(std::numeric_limits::min())); + else + cuts.push_back(-(std::int64_t{1} << exp)); + } + return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { + return detail::unit_poly(detail::clrsb_of(raw), fractional_bits); + }); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_ilogb_lut(unsigned fractional_bits = 0) +{ + constexpr int width = detail::raw_width(); + std::vector cuts{0, 1}; + for (int b = 1; b <= width - 2; ++b) + cuts.push_back(std::int64_t{1} << b); + for (int exp = 1; exp <= width - 1; ++exp) + { + if (exp >= width - 1) + cuts.push_back(static_cast(std::numeric_limits::min()) + 1); + else + cuts.push_back(-(std::int64_t{1} << exp) + 1); + } + return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { + if (raw == 0) + return detail::easy_poly{ilog_of_zero, 0, 0, 1}; + return detail::unit_poly(detail::ilogb_units(raw, fractional_bits), + fractional_bits); + }); +} + +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_ilog10_lut(unsigned fractional_bits = 0) +{ + using lim = std::numeric_limits; + const std::int64_t maxv = static_cast(lim::max()); + const int lo = detail::ilog10_units(1, fractional_bits); + const int hi_pos = detail::ilog10_units(maxv, fractional_bits); + const int hi_neg = detail::ilog10_units(static_cast(lim::min()), + fractional_bits); + const int hi = hi_pos > hi_neg ? hi_pos : hi_neg; + + std::vector cuts{0, 1}; + std::int64_t prev = 0; + for (int e = lo; e <= hi; ++e) + { + std::int64_t left = 1; + std::int64_t right = maxv; + while (left < right) + { + const std::int64_t mid = left + (right - left) / 2; + if (detail::ilog10_units(mid, fractional_bits) >= e) + right = mid; + else + left = mid + 1; + } + if (left == prev) + continue; + prev = left; + cuts.push_back(left); + if (left < maxv) + { + std::int64_t end = left; + std::int64_t scan_left = left; + std::int64_t scan_right = maxv; + while (scan_left < scan_right) + { + const std::int64_t mid = scan_left + (scan_right - scan_left + 1) / 2; + if (detail::ilog10_units(mid, fractional_bits) == e) + scan_left = mid; + else + scan_right = mid - 1; + } + end = scan_left; + const std::int64_t neg = -end; + if (neg > static_cast(lim::min())) + cuts.push_back(neg); + } + } + return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { + if (raw == 0) + return detail::easy_poly{ilog_of_zero, 0, 0, 1}; + return detail::unit_poly(detail::ilog10_units(raw, fractional_bits), + fractional_bits); + }); +} + +/// Bit `index` counting down from the most significant bit of `Raw`. +/// Index 0 is the sign bit. Larger indexes are refused: the bit is constant +/// on `2^{index+1}` intervals. +template +HEDLEY_WARN_UNUSED_RESULT +easy_lut make_msb_lut(unsigned index, unsigned fractional_bits = 0) +{ + constexpr int width = detail::raw_width(); + if (index >= msb_bit_limit || static_cast(index) >= width) + throw std::invalid_argument( + "msb lut: only the most significant bits are piecewise-cheap"); + const int shift = width - 1 - static_cast(index); + if (shift >= 63) + { + return detail::steps_from_cuts({0}, [=](std::int64_t raw) { + const bool on = ((detail::raw_bits(raw) >> shift) & 1u) != 0; + return detail::indicator_poly(on, fractional_bits); + }); + } + const std::int64_t step = std::int64_t{1} << shift; + std::vector cuts; + const auto minv = static_cast(std::numeric_limits::min()); + for (std::int64_t boundary = minv; ; ) + { + cuts.push_back(boundary); + if (boundary > static_cast(std::numeric_limits::max()) - step) + break; + boundary += step; + } + return detail::steps_from_cuts(std::move(cuts), [=](std::int64_t raw) { + const bool on = ((detail::raw_bits(raw) >> shift) & 1u) != 0; + return detail::indicator_poly(on, fractional_bits); + }); +} + +} // namespace grotto + +#endif // LIBDPF_INCLUDE_GROTTO_DYADIC_LUT_HPP__ diff --git a/include/grotto/easy_lut.hpp b/include/grotto/easy_lut.hpp index d2579d4..ce4cce9 100644 --- a/include/grotto/easy_lut.hpp +++ b/include/grotto/easy_lut.hpp @@ -9,6 +9,8 @@ #ifndef LIBDPF_INCLUDE_GROTTO_EASY_LUT_HPP__ #define LIBDPF_INCLUDE_GROTTO_EASY_LUT_HPP__ +#include "hedley/hedley.h" + #include #include #include @@ -35,6 +37,7 @@ struct easy_lut std::vector c2; std::vector den; + HEDLEY_NO_THROW std::size_t parts() const noexcept { return c0.size(); } std::int64_t operator()(Raw x) const @@ -82,6 +85,7 @@ struct easy_poly std::int64_t den = 1; }; +HEDLEY_NO_THROW inline bool operator==(easy_poly a, easy_poly b) noexcept { return a.c0 == b.c0 && a.c1 == b.c1 && a.c2 == b.c2 && a.den == b.den; diff --git a/include/grotto/fixedpoint.hpp b/include/grotto/fixedpoint.hpp index 0af52f7..4becb00 100644 --- a/include/grotto/fixedpoint.hpp +++ b/include/grotto/fixedpoint.hpp @@ -40,6 +40,7 @@ namespace detail /// @brief Integer value of an already-rounded finite double, as a 256-bit word. /// Values that do not fit saturate to all-ones. +HEDLEY_NO_THROW inline uint256_t uint256_from_rounded_double(double rounded) noexcept { if (!(rounded > 0.0) || !std::isfinite(rounded)) @@ -99,6 +100,7 @@ struct is_static_castable +HEDLEY_NO_THROW Raw saturate_low_bits(uint256_t wide) noexcept { static_assert(Bits > 0 && Bits <= 256); @@ -151,6 +153,7 @@ Raw saturate_low_bits(uint256_t wide) noexcept } template +HEDLEY_NO_THROW inline IntegralType rounded_double_to_integral(double rounded) noexcept { if constexpr (std::is_integral_v @@ -206,6 +209,7 @@ template HEDLEY_ALWAYS_INLINE HEDLEY_CONST +HEDLEY_NO_THROW constexpr IntegralType scale_integer_to_fixed_raw(T integer_value) noexcept { using unsigned_type = dpf::utils::make_unsigned_t; @@ -228,6 +232,7 @@ inline constexpr bool is_signed_rep_v = template HEDLEY_ALWAYS_INLINE HEDLEY_CONST +HEDLEY_NO_THROW constexpr IntegralType raw_neg(IntegralType x) noexcept { using unsigned_type = dpf::utils::make_unsigned_t; @@ -238,6 +243,7 @@ constexpr IntegralType raw_neg(IntegralType x) noexcept template HEDLEY_ALWAYS_INLINE HEDLEY_CONST +HEDLEY_NO_THROW constexpr IntegralType raw_abs(IntegralType x) noexcept { if constexpr (is_signed_rep_v) @@ -252,6 +258,7 @@ constexpr IntegralType raw_abs(IntegralType x) noexcept template HEDLEY_ALWAYS_INLINE HEDLEY_CONST +HEDLEY_NO_THROW constexpr IntegralType raw_fmod(IntegralType a, IntegralType b) noexcept { if (b == IntegralType{}) @@ -265,6 +272,7 @@ constexpr IntegralType raw_fmod(IntegralType a, IntegralType b) noexcept template +HEDLEY_NO_THROW auto constexpr make_fixed_from_integral_type(IntegralType value) noexcept; /// @tparam FractionalBits Number of fractional bits used in the fixed-point @@ -378,6 +386,7 @@ public: ~fixedpoint() = default; /// @brief Cast to `double` + HEDLEY_NO_THROW HEDLEY_ALWAYS_INLINE HEDLEY_PURE explicit constexpr operator double() const noexcept @@ -728,6 +737,7 @@ public: template +HEDLEY_NO_THROW constexpr bool operator&(const Mask & mask, const fixedpoint & x) noexcept { @@ -739,6 +749,7 @@ template +HEDLEY_NO_THROW std::basic_ostream & operator<<(std::basic_ostream & os, const fixedpoint & f) noexcept @@ -762,6 +773,7 @@ operator>>(std::basic_istream & is, template +HEDLEY_NO_THROW auto constexpr make_fixed_from_integral_type(IntegralType value) noexcept { return fixedpoint::from_raw(value); @@ -811,6 +823,7 @@ static auto make_fixed_safe(double d) template +HEDLEY_NO_THROW constexpr auto precision_cast(const fixedpoint & f) noexcept { auto value = f.integral_representation(); @@ -826,6 +839,7 @@ constexpr auto precision_cast(const fixedpoint template +HEDLEY_NO_THROW static constexpr auto precision_of(fixedpoint) noexcept { return FractionalBits; @@ -1466,6 +1480,7 @@ struct countl_zero_symmetric_difference; static constexpr auto clz = dpf::utils::countl_zero_symmetric_difference{}; + HEDLEY_NO_THROW HEDLEY_PURE HEDLEY_ALWAYS_INLINE constexpr std::size_t operator()(const T & lhs, const T & rhs) const noexcept @@ -1499,6 +1514,7 @@ struct mod_pow_2> using fixed_type = grotto::fixedpoint; static constexpr auto mod = mod_pow_2{}; + HEDLEY_NO_THROW std::size_t operator()(fixed_type val, std::size_t n) const noexcept { return mod(val.integral_representation(), n); @@ -1512,6 +1528,7 @@ struct make_from_integral_value using fixed_type = grotto::fixedpoint; using integral_type = typename to_integral_type::integral_type; + HEDLEY_NO_THROW constexpr fixed_type operator()(integral_type val) const noexcept { return grotto::make_fixed_from_integral_type( @@ -1617,6 +1634,7 @@ class numeric_limits> static constexpr int bitwidth = static_cast(dpf::utils::bitlength_of_v); + HEDLEY_NO_THROW static constexpr I raw_lowest() noexcept { if constexpr (signed_rep) @@ -1627,6 +1645,7 @@ class numeric_limits> return I{}; } + HEDLEY_NO_THROW static constexpr I raw_max() noexcept { if constexpr (signed_rep) @@ -1664,8 +1683,11 @@ class numeric_limits> static constexpr bool traps = false; static constexpr bool tinyness_before = false; + HEDLEY_NO_THROW static constexpr T lowest() noexcept { return T::from_raw(raw_lowest()); } + HEDLEY_NO_THROW static constexpr T max() noexcept { return T::from_raw(raw_max()); } + HEDLEY_NO_THROW static constexpr T min() noexcept { if constexpr (FractionalBits == 0) @@ -1674,17 +1696,23 @@ class numeric_limits> } return T::from_raw(I{1}); } + HEDLEY_NO_THROW static constexpr T epsilon() noexcept { return FractionalBits ? T::from_raw(I{1}) : T::from_raw(I{}); } + HEDLEY_NO_THROW static constexpr T round_error() noexcept { return FractionalBits ? T(0.5) : T(0); } + HEDLEY_NO_THROW static constexpr T infinity() noexcept { return max(); } + HEDLEY_NO_THROW static constexpr T quiet_NaN() noexcept { return T::from_raw(I{}); } + HEDLEY_NO_THROW static constexpr T signalling_NaN() noexcept { return T::from_raw(I{}); } + HEDLEY_NO_THROW static constexpr T denorm_min() noexcept { return min(); } }; diff --git a/include/grotto/fixedpoint_mul.hpp b/include/grotto/fixedpoint_mul.hpp index bf7f94c..8c3fa3b 100644 --- a/include/grotto/fixedpoint_mul.hpp +++ b/include/grotto/fixedpoint_mul.hpp @@ -6,6 +6,8 @@ #ifndef LIBDPF_INCLUDE_GROTTO_FIXEDPOINT_MUL_HPP__ #define LIBDPF_INCLUDE_GROTTO_FIXEDPOINT_MUL_HPP__ +#include "hedley/hedley.h" + #ifndef LIBDPF_INCLUDE_DPF_FIXEDPOINT_HPP__ #include "grotto/fixedpoint.hpp" #endif @@ -106,6 +108,7 @@ namespace detail inline constexpr std::size_t fixed_mul_buf_limbs = 12; +HEDLEY_NO_THROW constexpr void mask_to_bits(std::uint64_t * limbs, std::size_t nlimbs, unsigned bits) noexcept { if (bits >= nlimbs * 64u) @@ -129,11 +132,13 @@ constexpr void mask_to_bits(std::uint64_t * limbs, std::size_t nlimbs, unsigned } } +HEDLEY_NO_THROW constexpr bool test_bit(const std::uint64_t * limbs, unsigned bit) noexcept { return ((limbs[bit / 64u] >> (bit % 64u)) & 1u) != 0u; } +HEDLEY_NO_THROW constexpr void fill_ones(std::uint64_t * limbs, unsigned from, unsigned to) noexcept { for (unsigned bit = from; bit < to; ) @@ -149,6 +154,7 @@ constexpr void fill_ones(std::uint64_t * limbs, unsigned from, unsigned to) noex } } +HEDLEY_NO_THROW constexpr void sign_extend_range(std::uint64_t * limbs, unsigned from_bits, unsigned to_bits) noexcept { if (to_bits <= from_bits || from_bits == 0u) @@ -162,6 +168,7 @@ constexpr void sign_extend_range(std::uint64_t * limbs, unsigned from_bits, unsi } template +HEDLEY_NO_THROW constexpr void store_raw_limbs(const T & value, std::uint64_t out[4]) noexcept { out[0] = out[1] = out[2] = out[3] = 0; @@ -192,6 +199,7 @@ constexpr void store_raw_limbs(const T & value, std::uint64_t out[4]) noexcept /// Low `dest_bits` of `value`, sign-extended when `value` is a narrower signed integer. template +HEDLEY_NO_THROW constexpr void reduce_operand(const T & value, unsigned src_bits, bool is_signed, unsigned dest_bits, std::uint64_t * dest, unsigned nlimbs) noexcept { @@ -215,6 +223,7 @@ constexpr void reduce_operand(const T & value, unsigned src_bits, bool is_signed } /// Product modulo `2^(64*nlimbs)`, using exactly `nlimbs` limbs of each operand. +HEDLEY_NO_THROW constexpr void mul_low_limbs(std::uint64_t * out, const std::uint64_t * lhs, const std::uint64_t * rhs, unsigned nlimbs) noexcept { @@ -235,6 +244,7 @@ constexpr void mul_low_limbs(std::uint64_t * out, const std::uint64_t * lhs, } } +HEDLEY_NO_THROW constexpr void shift_left_limbs(std::uint64_t * limbs, std::size_t nlimbs, unsigned shift) noexcept { if (shift == 0u) @@ -261,6 +271,7 @@ constexpr void shift_left_limbs(std::uint64_t * limbs, std::size_t nlimbs, unsig } } +HEDLEY_NO_THROW constexpr void shift_right_limbs(std::uint64_t * limbs, std::size_t nlimbs, unsigned shift) noexcept { if (shift == 0u) @@ -288,6 +299,7 @@ constexpr void shift_right_limbs(std::uint64_t * limbs, std::size_t nlimbs, unsi } template +HEDLEY_NO_THROW constexpr T limbs_to_integral(const std::uint64_t * limbs) noexcept { if constexpr (std::is_same_v) @@ -331,6 +343,7 @@ template +HEDLEY_NO_THROW constexpr auto fixed_mul( fixedpoint lhs, fixedpoint rhs) noexcept diff --git a/include/grotto/hexfloat.hpp b/include/grotto/hexfloat.hpp index dc08278..212d28f 100644 --- a/include/grotto/hexfloat.hpp +++ b/include/grotto/hexfloat.hpp @@ -9,6 +9,8 @@ #ifndef LIBDPF_INCLUDE_GROTTO_HEXFLOAT_HPP__ #define LIBDPF_INCLUDE_GROTTO_HEXFLOAT_HPP__ +#include "hedley/hedley.h" + #include #include #include @@ -109,6 +111,7 @@ constexpr bool is_hexfloat_fixedpoint = false; template constexpr bool is_hexfloat_fixedpoint> = true; +HEDLEY_NO_THROW constexpr void mask_low_bits(std::uint64_t * limbs, unsigned width) noexcept { if (width >= 256u) @@ -132,11 +135,13 @@ constexpr void mask_low_bits(std::uint64_t * limbs, unsigned width) noexcept } } +HEDLEY_NO_THROW constexpr bool bit_is_set(const std::uint64_t * limbs, unsigned bit) noexcept { return ((limbs[bit / 64u] >> (bit % 64u)) & 1u) != 0u; } +HEDLEY_NO_THROW constexpr void negate_low_bits(std::uint64_t * limbs, unsigned width) noexcept { for (unsigned i = 0; i < 4u; ++i) @@ -154,6 +159,7 @@ constexpr void negate_low_bits(std::uint64_t * limbs, unsigned width) noexcept } template +HEDLEY_NO_THROW constexpr void store_integer_bits(const T & value, std::uint64_t out[4]) noexcept { out[0] = out[1] = out[2] = out[3] = 0; @@ -183,6 +189,7 @@ constexpr void store_integer_bits(const T & value, std::uint64_t out[4]) noexcep } template +HEDLEY_NO_THROW constexpr T load_integer_bits(const std::uint64_t * limbs) noexcept { if constexpr (std::is_same_v) @@ -205,6 +212,7 @@ constexpr T load_integer_bits(const std::uint64_t * limbs) noexcept } } +HEDLEY_NO_THROW inline int highest_bit(const std::uint64_t * limbs) noexcept { for (int i = 3; i >= 0; --i) @@ -268,6 +276,7 @@ std::string format_hexfloat(const T & value, int fractional_bits) return std::string(negative ? "-" : "+") + "0x1." + fraction + "p" + expbuf; } +HEDLEY_NO_THROW inline int hex_value(char c) noexcept { if (c >= '0' && c <= '9') return c - '0'; diff --git a/include/grotto/nmod.hpp b/include/grotto/nmod.hpp new file mode 100644 index 0000000..d344130 --- /dev/null +++ b/include/grotto/nmod.hpp @@ -0,0 +1,194 @@ +/// @file grotto/nmod.hpp +/// @brief Reduction modulo an arbitrary public modulus. +/// @details `nmod` multiplies by a rounded reciprocal `1/M` and splits that +/// product with a floor. The quotient is `floor(x/M)`. The residue +/// is `{x/M}` truncated onto `residue_bits` fractional bits, so it +/// lies in `[0, 2^residue_bits)`. A negative product borrows, which +/// keeps the residue non-negative. +/// +/// The reciprocal magnitude is 128 bits, so `1/M` can be carried +/// wider than the input. A power-of-two modulus is the same split +/// with an exact shift (`nmod_pow2`). +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + +#ifndef LIBDPF_INCLUDE_GROTTO_NMOD_HPP__ +#define LIBDPF_INCLUDE_GROTTO_NMOD_HPP__ + +#include +#include + +#include "hedley/hedley.h" + +namespace grotto +{ + +namespace nmod_detail +{ +using u128 = unsigned __int128; +} + +struct nmod_result +{ + /// `floor({x/M} * 2^residue_bits)`, in `[0, 2^residue_bits)`. + std::int64_t residue = 0; + /// `floor(x/M)`. + std::int64_t quotient = 0; +}; + +/// `x_raw / 2^x_bits` modulo `M`, with `1/M ≈ recip_raw / 2^recip_bits`. +/// `recip_raw` is a positive magnitude of at most 128 bits. +/// @throws std::invalid_argument if the reciprocal is zero or a width is illegal. +/// @throws std::overflow_error if the quotient does not fit in `int64_t`. +HEDLEY_WARN_UNUSED_RESULT +inline nmod_result nmod(std::int64_t x_raw, unsigned x_bits, + unsigned __int128 recip_raw, unsigned recip_bits, unsigned residue_bits) +{ + if (recip_raw == 0) + throw std::invalid_argument("nmod: reciprocal must be positive"); + if (residue_bits > 63) + throw std::invalid_argument("nmod: residue must fit in int64"); + if (x_bits > 100000u || recip_bits > 100000u) + throw std::invalid_argument("nmod: fractional width is too large"); + + const bool neg = x_raw < 0; + const auto x_mag = static_cast( + neg ? -static_cast<__int128>(x_raw) : x_raw); + const auto recip_lo = static_cast(recip_raw); + const auto recip_hi = static_cast(recip_raw >> 64); + + unsigned __int128 low = static_cast(x_mag) * recip_lo; + unsigned __int128 high = static_cast(x_mag) * recip_hi; + std::uint64_t limb[4] = {}; + limb[0] = static_cast(low); + const unsigned __int128 mid = (low >> 64) + static_cast(high); + limb[1] = static_cast(mid); + const unsigned __int128 top = (high >> 64) + (mid >> 64); + limb[2] = static_cast(top); + limb[3] = static_cast(top >> 64); + + const auto bit_set_at_or_above = [&](unsigned bit) { + if (bit >= 256) + return false; + const unsigned index = bit / 64u; + const unsigned offset = bit % 64u; + if ((limb[index] >> offset) != 0) + return true; + for (unsigned i = index + 1; i < 4; ++i) + { + if (limb[i] != 0) + return true; + } + return false; + }; + const auto extract = [&](unsigned low_bit, unsigned count) -> std::uint64_t { + if (count == 0 || low_bit >= 256) + return 0; + const unsigned index = low_bit / 64u; + const unsigned offset = low_bit % 64u; + unsigned __int128 chunk = limb[index]; + if (index + 1 < 4) + chunk |= static_cast(limb[index + 1]) << 64; + chunk >>= offset; + if (count == 64) + return static_cast(chunk); + return static_cast(chunk) & ((std::uint64_t{1} << count) - 1); + }; + const auto low_bits_set = [&](unsigned width) { + if (width == 0) + return false; + if (width >= 256) + return limb[0] != 0 || limb[1] != 0 || limb[2] != 0 || limb[3] != 0; + const unsigned index = width / 64u; + const unsigned offset = width % 64u; + for (unsigned i = 0; i < index; ++i) + { + if (limb[i] != 0) + return true; + } + if (offset == 0) + return false; + const std::uint64_t mask = (std::uint64_t{1} << offset) - 1; + return (limb[index] & mask) != 0; + }; + + const unsigned scale = x_bits + recip_bits; + const bool remainder = low_bits_set(scale); + std::uint64_t quotient_mag = 0; + if (scale < 256) + { + if (bit_set_at_or_above(scale + 64)) + throw std::overflow_error("nmod: quotient does not fit int64"); + quotient_mag = extract(scale, 64); + } + + nmod_result out; + if (!neg) + { + if (quotient_mag > static_cast(INT64_MAX)) + throw std::overflow_error("nmod: quotient does not fit int64"); + out.quotient = static_cast(quotient_mag); + } + else if (!remainder) + { + if (quotient_mag > (static_cast(INT64_MAX) + 1)) + throw std::overflow_error("nmod: quotient does not fit int64"); + out.quotient = quotient_mag == (std::uint64_t{1} << 63) + ? INT64_MIN + : -static_cast(quotient_mag); + } + else + { + if (quotient_mag > static_cast(INT64_MAX)) + throw std::overflow_error("nmod: quotient does not fit int64"); + out.quotient = -static_cast(quotient_mag) - 1; + } + + if (residue_bits == 0 || (!neg && !remainder) || (neg && !remainder)) + return out; + + // A 128-bit reciprocal times an int64 magnitude stays under 2^192. + // With a residue of at most 63 bits, a scale at or above 256 puts every + // product bit strictly below the residue window. + if (scale >= 256) + { + if (neg) + out.residue = static_cast((std::uint64_t{1} << residue_bits) - 1); + return out; + } + + std::uint64_t field = 0; + if (scale >= residue_bits) + field = extract(scale - residue_bits, residue_bits); + else + field = extract(0, scale) << (residue_bits - scale); + + if (neg) + { + const std::uint64_t unit = std::uint64_t{1} << residue_bits; + const unsigned discarded = scale >= residue_bits ? scale - residue_bits : 0; + const bool borrow = discarded > 0 && low_bits_set(discarded); + field = borrow ? unit - field - 1 : unit - field; + } + out.residue = static_cast(field); + return out; +} + +/// Modulus `2^{-exp}` by an exact shift. Positive `exp` multiplies by +/// `2^exp` (`exp`'s `2^{-13}` split). Zero splits at the integer (`2^x`). +/// Negative `exp` is a modulus above one. +/// @throws std::overflow_error if `exp` does not fit the reciprocal width. +HEDLEY_WARN_UNUSED_RESULT +inline nmod_result nmod_pow2(std::int64_t x_raw, unsigned x_bits, int exp, + unsigned residue_bits) +{ + if (exp >= 128 || exp < -100000) + throw std::overflow_error("nmod: power-of-two reciprocal does not fit"); + if (exp >= 0) + return nmod(x_raw, x_bits, nmod_detail::u128{1} << static_cast(exp), 0, residue_bits); + return nmod(x_raw, x_bits, 1, static_cast(-exp), residue_bits); +} + +} // namespace grotto + +#endif // LIBDPF_INCLUDE_GROTTO_NMOD_HPP__ diff --git a/include/grotto/offset_horner.hpp b/include/grotto/offset_horner.hpp index 96babbd..a92ce42 100644 --- a/include/grotto/offset_horner.hpp +++ b/include/grotto/offset_horner.hpp @@ -22,6 +22,8 @@ #ifndef LIBDPF_INCLUDE_GROTTO_OFFSET_HORNER_HPP__ #define LIBDPF_INCLUDE_GROTTO_OFFSET_HORNER_HPP__ +#include "hedley/hedley.h" + #include #include #include @@ -43,6 +45,7 @@ namespace grotto inline constexpr std::size_t offset_horner_max_degree = 3; template +HEDLEY_NO_THROW T offset_horner_group_add(T a, T b) noexcept { using u = std::make_unsigned_t; @@ -50,6 +53,7 @@ T offset_horner_group_add(T a, T b) noexcept } template +HEDLEY_NO_THROW T offset_horner_group_sub(T a, T b) noexcept { using u = std::make_unsigned_t; @@ -65,6 +69,7 @@ struct offset_horner_x_plus_r }; template +HEDLEY_NO_THROW offset_horner_x_plus_r offset_horner_at_x_plus_r(T x, T r) noexcept { return offset_horner_x_plus_r{ @@ -86,6 +91,7 @@ inline constexpr uint64_t binom[4][4] = { }; template +HEDLEY_NO_THROW uint64_t lift(T v) noexcept { if constexpr (std::is_signed_v) @@ -95,6 +101,7 @@ uint64_t lift(T v) noexcept } template +HEDLEY_NO_THROW uint64_t horner_at(const std::array & coeff, uint64_t point) noexcept { uint64_t acc = coeff[Degree]; @@ -104,6 +111,7 @@ uint64_t horner_at(const std::array & coeff, uint64_t poin } template +HEDLEY_NO_THROW void fill_payloads(uint64_t base, uint64_t (&payload)[Degree + 1]) noexcept { uint64_t pow = 1; @@ -191,6 +199,7 @@ std::vector segments_of(const Key & key, const std::vector & k } template +HEDLEY_NO_THROW int64_t math_lift(T value) noexcept { if constexpr (std::is_signed_v) @@ -200,6 +209,7 @@ int64_t math_lift(T value) noexcept } template +HEDLEY_NO_THROW T domain_min() noexcept { if constexpr (std::is_signed_v) @@ -211,6 +221,7 @@ T domain_min() noexcept /// Public center-space cut where `center + eta` crosses the domain end. /// Empty when that cut is outside the domain, including `eta == 0`. template +HEDLEY_NO_THROW std::optional carry_threshold(T eta) noexcept { constexpr unsigned bits = dpf::utils::bitlength_of_v; @@ -237,6 +248,7 @@ std::optional carry_threshold(T eta) noexcept /// `center + kappa` is the wrapped representative, as a mathematical integer. template +HEDLEY_NO_THROW int64_t kappa_for(T left, T eta) noexcept { constexpr unsigned bits = dpf::utils::bitlength_of_v; @@ -528,7 +540,7 @@ namespace offset_horner_detail template geneval_offset_horner_result geneval_at( - InputT center0, InputT center1, InputT center, InputT eta, + bool arith, InputT center0, InputT center1, InputT center, InputT eta, const std::vector & knots, const std::vector> & coeff, Rng rng) @@ -553,8 +565,11 @@ geneval_offset_horner_result geneval_at( std::array, Degree + 1> seg1; for (std::size_t m = 0; m <= Degree; ++m) { - const auto opened = dpf::geneval_cmp(center0, center1, - shifted.begin(), shifted.end(), rng, payload[m]); + const auto opened = arith + ? dpf::geneval_cmp(dpf::arith_input, center0, center1, + shifted.begin(), shifted.end(), rng, payload[m]) + : dpf::geneval_cmp(center0, center1, + shifted.begin(), shifted.end(), rng, payload[m]); const uint64_t blind = dpf::uniform_sample(); wrap[m][0] = blind; wrap[m][1] = payload[m] - blind; @@ -574,6 +589,17 @@ geneval_offset_horner_result geneval_at( return out; } +template +geneval_offset_horner_result geneval_at( + InputT center0, InputT center1, InputT center, InputT eta, + const std::vector & knots, + const std::vector> & coeff, + Rng rng) +{ + return geneval_at(false, center0, center1, center, eta, knots, coeff, + std::move(rng)); +} + } // namespace offset_horner_detail /// Geneval-style offset Horner. The center is XOR-shared as in `geneval_point`. @@ -609,10 +635,25 @@ geneval_offset_horner_result geneval_offset_horner( center0, center1, eta, knots, coeff, std::move(rng)); } +/// Additive shares of the center: `center0 + center1` is the comparison point. +template +geneval_offset_horner_result geneval_offset_horner( + dpf::arith_input_t, InputT center0, InputT center1, InputT eta, + const std::vector & knots, + const std::vector> & coeff, + Rng rng) +{ + const InputT center = offset_horner_group_add(center0, center1); + return offset_horner_detail::geneval_at( + true, center0, center1, center, eta, knots, coeff, std::move(rng)); +} + /// Additive shares of the input `x` and the mask `r`. Reconstructs -/// `eta = x - r` and `center = 2r`, XOR-shares that center as `(center, 0)`, -/// and returns both parties' Horner shares of the cubic at `x + r` -/// (the group element `x + r`). +/// `eta = x - r` and passes additive shares of `center = 2r` (`2·r0`, `2·r1`) +/// to arithmetic `geneval_cmp`. Returns both parties' Horner shares of the +/// cubic at `x + r` (the group element `x + r`). template @@ -625,10 +666,11 @@ geneval_offset_horner_result geneval_offset_horner( const InputT x = offset_horner_group_add(x0, x1); const InputT r = offset_horner_group_add(r0, r1); const InputT eta = offset_horner_group_sub(x, r); - const InputT center = offset_horner_group_add(r, r); - InputT zero{}; + const InputT center0 = offset_horner_group_add(r0, r0); + const InputT center1 = offset_horner_group_add(r1, r1); + const InputT center = offset_horner_group_add(center0, center1); return offset_horner_detail::geneval_at( - center, zero, center, eta, knots, coeff, std::move(rng)); + true, center0, center1, center, eta, knots, coeff, std::move(rng)); } template diff --git a/include/grotto/offset_iterable.hpp b/include/grotto/offset_iterable.hpp index b4436c4..aaa1be8 100644 --- a/include/grotto/offset_iterable.hpp +++ b/include/grotto/offset_iterable.hpp @@ -115,6 +115,7 @@ struct offset_iterator_base std::add_const_t>; using size_type = std::size_t; + HEDLEY_NO_THROW constexpr offset_iterator_base(wrapped_iterator iter, size_type offset) noexcept : it{iter}, offset_{offset} { } diff --git a/include/grotto/piecewise.hpp b/include/grotto/piecewise.hpp index 3d61ff6..b3c6d17 100644 --- a/include/grotto/piecewise.hpp +++ b/include/grotto/piecewise.hpp @@ -1,10 +1,12 @@ /// @file grotto/piecewise.hpp +/// @brief Horner evaluation of a cubic and a bound-selected piece. +/// @details `eval_horner` evaluates one polynomial. `piecewise_eval` selects +/// the piece whose upper bound is the first entry of `bounds` +/// strictly greater than `x`. /// @author Ryan Henry -/// @brief -/// @details -/// @copyright Copyright (c) 2019-2023 Ryan Henry and others +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) /// @license Released under a GNU General Public v2.0 (GPLv2) license; -/// see [LICENSE.md](@ref GPLv2) for details. +/// see [LICENSE.md](@ref license) for details. #ifndef LIBDPF_INCLUDE_GROTTO_PIECEWISE_HPP__ #define LIBDPF_INCLUDE_GROTTO_PIECEWISE_HPP__ @@ -13,6 +15,8 @@ #include #include +#include "hedley/hedley.h" + namespace grotto { @@ -25,16 +29,26 @@ template using poly_quadratic = std::array; template using poly_cubic = std::array; template -constexpr auto eval_horner(const poly_constant & f, T x) { return f[0]; } +HEDLEY_PURE +HEDLEY_NO_THROW +constexpr auto eval_horner(const poly_constant & f, T x) noexcept { return f[0]; } template -constexpr auto eval_horner(const poly_linear & f, T x) { return f[1] * x + f[0]; } +HEDLEY_PURE +HEDLEY_NO_THROW +constexpr auto eval_horner(const poly_linear & f, T x) noexcept { return f[1] * x + f[0]; } template -constexpr auto eval_horner(const poly_quadratic & f, T x) { return (f[2] * x + f[1]) * x + f[0]; } +HEDLEY_PURE +HEDLEY_NO_THROW +constexpr auto eval_horner(const poly_quadratic & f, T x) noexcept { return (f[2] * x + f[1]) * x + f[0]; } template -constexpr auto eval_horner(const poly_cubic & f, T x) { return ((f[3] * x + f[2]) * x + f[1]) * x + f[0]; } +HEDLEY_PURE +HEDLEY_NO_THROW +constexpr auto eval_horner(const poly_cubic & f, T x) noexcept { return ((f[3] * x + f[2]) * x + f[1]) * x + f[0]; } template -auto piecewise_eval(const std::array, N1> & polys, const std::array & bounds, T x) +HEDLEY_PURE +HEDLEY_NO_THROW +auto piecewise_eval(const std::array, N1> & polys, const std::array & bounds, T x) noexcept { auto it = std::upper_bound(std::cbegin(bounds), std::cend(bounds), x, [](const T & lhs, const T & rhs){ return lhs < rhs; }); diff --git a/include/grotto/prefix_parity.hpp b/include/grotto/prefix_parity.hpp index 9553f97..0985f5e 100644 --- a/include/grotto/prefix_parity.hpp +++ b/include/grotto/prefix_parity.hpp @@ -17,6 +17,7 @@ #include "dpf/twiddle.hpp" #include "dpf/leaf_node.hpp" #include "dpf/dcf.hpp" +#include "dpf/blocked_dcf.hpp" #include "grotto/offset_iterable.hpp" #include "dpf/path_memoizer.hpp" #include "dpf/utils.hpp" @@ -29,6 +30,7 @@ namespace grotto template +HEDLEY_NO_THROW auto parity_of_substring_prefix(const NodeT & node, InputT x) noexcept { static constexpr auto bits_per_limb = dpf::utils::bitlength_of_v; @@ -211,6 +213,7 @@ struct key_has_cmp().has_cmp())> : std::true_type {}; template +HEDLEY_NO_THROW uint64_t cmp_addend_raw(const KeyT & key) noexcept { if constexpr (dpf::is_party_key_v) @@ -263,7 +266,12 @@ static auto signed_prefix_parities(const DpfKey & dpf, constexpr std::size_t depth = key_type::depth; const auto & ch = dpf.cmp(); const std::size_t nbits = static_cast(ch.nbits); - if (nbits > depth) + if constexpr (key_type::cmp_block > 0) + { + if (key_type::cmp_h > depth) + throw std::invalid_argument("signed_prefix_parities: comparison is deeper than the key"); + } + else if (nbits > depth) throw std::invalid_argument("signed_prefix_parities: comparison is deeper than the key"); const uint64_t mask = ch.mask; @@ -284,6 +292,11 @@ static auto signed_prefix_parities(const DpfKey & dpf, prefixes[which] = addend & mask; continue; } + if constexpr (key_type::cmp_block > 0) + { + prefixes[which] = dpf::detail::blocked::eval_share(dpf, tx, path); + continue; + } const std::size_t resume = dpf::detail::path_resume_for_level( path, dpf, tx, nbits); @@ -363,7 +376,12 @@ static void signed_prefix_parities_into(const DpfKey & dpf, constexpr std::size_t depth = key_type::depth; const auto & ch = dpf.cmp(); const std::size_t nbits = static_cast(ch.nbits); - if (nbits > depth) + if constexpr (key_type::cmp_block > 0) + { + if (key_type::cmp_h > depth) + throw std::invalid_argument("signed_prefix_parities: comparison is deeper than the key"); + } + else if (nbits > depth) throw std::invalid_argument("signed_prefix_parities: comparison is deeper than the key"); const uint64_t mask = ch.mask; @@ -382,6 +400,11 @@ static void signed_prefix_parities_into(const DpfKey & dpf, out[which] = addend & mask; continue; } + if constexpr (key_type::cmp_block > 0) + { + out[which] = dpf::detail::blocked::eval_share(dpf, tx, path); + continue; + } const std::size_t resume = dpf::detail::path_resume_for_level( path, dpf, tx, nbits); diff --git a/include/grotto/principal_lut.hpp b/include/grotto/principal_lut.hpp index a21ef88..8998a7b 100644 --- a/include/grotto/principal_lut.hpp +++ b/include/grotto/principal_lut.hpp @@ -17,6 +17,8 @@ #ifndef LIBDPF_INCLUDE_GROTTO_PRINCIPAL_LUT_HPP__ #define LIBDPF_INCLUDE_GROTTO_PRINCIPAL_LUT_HPP__ +#include "hedley/hedley.h" + #include #include @@ -44,6 +46,7 @@ enum class principal : unsigned inline constexpr unsigned principal_precisions[] = {8u, 12u, 16u, 20u, 24u, 28u, 32u}; +HEDLEY_NO_THROW inline constexpr bool principal_precision(unsigned fractional_bits) noexcept { for (unsigned k : principal_precisions) @@ -150,6 +153,7 @@ inline w256 w_mul_u64(w256 value, std::uint64_t factor) return out; } +HEDLEY_NO_THROW inline bool w_negative(w256 value) noexcept { return value.hi < 0; @@ -311,7 +315,10 @@ inline w256 w_mul_i64(w256 value, std::int64_t factor) { if (factor >= 0) return w_mul_u64(value, static_cast(factor)); - return w_neg(w_mul_u64(value, static_cast(-factor))); + // `-factor` is undefined at INT64_MIN. The magnitude is the unsigned wrap. + const auto mag = static_cast(0) + - static_cast(factor); + return w_neg(w_mul_u64(value, mag)); } inline std::int64_t horner(const cubic_bits & piece, unsigned q, std::int64_t raw, unsigned fractional_bits) diff --git a/include/grotto/range_lut.hpp b/include/grotto/range_lut.hpp new file mode 100644 index 0000000..36e7522 --- /dev/null +++ b/include/grotto/range_lut.hpp @@ -0,0 +1,676 @@ +/// @file grotto/range_lut.hpp +/// @brief Full-domain maps built from the principal-domain cubics. +/// @details Each reduced map is one of the elementary range reductions, and +/// the polynomial it evaluates is the matching principal table: +/// `ln` / `lg` / `log10` share the mantissa logarithm; +/// `exp` / `exp2` / `exp10` share the `2^{-13}` exponential; +/// `sin` / `cos` share the quarter-turn sine; +/// `tan` / `cot` share `tanf` and `tang`; +/// `sec` / `csc` share `sec` and `gsec`; +/// `sinh` / `cosh` / `tanh` / `sech` share the hyperbolic addition; +/// `coth` and `csch` use their principal small-argument tables; +/// `sqrt` / `inv` / `rsqrt` / `invsq` are dyadic lifts of `[1/2, 1]`. +/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors) +/// @license Released under a GNU General Public v2.0 (GPLv2) license. + +#ifndef LIBDPF_INCLUDE_GROTTO_RANGE_LUT_HPP__ +#define LIBDPF_INCLUDE_GROTTO_RANGE_LUT_HPP__ + +#include +#include + +#include "hedley/hedley.h" + +#include "grotto/principal_lut.hpp" + +namespace grotto +{ + +enum class reduced : unsigned +{ + ln = 0, + lg, + log10, + exp, + exp2, + exp10, + sin, + cos, + tan, + cot, + sec, + csc, + sinh, + cosh, + tanh, + coth, + sech, + csch, + sqrt, + inv, + rsqrt, + invsq, +}; + +HEDLEY_WARN_UNUSED_RESULT +inline std::int64_t eval_reduced(reduced which, unsigned fractional_bits, std::int64_t raw); + +namespace range_detail +{ + +using u128 = unsigned __int128; + +constexpr u128 words(std::uint64_t hi, std::uint64_t lo) +{ + return (u128{hi} << 64) | lo; +} + +/// `value * 2^64`, rounded half away from zero. Values above `2^64` keep the +/// high limb so the constant is not truncated. +inline constexpr u128 ln2_64 = words(0, 12786308645202655660ULL); +inline constexpr u128 inv_ln2_64 = words(1, 8166282121979093367ULL); +inline constexpr u128 log10_2_64 = words(0, 5553023288523357132ULL); +inline constexpr u128 ln10_64 = words(2, 5581709770980765788ULL); +inline constexpr u128 inv_ln10_64 = words(0, 8011319160293570763ULL); +inline constexpr u128 sqrt2_64 = words(1, 7640891576956012809ULL); +inline constexpr u128 rsqrt2_64 = words(0, 13043817825332782212ULL); +inline constexpr u128 two_over_pi_64 = words(0, 11743562013128004906ULL); +inline constexpr u128 four_over_pi_64 = words(1, 5040379952546458196ULL); +inline constexpr u128 pi_over_4_64 = words(0, 14488038916154245685ULL); + +/// `exp(2^{i-13}) * 2^64`. +inline constexpr u128 exp_chunk_64[13] = { + words(1, 2251937258231296ULL), + words(1, 4504149427926357ULL), + words(1, 9009398635954180ULL), + words(1, 18023197466514910ULL), + words(1, 36064004308734226ULL), + words(1, 72198514957318099ULL), + words(1, 144679606912572172ULL), + words(1, 290493950045950331ULL), + words(1, 585562514163419534ULL), + words(1, 1189712777830127574ULL), + words(1, 2456155437534072733ULL), + words(1, 5239344172067481206ULL), + words(1, 11966795255776918679ULL), +}; + +inline std::int64_t round_mag(u128 mag, unsigned shift, bool neg) +{ + if (shift >= 128) + return 0; + if (shift > 0) + { + mag += u128{1} << (shift - 1); + mag >>= shift; + } + if (mag > static_cast(INT64_MAX)) + throw std::overflow_error("range lut: value does not fit int64"); + const auto out = static_cast(mag); + return neg ? -out : out; +} + +inline std::int64_t round_i128(__int128 value, unsigned shift) +{ + const bool neg = value < 0; + const auto mag = static_cast(neg ? -value : value); + return round_mag(mag, shift, neg); +} + +inline std::int64_t scale_unit(u128 mag64, unsigned fractional_bits) +{ + return round_mag(mag64, 64u - fractional_bits, false); +} + +inline std::int64_t mul_raw(std::int64_t lhs, std::int64_t rhs, unsigned fractional_bits) +{ + return round_i128(static_cast<__int128>(lhs) * rhs, fractional_bits); +} + +inline std::int64_t div_raw(std::int64_t num, std::int64_t den, unsigned fractional_bits) +{ + if (den == 0) + throw std::domain_error("range lut: division by zero"); + const bool neg = (num < 0) != (den < 0); + auto n = static_cast(num < 0 ? -static_cast<__int128>(num) : num); + auto d = static_cast(den < 0 ? -static_cast<__int128>(den) : den); + n <<= fractional_bits; + const u128 quot = (n + d / 2) / d; + return round_mag(quot, 0, neg); +} + +inline std::int64_t shift_pow2(std::int64_t value, int places) +{ + if (places == 0 || value == 0) + return value; + if (places > 0) + { + if (places >= 62) + throw std::overflow_error("range lut: exponent overflow"); + const __int128 wide = static_cast<__int128>(value) << places; + if (wide > INT64_MAX || wide < INT64_MIN) + throw std::overflow_error("range lut: exponent overflow"); + return static_cast(wide); + } + return round_i128(value, static_cast(-places)); +} + +HEDLEY_CONST +HEDLEY_NO_THROW +constexpr std::int64_t one_raw(unsigned fractional_bits) noexcept +{ + return std::int64_t{1} << fractional_bits; +} + +HEDLEY_CONST +HEDLEY_NO_THROW +constexpr u128 magnitude_of(std::int64_t raw) noexcept +{ + if (raw >= 0) + return static_cast(raw); + return static_cast(-static_cast<__int128>(raw)); +} + +inline std::int64_t abs_raw(std::int64_t raw) +{ + const u128 mag = magnitude_of(raw); + if (mag > static_cast(INT64_MAX)) + throw std::overflow_error("range lut: magnitude does not fit int64"); + return static_cast(mag); +} + +struct dyadic +{ + std::int64_t mantissa_raw; + int power; +}; + +inline dyadic split_positive(std::int64_t raw, unsigned fractional_bits) +{ + if (raw <= 0) + throw std::domain_error("range lut: reduction requires a positive input"); + const auto mag = static_cast(raw); + const int floor_log = 63 - __builtin_clzll(mag); + const int shift = static_cast(fractional_bits) - floor_log - 1; + std::int64_t mantissa = shift >= 0 + ? raw << shift + : round_i128(raw, static_cast(-shift)); + int power = floor_log + 1 - static_cast(fractional_bits); + const std::int64_t one = one_raw(fractional_bits); + const std::int64_t half = one >> 1; + if (mantissa >= one) + { + mantissa >>= 1; + ++power; + } + if (mantissa < half) + mantissa = half; + return dyadic{mantissa, power}; +} + +inline std::int64_t ln2_raw(unsigned fractional_bits) +{ + return scale_unit(ln2_64, fractional_bits); +} + +inline std::int64_t eval_ln_positive(unsigned fractional_bits, std::int64_t raw) +{ + const dyadic part = split_positive(raw, fractional_bits); + const std::int64_t ln_m = eval_principal(principal::ln, fractional_bits, part.mantissa_raw); + return ln_m + static_cast(part.power) * ln2_raw(fractional_bits); +} + +inline std::int64_t eval_exp_at_scale(unsigned fractional_bits, std::int64_t raw) +{ + if (fractional_bits < 13) + { + const int lift = static_cast(16u - fractional_bits); + const __int128 lifted_arg = static_cast<__int128>(raw) << lift; + if (lifted_arg > INT64_MAX || lifted_arg < INT64_MIN) + throw std::overflow_error("range lut: exponent overflow"); + const std::int64_t lifted = eval_exp_at_scale(16, static_cast(lifted_arg)); + return round_i128(lifted, static_cast(lift)); + } + const std::int64_t ln2 = ln2_raw(fractional_bits); + if (ln2 <= 0) + throw std::logic_error("range lut: ln 2 constant"); + std::int64_t n_bin = raw / ln2; + std::int64_t remainder = raw - n_bin * ln2; + if (remainder < 0) + { + remainder += ln2; + --n_bin; + } + while (remainder >= ln2) + { + remainder -= ln2; + ++n_bin; + } + + const std::int64_t step = std::int64_t{1} << (fractional_bits - 13); + const std::int64_t chunks = remainder / step; + const std::int64_t tiny = remainder - chunks * step; + std::int64_t table_raw = tiny << 13; + const std::int64_t one = one_raw(fractional_bits); + if (table_raw > one) + table_raw = one; + std::int64_t exp_s = eval_principal(principal::exp, fractional_bits, table_raw); + for (unsigned bit = 0; bit < 13; ++bit) + { + if ((static_cast(chunks) & (1ull << bit)) == 0) + continue; + // `chunk` is `exp(2^{i-13}) * 2^64`, so the product's high limb is the raw product. + const u128 prod = static_cast(exp_s) * exp_chunk_64[bit]; + exp_s = round_mag(prod, 64, false); + } + return shift_pow2(exp_s, static_cast(n_bin)); +} + +inline std::int64_t fractional_raw(std::int64_t raw, unsigned fractional_bits, std::int64_t & whole) +{ + const std::int64_t one = one_raw(fractional_bits); + std::int64_t q = raw / one; + std::int64_t f = raw - q * one; + if (f < 0) + { + f += one; + --q; + } + whole = q; + return f; +} + +inline std::int64_t pow10_raw(int exponent, unsigned fractional_bits) +{ + const std::int64_t one = one_raw(fractional_bits); + if (exponent == 0) + return one; + if (exponent < 0) + return div_raw(one, pow10_raw(-exponent, fractional_bits), fractional_bits); + u128 acc = static_cast(one); + for (int i = 0; i < exponent; ++i) + { + if (acc > static_cast(INT64_MAX) / 10) + throw std::overflow_error("range lut: exponent overflow"); + acc *= 10; + } + return static_cast(acc); +} + +struct angle +{ + unsigned index; + std::int64_t frac_raw; +}; + +/// `{ |x| * multiplier }` at this precision, with the integer part reduced +/// only as far as the low bits the quadrant logic reads. +inline angle reduce_positive(unsigned fractional_bits, std::int64_t raw, u128 multiplier_64) +{ + const u128 scaled = magnitude_of(raw) * multiplier_64; + const u128 rounded = (scaled + (u128{1} << 63)) >> 64; + const u128 one = u128{1} << fractional_bits; + return angle{ + static_cast(rounded >> fractional_bits), + static_cast(rounded & (one - 1)), + }; +} + +inline std::int64_t principal_sin_fraction(unsigned fractional_bits, std::int64_t fraction_raw, bool complement) +{ + const std::int64_t one = one_raw(fractional_bits); + std::int64_t argument = complement ? one - fraction_raw : fraction_raw; + if (argument < 0) + argument = 0; + if (argument > one) + argument = one; + return eval_principal(principal::sin, fractional_bits, argument); +} + +inline std::int64_t sin_from_angle(unsigned fractional_bits, const angle & turned, int sign) +{ + const unsigned which = turned.index & 3u; + const bool complement = which == 1 || which == 3; + const int quadrant_sign = (which == 2 || which == 3) ? -1 : 1; + const std::int64_t magnitude = principal_sin_fraction( + fractional_bits, turned.frac_raw, complement); + return magnitude * quadrant_sign * sign; +} + +inline std::int64_t cos_from_angle(unsigned fractional_bits, const angle & turned) +{ + angle shifted = turned; + shifted.index += 1; + return sin_from_angle(fractional_bits, shifted, 1); +} + +inline std::int64_t pi_over_4_raw(unsigned fractional_bits) +{ + return scale_unit(pi_over_4_64, fractional_bits); +} + +inline std::int64_t tan_positive(unsigned fractional_bits, std::int64_t magnitude, int quarter_shift) +{ + const angle turned = reduce_positive(fractional_bits, magnitude, four_over_pi_64); + const unsigned q = (turned.index + static_cast(quarter_shift)) & 3u; + const std::int64_t one = one_raw(fractional_bits); + std::int64_t t = (q == 0 || q == 2) ? turned.frac_raw : one - turned.frac_raw; + if (t < 0) + t = 0; + if (t > one) + t = one; + const std::int64_t z = mul_raw(t, pi_over_4_raw(fractional_bits), fractional_bits); + if (q == 0 || q == 3) + { + const std::int64_t tanf = eval_principal(principal::tanf, fractional_bits, t); + const std::int64_t y = mul_raw(z, tanf, fractional_bits); + return q == 3 ? -y : y; + } + if (z == 0) + throw std::domain_error("range lut: tan pole"); + const std::int64_t tang = eval_principal(principal::tang, fractional_bits, t); + const std::int64_t y = div_raw(one, z, fractional_bits) + tang; + return q == 2 ? -y : y; +} + +inline std::int64_t sec_positive(unsigned fractional_bits, std::int64_t magnitude, int octant_shift) +{ + const angle turned = reduce_positive(fractional_bits, magnitude, four_over_pi_64); + const unsigned q8 = (turned.index + static_cast(octant_shift)) & 7u; + const unsigned q = q8 & 3u; + const int sigma = (q8 & 4u) == 0 ? 1 : -1; + const std::int64_t one = one_raw(fractional_bits); + std::int64_t t = (q == 0 || q == 2) ? turned.frac_raw : one - turned.frac_raw; + if (t < 0) + t = 0; + if (t > one) + t = one; + if (q == 0 || q == 3) + { + const std::int64_t sec = eval_principal(principal::sec, fractional_bits, t); + const int sign = (q == 3 ? -1 : 1) * sigma; + return sec * sign; + } + const std::int64_t z = mul_raw(t, pi_over_4_raw(fractional_bits), fractional_bits); + if (z == 0) + throw std::domain_error("range lut: sec pole"); + const std::int64_t gsec = eval_principal(principal::gsec, fractional_bits, t); + std::int64_t y = div_raw(one, z, fractional_bits) + gsec; + if (q == 2) + y = -y; + return y * sigma; +} + +inline void quotient_2_13(unsigned fractional_bits, std::int64_t magnitude, + std::int64_t & quotient, std::int64_t & remainder) +{ + if (fractional_bits >= 13) + { + const unsigned shift = fractional_bits - 13; + quotient = magnitude >> shift; + const std::int64_t mask = shift >= 63 ? INT64_MAX : (std::int64_t{1} << shift) - 1; + remainder = shift == 0 ? 0 : magnitude & mask; + return; + } + const int lift = static_cast(13u - fractional_bits); + const __int128 wide = static_cast<__int128>(magnitude) << lift; + if (wide > INT64_MAX) + throw std::overflow_error("range lut: exponent overflow"); + quotient = static_cast(wide); + remainder = 0; +} + +inline std::int64_t exp_of_quotient(unsigned fractional_bits, std::int64_t quotient, std::int64_t magnitude) +{ + if (quotient == 0) + return one_raw(fractional_bits); + __int128 argument; + if (fractional_bits >= 13) + argument = static_cast<__int128>(quotient) << (fractional_bits - 13); + else + argument = magnitude; + if (argument > INT64_MAX) + throw std::overflow_error("range lut: exponent overflow"); + return eval_exp_at_scale(fractional_bits, static_cast(argument)); +} + +struct hyp +{ + std::int64_t sinh_raw; + std::int64_t cosh_raw; +}; + +inline hyp sinh_cosh(unsigned fractional_bits, std::int64_t raw) +{ + const bool neg = raw < 0; + const auto mag_wide = magnitude_of(raw); + if (mag_wide > static_cast(INT64_MAX)) + throw std::overflow_error("range lut: exponent overflow"); + const std::int64_t mag = static_cast(mag_wide); + std::int64_t quotient = 0; + std::int64_t remainder = 0; + quotient_2_13(fractional_bits, mag, quotient, remainder); + + const std::int64_t one = one_raw(fractional_bits); + std::int64_t table = 0; + if (fractional_bits >= 13 && remainder != 0) + { + const __int128 lifted = static_cast<__int128>(remainder) << 13; + table = lifted > one ? one : static_cast(lifted); + } + const std::int64_t sr = eval_principal(principal::sinh, fractional_bits, table); + const std::int64_t cr = eval_principal(principal::cosh, fractional_bits, table); + std::int64_t sh = sr; + std::int64_t ch = cr; + if (quotient != 0) + { + const std::int64_t grown = exp_of_quotient(fractional_bits, quotient, mag); + std::int64_t inv = 0; + if (grown != 0) + inv = div_raw(one, grown, fractional_bits); + const std::int64_t sq = round_i128(static_cast<__int128>(grown) - inv, 1); + const std::int64_t cq = round_i128(static_cast<__int128>(grown) + inv, 1); + const __int128 sinh_sum = static_cast<__int128>(sq) * cr + static_cast<__int128>(cq) * sr; + const __int128 cosh_sum = static_cast<__int128>(cq) * cr + static_cast<__int128>(sq) * sr; + sh = round_i128(sinh_sum, fractional_bits); + ch = round_i128(cosh_sum, fractional_bits); + } + if (neg) + sh = -sh; + return hyp{sh, ch}; +} + +/// `ln(2^{k+1} ± 1) / 2`, the saturation threshold used by `tanh` and `coth`. +inline std::int64_t beta_raw(unsigned fractional_bits, bool plus) +{ + u128 ln = u128{fractional_bits + 1} * ln2_64; + const u128 eps = u128{1} << (63u - fractional_bits); + if (plus) + ln += eps; + else + ln -= eps; + return round_mag(ln, 65u - fractional_bits, false); +} + +inline int half_pow_of(int power) +{ + return (power & 1) != 0 ? (power - 1) / 2 : power / 2; +} + +} // namespace range_detail + +HEDLEY_WARN_UNUSED_RESULT +inline std::int64_t eval_reduced(reduced which, unsigned fractional_bits, std::int64_t raw) +{ + using namespace range_detail; + if (!principal_precision(fractional_bits)) + throw std::invalid_argument("range lut: precision must be 8, 12, ..., 32"); + const std::int64_t one = one_raw(fractional_bits); + switch (which) + { + case reduced::ln: + return eval_ln_positive(fractional_bits, raw); + case reduced::lg: + { + const dyadic part = split_positive(raw, fractional_bits); + const std::int64_t ln_m = eval_principal(principal::ln, fractional_bits, part.mantissa_raw); + const std::int64_t lg_m = mul_raw( + ln_m, scale_unit(inv_ln2_64, fractional_bits), fractional_bits); + return lg_m + (static_cast(part.power) << fractional_bits); + } + case reduced::log10: + { + const dyadic part = split_positive(raw, fractional_bits); + const std::int64_t ln_m = eval_principal(principal::ln, fractional_bits, part.mantissa_raw); + const std::int64_t mantissa = mul_raw( + ln_m, scale_unit(inv_ln10_64, fractional_bits), fractional_bits); + const std::int64_t lift = static_cast(part.power) + * scale_unit(log10_2_64, fractional_bits); + return mantissa + lift; + } + case reduced::exp: + return eval_exp_at_scale(fractional_bits, raw); + case reduced::exp2: + { + std::int64_t whole = 0; + const std::int64_t frac = fractional_raw(raw, fractional_bits, whole); + const std::int64_t natural = mul_raw(frac, ln2_raw(fractional_bits), fractional_bits); + return shift_pow2(eval_exp_at_scale(fractional_bits, natural), static_cast(whole)); + } + case reduced::exp10: + { + std::int64_t whole = 0; + const std::int64_t frac = fractional_raw(raw, fractional_bits, whole); + const std::int64_t natural = mul_raw( + frac, scale_unit(ln10_64, fractional_bits), fractional_bits); + if (whole > 18 || whole < -18) + throw std::overflow_error("range lut: exponent overflow"); + return mul_raw( + eval_exp_at_scale(fractional_bits, natural), + pow10_raw(static_cast(whole), fractional_bits), + fractional_bits); + } + case reduced::sin: + return sin_from_angle( + fractional_bits, + reduce_positive(fractional_bits, raw, two_over_pi_64), + raw < 0 ? -1 : 1); + case reduced::cos: + return cos_from_angle( + fractional_bits, reduce_positive(fractional_bits, raw, two_over_pi_64)); + case reduced::tan: + { + const std::int64_t y = tan_positive(fractional_bits, abs_raw(raw), 0); + return raw < 0 ? -y : y; + } + case reduced::cot: + { + if (raw == 0) + throw std::domain_error("range lut: cot pole"); + const std::int64_t y = -tan_positive(fractional_bits, abs_raw(raw), 2); + return raw < 0 ? -y : y; + } + case reduced::sec: + return sec_positive(fractional_bits, abs_raw(raw), 0); + case reduced::csc: + { + if (raw == 0) + throw std::domain_error("range lut: csc pole"); + const std::int64_t y = sec_positive(fractional_bits, abs_raw(raw), -2); + return raw < 0 ? -y : y; + } + case reduced::sinh: + return sinh_cosh(fractional_bits, raw).sinh_raw; + case reduced::cosh: + return sinh_cosh(fractional_bits, raw).cosh_raw; + case reduced::tanh: + { + if (raw == 0) + return 0; + const std::int64_t limit = beta_raw(fractional_bits, false); + const std::int64_t mag = abs_raw(raw); + if (mag >= limit) + return raw < 0 ? -one : one; + const hyp pair = sinh_cosh(fractional_bits, raw); + return div_raw(pair.sinh_raw, pair.cosh_raw, fractional_bits); + } + case reduced::coth: + { + if (raw == 0) + throw std::domain_error("range lut: coth pole"); + const std::int64_t limit = beta_raw(fractional_bits, true); + const std::int64_t mag = abs_raw(raw); + std::int64_t y; + if (mag >= limit) + y = one; + else + { + const std::int64_t t = div_raw(mag, limit, fractional_bits); + const std::int64_t argument = t > one ? one : t; + const std::int64_t removed = eval_principal(principal::coth, fractional_bits, argument); + y = removed + div_raw(one, mag, fractional_bits); + } + return raw < 0 ? -y : y; + } + case reduced::sech: + { + const std::int64_t ch = sinh_cosh(fractional_bits, raw).cosh_raw; + return div_raw(one, ch, fractional_bits); + } + case reduced::csch: + { + if (raw == 0) + throw std::domain_error("range lut: csch pole"); + const bool neg = raw < 0; + const std::int64_t mag = abs_raw(raw); + std::int64_t y; + if (mag <= one) + { + const std::int64_t removed = eval_principal(principal::csch, fractional_bits, mag); + y = removed + div_raw(one, mag, fractional_bits); + } + else + { + y = div_raw(one, sinh_cosh(fractional_bits, mag).sinh_raw, fractional_bits); + } + return neg ? -y : y; + } + case reduced::sqrt: + { + if (raw == 0) + return 0; + const dyadic part = split_positive(raw, fractional_bits); + std::int64_t root = eval_principal(principal::sqrt, fractional_bits, part.mantissa_raw); + if ((part.power & 1) != 0) + root = mul_raw(root, scale_unit(sqrt2_64, fractional_bits), fractional_bits); + return shift_pow2(root, half_pow_of(part.power)); + } + case reduced::inv: + { + const dyadic part = split_positive(raw, fractional_bits); + const std::int64_t reciprocal = eval_principal( + principal::inv, fractional_bits, part.mantissa_raw); + return shift_pow2(reciprocal, -part.power); + } + case reduced::rsqrt: + { + const dyadic part = split_positive(raw, fractional_bits); + std::int64_t root = eval_principal(principal::rsqrt, fractional_bits, part.mantissa_raw); + if ((part.power & 1) != 0) + root = mul_raw(root, scale_unit(rsqrt2_64, fractional_bits), fractional_bits); + return shift_pow2(root, -half_pow_of(part.power)); + } + case reduced::invsq: + { + const dyadic part = split_positive(raw, fractional_bits); + const std::int64_t square = eval_principal( + principal::invsq, fractional_bits, part.mantissa_raw); + return shift_pow2(square, -2 * part.power); + } + } + throw std::invalid_argument("range lut: unknown map"); +} + +} // namespace grotto + +#endif // LIBDPF_INCLUDE_GROTTO_RANGE_LUT_HPP__ diff --git a/test/CMakeLists.txt b/test/CMakeLists.txt index 573cab9..2fe7923 100644 --- a/test/CMakeLists.txt +++ b/test/CMakeLists.txt @@ -40,6 +40,10 @@ add_executable(bit_array_test tests/bit_array_test.cpp) add_executable(parallel_bit_iterable_test tests/parallel_bit_iterable_test.cpp) add_executable(setbit_index_iterable_test tests/setbit_index_iterable_test.cpp) add_executable(incremental_test tests/incremental_test.cpp) +add_executable(path_recipe_test tests/path_recipe_test.cpp) +add_executable(blocked_dcf_test tests/blocked_dcf_test.cpp) +target_compile_definitions(blocked_dcf_test PRIVATE LIBDPF_HAS_NLOHMANN_JSON) +target_include_directories(blocked_dcf_test PRIVATE ../thirdparty/json/include) add_executable(geneval_test tests/geneval_test.cpp) add_executable(stress_scenarios_test tests/stress_scenarios_test.cpp) add_executable(incremental_json_test tests/incremental_json_test.cpp) @@ -58,14 +62,19 @@ target_compile_options(random_test PRIVATE -ftrapv) find_package(Threads REQUIRED) target_link_libraries(random_test Threads::Threads) add_executable(prg_lowmc_test tests/prg_lowmc_test.cpp) +add_executable(prg_chacha_test tests/prg_chacha_test.cpp) add_executable(secret_share_test tests/secret_share_test.cpp) add_executable(beaver_test tests/beaver_test.cpp) add_executable(constant_lut_test tests/constant_lut_test.cpp) add_executable(signed_prefix_test tests/signed_prefix_test.cpp) add_executable(easy_lut_test tests/easy_lut_test.cpp) +add_executable(dyadic_lut_test tests/dyadic_lut_test.cpp) +add_executable(nmod_test tests/nmod_test.cpp) add_executable(principal_lut_test tests/principal_lut_test.cpp) +add_executable(range_lut_test tests/range_lut_test.cpp) add_executable(window_lut_test tests/window_lut_test.cpp) add_executable(offset_horner_test tests/offset_horner_test.cpp) +add_executable(corner_gaps_test tests/corner_gaps_test.cpp) include(GoogleTest) gtest_discover_tests(dpf_key_test) @@ -86,6 +95,8 @@ gtest_discover_tests(bit_array_test) gtest_discover_tests(parallel_bit_iterable_test) gtest_discover_tests(setbit_index_iterable_test) gtest_discover_tests(incremental_test) +gtest_discover_tests(path_recipe_test) +gtest_discover_tests(blocked_dcf_test) gtest_discover_tests(geneval_test) gtest_discover_tests(stress_scenarios_test) gtest_discover_tests(incremental_json_test) @@ -96,11 +107,18 @@ gtest_discover_tests(lane_blast_test) gtest_discover_tests(context_blast_test) gtest_discover_tests(random_test) gtest_discover_tests(prg_lowmc_test) +gtest_discover_tests(prg_chacha_test) gtest_discover_tests(secret_share_test) gtest_discover_tests(beaver_test) gtest_discover_tests(constant_lut_test) gtest_discover_tests(signed_prefix_test) gtest_discover_tests(easy_lut_test) +gtest_discover_tests(dyadic_lut_test) +gtest_discover_tests(nmod_test) gtest_discover_tests(principal_lut_test) +gtest_discover_tests(range_lut_test) gtest_discover_tests(window_lut_test) gtest_discover_tests(offset_horner_test) +add_executable(ic_test tests/ic_test.cpp) +gtest_discover_tests(ic_test) +gtest_discover_tests(corner_gaps_test) diff --git a/test/tests/beaver_test.cpp b/test/tests/beaver_test.cpp index 42c42cd..b8b97c3 100644 --- a/test/tests/beaver_test.cpp +++ b/test/tests/beaver_test.cpp @@ -1552,3 +1552,26 @@ TEST(Beaver, RejectsBadUse) EXPECT_THROW((void)[&] { return a.bit_mul(x, x); }(), std::invalid_argument); (void)z; } + +TEST(Beaver, ProductExtremes) +{ + const std::tuple cases[] = { + {0ull, 5ull, 0ull}, + {7ull, 0ull, 0ull}, + {~u64{0}, ~u64{0}, 1ull}, + {1ull, ~u64{0}, ~u64{0}}, + }; + for (const auto & [a, b, want] : cases) + { + session64 s; + auto x = s.input(); + auto y = s.input(); + auto z = s(x * y); + Counter rng; + s.sample(rng); + s.bind(x, a, rng); + s.bind(y, b, rng); + s.evaluate(); + EXPECT_EQ(s.open(z), want) << a << " * " << b; + } +} diff --git a/test/tests/blocked_dcf_test.cpp b/test/tests/blocked_dcf_test.cpp new file mode 100644 index 0000000..da539fe --- /dev/null +++ b/test/tests/blocked_dcf_test.cpp @@ -0,0 +1,404 @@ +#include + +#include "dpf.hpp" +#include "dpf/json.hpp" +#include "grotto/offset_horner.hpp" +#include "grotto/prefix_parity.hpp" + +#include +#include +#include +#include +#include + +namespace +{ + +simde__m128i g_roots[16]; +int g_ri = 0; +simde__m128i take_root() { return g_roots[g_ri++]; } + +struct PadA +{ + uint64_t n = 1; + simde__m128i block() + { + auto v = simde_mm_set_epi64x(static_cast(n), + static_cast(n * 9 + 3)); + n += 2; + return v; + } + void fill(void * p, std::size_t nbytes) + { + auto * b = static_cast(p); + for (std::size_t i = 0; i < nbytes; ++i) + b[i] = static_cast(n + i * 17); + n += nbytes; + } + uint8_t bit() { return static_cast(n++ & 1u); } +}; + +void reset_tape() +{ + g_ri = 0; + for (int i = 0; i < 16; ++i) + g_roots[i] = simde_mm_set_epi64x(0x11110000LL + i, 0x22220000LL + i * 3); +} + +template +auto recon(const A & a, const B & b) +{ + return dpf::reconstruct(a, b); +} + +template +uint64_t recon_cmp(const K0 & k0, const K1 & k1, X x) +{ + return recon(dpf::eval_point(dpf::cmp, k0, x), + dpf::eval_point(dpf::cmp, k1, x)) & k0.cmp().mask; +} + +template +void expect_u8_kind(uint8_t alpha, Spec spec, uint64_t below, uint64_t at, + uint64_t above) +{ + auto [k0, k1] = dpf::make_dpf(alpha, dpf::block_width(std::move(spec))); + using KT = std::decay_t; + EXPECT_GT(KT::cmp_block, 0u); + EXPECT_EQ(KT::cmp_q, 2u); + EXPECT_EQ(KT::cmp_h, 6u); + EXPECT_EQ(k0.value_cw().size(), KT::cmp_checkpoints); + EXPECT_EQ(k0.tail_cw().size(), KT::cmp_tail); + EXPECT_EQ(KT::depth, KT::cmp_h); + for (int x = 0; x < 256; ++x) + { + const auto q = static_cast(x); + const uint64_t got = recon_cmp(k0, k1, q); + const uint64_t want = q < alpha ? below : (q == alpha ? at : above); + ASSERT_EQ(got, want) << "B=" << B << " x=" << x << " alpha=" << int(alpha); + } +} + +template +void exhaustive_lt(uint8_t alpha, uint64_t yt, uint64_t yf) +{ + expect_u8_kind(alpha, dpf::lt(yt, yf), yt, yf, yf); +} + +} // namespace + +TEST(BlockedDcf, ExhaustiveUint8AllWidths) +{ + const uint64_t yt = 9, yf = 0; + for (uint8_t alpha : {uint8_t{0}, uint8_t{1}, uint8_t{7}, uint8_t{64}, + uint8_t{127}, uint8_t{200}, uint8_t{255}}) + { + exhaustive_lt<1>(alpha, yt, yf); + exhaustive_lt<2>(alpha, yt, yf); + exhaustive_lt<3>(alpha, yt, yf); + exhaustive_lt<4>(alpha, yt, yf); + } +} + +TEST(BlockedDcf, KindsIfFalseAndPayloadWidths) +{ + const uint8_t alpha = 40; + expect_u8_kind<4>(alpha, dpf::lt(uint64_t{9}, uint64_t{2}), 9, 2, 2); + expect_u8_kind<4>(alpha, dpf::leq(uint64_t{9}, uint64_t{2}), 9, 9, 2); + expect_u8_kind<4>(alpha, dpf::gt(uint64_t{9}, uint64_t{2}), 2, 2, 9); + expect_u8_kind<4>(alpha, dpf::geq(uint64_t{9}, uint64_t{2}), 2, 9, 9); + expect_u8_kind<4>(alpha, dpf::lt(dpf::bit::one), 1, 0, 0); + expect_u8_kind<2>(alpha, dpf::lt(uint16_t{7}, uint16_t{1}), 7, 1, 1); +} + +TEST(BlockedDcf, Block1MatchesFunctionNotBytes) +{ + const uint8_t alpha = 30; + auto bare = dpf::make_dpf(alpha, dpf::lt(uint64_t{4})); + auto blocked = dpf::make_dpf(alpha, dpf::block_width<1>(dpf::lt(uint64_t{4}))); + using Bare = std::decay_t; + using Blk = std::decay_t; + EXPECT_NE(Bare::depth, Blk::depth); + EXPECT_NE(bare.first.value_cw().size(), blocked.first.value_cw().size()); + for (int x = 0; x < 256; ++x) + { + const auto q = static_cast(x); + EXPECT_EQ(recon_cmp(bare.first, bare.second, q), + recon_cmp(blocked.first, blocked.second, q)); + } +} + +TEST(BlockedDcf, DeeperOutputForcesFullTree) +{ + const uint32_t alpha = 0x01020304u; + auto [k0, k1] = dpf::make_dpf(alpha, uint32_t{11}, + dpf::block_width<4>(dpf::lt_at<8>(uint64_t{5}, uint64_t{1}))); + using KT = std::decay_t; + EXPECT_EQ(KT::cmp_q, 0u); + EXPECT_EQ(KT::cmp_h, 8u); + EXPECT_GT(KT::depth, KT::cmp_h); + EXPECT_EQ(recon(*dpf::eval_point(k0, alpha), *dpf::eval_point(k1, alpha)), 11u); + EXPECT_EQ(recon(*dpf::eval_point(k0, alpha ^ 1u), + *dpf::eval_point(k1, alpha ^ 1u)), 0u); + auto top = [](uint32_t v) { return static_cast(v >> 24); }; + auto cmp = [&](uint32_t q) { return recon_cmp(k0, k1, q); }; + EXPECT_EQ(cmp((uint32_t{top(alpha)} - 1u) << 24), 5u); + EXPECT_EQ(cmp(alpha), 1u); + EXPECT_EQ(cmp((uint32_t{top(alpha)} + 1u) << 24), 1u); +} + +TEST(BlockedDcf, ShallowerLeafKeepsTail) +{ + const uint16_t alpha = 0x1234; + auto [k0, k1] = dpf::make_dpf(alpha, dpf::at<4>(uint8_t{3}), + dpf::block_width<4>(dpf::lt(uint64_t{8}))); + using KT = std::decay_t; + EXPECT_EQ(KT::cmp_q, 2u); + EXPECT_EQ(recon(*dpf::eval_point(dpf::out<0, 4>, k0, alpha), + *dpf::eval_point(dpf::out<0, 4>, k1, alpha)), 3u); + EXPECT_EQ(recon_cmp(k0, k1, uint16_t{alpha - 1}), 8u); + EXPECT_EQ(recon_cmp(k0, k1, alpha), 0u); +} + +TEST(BlockedDcf, PointIntervalFullSequenceInnerProduct) +{ + const uint8_t alpha = 100; + auto [k0, k1] = dpf::make_dpf(alpha, + dpf::block_width<4>(dpf::lt(uint64_t{6}, uint64_t{1}))); + using KT = std::decay_t; + auto path0 = dpf::make_basic_path_memoizer(k0); + auto path1 = dpf::make_basic_path_memoizer(k1); + EXPECT_EQ(recon_cmp(k0, k1, uint8_t{50}), 6u); + EXPECT_EQ(dpf::reconstruct( + dpf::eval_point(dpf::cmp, k0, uint8_t{50}, path0), + dpf::eval_point(dpf::cmp, k1, uint8_t{50}, path1)) & k0.cmp().mask, + 6u); + EXPECT_EQ(dpf::reconstruct( + dpf::eval_point(dpf::cmp, k0, uint8_t{150}, path0), + dpf::eval_point(dpf::cmp, k1, uint8_t{150}, path1)) & k0.cmp().mask, + 1u); + + auto narrow0 = dpf::make_output_buffer(dpf::cmp, k0, uint8_t{90}, uint8_t{110}); + auto narrow1 = dpf::make_output_buffer(dpf::cmp, k1, uint8_t{90}, uint8_t{110}); + dpf::eval_interval(dpf::cmp, k0, uint8_t{90}, uint8_t{110}, narrow0); + dpf::eval_interval(dpf::cmp, k1, uint8_t{90}, uint8_t{110}, narrow1); + for (uint8_t x = 90; x <= 110; ++x) + { + EXPECT_EQ(recon(narrow0[x - 90], narrow1[x - 90]) & k0.cmp().mask, + recon_cmp(k0, k1, x)); + } + + constexpr std::size_t stop = KT::cmp_depth; + dpf::detail::incr::cmp_full_interval_memo memo0{21}; + dpf::detail::incr::cmp_full_interval_memo memo1{21}; + auto again0 = dpf::make_output_buffer(dpf::cmp, k0, uint8_t{90}, uint8_t{110}); + auto again1 = dpf::make_output_buffer(dpf::cmp, k1, uint8_t{90}, uint8_t{110}); + dpf::eval_interval(dpf::cmp, k0, uint8_t{90}, uint8_t{110}, again0, memo0); + dpf::eval_interval(dpf::cmp, k1, uint8_t{90}, uint8_t{110}, again1, memo1); + EXPECT_EQ(recon(again0[0], again1[0]) & k0.cmp().mask, 6u); + EXPECT_EQ(recon(again0[10], again1[10]) & k0.cmp().mask, 1u); + + auto full0 = dpf::eval_full(dpf::cmp, k0); + auto full1 = dpf::eval_full(dpf::cmp, k1); + ASSERT_EQ(full0.size(), 256u); + for (int x = 0; x < 256; ++x) + { + EXPECT_EQ(recon(full0[x], full1[x]) & k0.cmp().mask, + recon_cmp(k0, k1, static_cast(x))); + } + + std::array pts{{0, 99, 100, 255}}; + auto seq0 = dpf::make_output_buffer(dpf::cmp, k0, pts.size()); + auto seq1 = dpf::make_output_buffer(dpf::cmp, k1, pts.size()); + dpf::eval_sequence(dpf::cmp, k0, pts.begin(), pts.end(), seq0, path0); + dpf::eval_sequence(dpf::cmp, k1, pts.begin(), pts.end(), seq1, path1); + for (std::size_t i = 0; i < pts.size(); ++i) + { + EXPECT_EQ(recon(seq0[i], seq1[i]) & k0.cmp().mask, + recon_cmp(k0, k1, pts[i])); + } + + std::vector w(21, 1); + const uint64_t dot = + dpf::eval_inner_product(dpf::cmp, k0, uint8_t{90}, uint8_t{110}, w) + + dpf::eval_inner_product(dpf::cmp, k1, uint8_t{90}, uint8_t{110}, w); + uint64_t want = 0; + for (uint8_t x = 90; x <= 110; ++x) + want = (want + recon_cmp(k0, k1, x)) & k0.cmp().mask; + EXPECT_EQ(dot & k0.cmp().mask, want); +} + +TEST(BlockedDcf, DealerMatchesDoernerShelat) +{ + const uint8_t alpha = 77; + const uint8_t x0 = 3; + const uint8_t x1 = static_cast(alpha ^ x0); + reset_tape(); + auto dealer = dpf::make_dpf(alpha, + dpf::root_sampler_t{take_root}, + dpf::block_width<4>(dpf::lt(uint64_t{15}, uint64_t{2}))); + reset_tape(); + dpf::ds_randomness rng{take_root, {}}; + auto ds = dpf::make_dpf_doerner_shelat(x0, x1, rng, + dpf::block_width<4>(dpf::lt(uint64_t{15}, uint64_t{2}))); + + EXPECT_EQ(std::memcmp(dealer.first.correction_words().data(), + ds.first.correction_words().data(), + dealer.first.correction_words().size() + * sizeof(dealer.first.correction_words()[0])), + 0); + EXPECT_EQ(dealer.first.correction_advice(), ds.first.correction_advice()); + EXPECT_EQ(dealer.first.value_cw(), ds.first.value_cw()); + EXPECT_EQ(dealer.first.tail_cw(), ds.first.tail_cw()); + EXPECT_EQ(dealer.first.cw_last(), ds.first.cw_last()); + EXPECT_EQ(dealer.first.cmp_addend().raw(), ds.first.cmp_addend().raw()); + EXPECT_EQ(dealer.second.cmp_addend().raw(), ds.second.cmp_addend().raw()); + for (int x = 0; x < 256; ++x) + { + const auto q = static_cast(x); + EXPECT_EQ(recon_cmp(dealer.first, dealer.second, q), + recon_cmp(ds.first, ds.second, q)); + } +} + +TEST(BlockedDcf, WildcardAssignMatchesConcrete) +{ + const uint8_t alpha = 19; + const uint64_t yt = 42; + reset_tape(); + auto wild = dpf::make_dpf(alpha, + dpf::root_sampler_t{take_root}, + dpf::block_width<4>(dpf::lt(dpf::wildcard_value{}))); + EXPECT_FALSE(wild.first.cmp_assigned()); + EXPECT_THROW(dpf::eval_point(dpf::cmp, wild.first, alpha), std::exception); + reset_tape(); + auto concrete = dpf::make_dpf(alpha, + dpf::root_sampler_t{take_root}, + dpf::block_width<4>(dpf::lt(yt))); + dpf::assign_cmp(wild.first, wild.second, yt); + EXPECT_TRUE(wild.first.cmp_assigned()); + EXPECT_EQ(wild.first.value_cw(), concrete.first.value_cw()); + EXPECT_EQ(wild.first.tail_cw(), concrete.first.tail_cw()); + EXPECT_EQ(wild.first.cw_last(), concrete.first.cw_last()); + for (int x = 0; x < 256; ++x) + { + const auto q = static_cast(x); + EXPECT_EQ(recon_cmp(wild.first, wild.second, q), + recon_cmp(concrete.first, concrete.second, q)); + } +} + +TEST(BlockedDcf, TrivialDomainEdges) +{ + auto hi_gt = dpf::make_dpf(uint8_t{255}, + dpf::block_width<4>(dpf::gt(uint64_t{3}))); + EXPECT_EQ(hi_gt.first.cmp().trivial, dpf::cmp_trivial::always_false); + EXPECT_EQ(recon_cmp(hi_gt.first, hi_gt.second, uint8_t{0}), 0u); + EXPECT_EQ(recon_cmp(hi_gt.first, hi_gt.second, uint8_t{255}), 0u); + + auto hi_leq = dpf::make_dpf(uint8_t{255}, + dpf::block_width<4>(dpf::leq(uint64_t{3}))); + EXPECT_EQ(hi_leq.first.cmp().trivial, dpf::cmp_trivial::always_true); + EXPECT_EQ(recon_cmp(hi_leq.first, hi_leq.second, uint8_t{0}), 3u); + EXPECT_EQ(recon_cmp(hi_leq.first, hi_leq.second, uint8_t{255}), 3u); +} + +TEST(BlockedDcf, JsonRoundTrip) +{ + const uint8_t alpha = 12; + auto [k0, k1] = dpf::make_dpf(alpha, + dpf::block_width<4>(dpf::gt(uint64_t{7}, uint64_t{1}))); + using KT = typename std::decay_t::key_type; + const std::string s0 = dpf::json::to_json(k0.key()); + const std::string s1 = dpf::json::to_json(k1.key()); + EXPECT_NE(s0.find("block_width"), std::string::npos); + EXPECT_NE(s0.find("tail_cw"), std::string::npos); + auto r0 = dpf::json::from_json(s0); + auto r1 = dpf::json::from_json(s1); + EXPECT_EQ(r0.value_cw(), k0.value_cw()); + EXPECT_EQ(r0.tail_cw(), k0.tail_cw()); + EXPECT_EQ(r0.cmp().block_width, 4); + EXPECT_EQ(r0.cmp().tail_bits, static_cast(KT::cmp_q)); + const uint64_t mask = k0.cmp().mask; + for (int x = 0; x < 256; ++x) + { + const auto q = static_cast(x); + const uint64_t got = + (dpf::eval_point(dpf::cmp, r0, q) + dpf::eval_point(dpf::cmp, r1, q)) + & mask; + EXPECT_EQ(got, recon_cmp(k0, k1, q)); + } +} + +TEST(BlockedDcf, GenevalOpensCheckpointWords) +{ + const uint8_t alpha = 20; + const uint8_t x0 = 1; + const uint8_t x1 = static_cast(alpha ^ x0); + reset_tape(); + dpf::ds_randomness rng{take_root, {}}; + std::array ends{{0, 20, 21}}; + auto g = dpf::geneval_cmp(x0, x1, ends.begin(), ends.end(), rng, + dpf::block_width<4>(dpf::lt(uint64_t{5}))); + using KT = std::decay_t(dpf::lt(uint64_t{5}))).first)>; + EXPECT_EQ(g.value_cw.size(), KT::cmp_checkpoints); + EXPECT_EQ(g.tail_cw.size(), KT::cmp_tail); + EXPECT_EQ(g.correction_words.size(), KT::depth); + EXPECT_EQ((g.party0[0] + g.party1[0]) & g.mask, 5u); + EXPECT_EQ((g.party0[1] + g.party1[1]) & g.mask, 0u); + EXPECT_EQ((g.party0[2] + g.party1[2]) & g.mask, 0u); +} + +TEST(BlockedDcf, GrottoPrefixSegmentAndHorner) +{ + const uint16_t alpha = 1000; + std::array ends{{0, 500, 1000, 4000}}; + auto bare = dpf::make_dpf(alpha, dpf::gt(uint64_t{1})); + auto blk = dpf::make_dpf(alpha, dpf::block_width<4>(dpf::gt(uint64_t{1}))); + const auto b0 = grotto::signed_prefix_parities(bare.first, ends); + const auto b1 = grotto::signed_prefix_parities(bare.second, ends); + const auto k0 = grotto::signed_prefix_parities(blk.first, ends); + const auto k1 = grotto::signed_prefix_parities(blk.second, ends); + const uint64_t maskp = bare.first.cmp().mask; + for (std::size_t i = 0; i < ends.size(); ++i) + EXPECT_EQ((b0[i] + b1[i]) & maskp, (k0[i] + k1[i]) & maskp); + + const auto s0 = grotto::signed_segment_parities(bare.first, ends); + const auto s1 = grotto::signed_segment_parities(bare.second, ends); + const auto t0 = grotto::signed_segment_parities(blk.first, ends); + const auto t1 = grotto::signed_segment_parities(blk.second, ends); + for (std::size_t i = 0; i < ends.size(); ++i) + EXPECT_EQ((s0[i] + s1[i]) & maskp, (t0[i] + t1[i]) & maskp); + + const uint16_t center = 30; + auto mat = grotto::make_offset_horner_keys(center); + uint64_t payload[2]; + payload[0] = 1; + payload[1] = center; + using bare_pair = decltype(dpf::make_dpf(center, dpf::gt(uint64_t{0}))); + using pair = decltype(dpf::make_dpf(center, + dpf::block_width<4>(dpf::gt(uint64_t{0})))); + std::vector bare_keys{ + dpf::make_dpf(center, dpf::gt(payload[0])), + dpf::make_dpf(center, dpf::gt(payload[1]))}; + std::vector keys{ + dpf::make_dpf(center, dpf::block_width<4>(dpf::gt(payload[0]))), + dpf::make_dpf(center, dpf::block_width<4>(dpf::gt(payload[1])))}; + std::vector knots{0, 10, 40}; + std::vector> coeff{ + {1, 0}, + {2, 3}, + {4, 1}}; + const uint16_t eta = 0; + const auto b0s = grotto::offset_horner_coefficient_share<0, 1>( + bare_keys, mat.wrap_share, knots, coeff, eta); + const auto b1s = grotto::offset_horner_coefficient_share<1, 1>( + bare_keys, mat.wrap_share, knots, coeff, eta); + const auto c0 = grotto::offset_horner_coefficient_share<0, 1>( + keys, mat.wrap_share, knots, coeff, eta); + const auto c1 = grotto::offset_horner_coefficient_share<1, 1>( + keys, mat.wrap_share, knots, coeff, eta); + for (std::size_t m = 0; m < 2; ++m) + EXPECT_EQ(b0s[m] + b1s[m], c0[m] + c1[m]); +} diff --git a/test/tests/corner_gaps_test.cpp b/test/tests/corner_gaps_test.cpp new file mode 100644 index 0000000..d963ab4 --- /dev/null +++ b/test/tests/corner_gaps_test.cpp @@ -0,0 +1,390 @@ +#include + +#include "dpf.hpp" +#include "grotto/fixedpoint.hpp" +#include "grotto/fixedpoint_mul.hpp" +#include "grotto/principal_lut.hpp" + +#include +#include +#include +#include +#include +#include + +namespace +{ + +template +void expect_same(const T & got, const T & point) +{ + EXPECT_EQ(std::memcmp(&got, &point, sizeof(T)), 0); +} + +template +void expect_wrapping_interval(In from, In to, In alpha, Out y, std::size_t leaves) +{ + auto [k0, k1] = dpf::make_dpf(alpha, y); + using key_t = std::decay_t; + EXPECT_EQ((dpf::utils::get_nodes_in_interval(from, to)), leaves); + + auto [buf0, it0] = dpf::eval_interval(k0, from, to); + auto [buf1, it1] = dpf::eval_interval(k1, from, to); + auto a = std::begin(it0); + auto b = std::begin(it1); + const auto a_end = std::end(it0); + const auto b_end = std::end(it1); + + std::size_t n = 0; + In cur = from; + for (;;) + { + ASSERT_NE(a, a_end); + ASSERT_NE(b, b_end); + auto p0 = *dpf::eval_point(k0, cur); + auto p1 = *dpf::eval_point(k1, cur); + expect_same(*a, p0); + expect_same(*b, p1); + ++a; + ++b; + ++n; + if (cur == to) + break; + cur = static_cast(static_cast(cur) + 1u); + ASSERT_LT(n, std::size_t{1} << 20); + } + EXPECT_EQ(a, a_end); + EXPECT_EQ(b, b_end); + const std::uint64_t width = std::uint64_t{1} << dpf::utils::bitlength_of_v; + const std::uint64_t masked = (static_cast(to) + - static_cast(from)) & (width - 1); + EXPECT_EQ(n, masked + 1); +} + +} // namespace + +TEST(CornerGaps, SameLeafWrapUint8Uint32) +{ + expect_wrapping_interval(10, 9, 40, 0x11111111u, 65); +} + +TEST(CornerGaps, AdjacentLeafWrapUint8Uint32) +{ + expect_wrapping_interval(8, 7, 40, 0x22222222u, 64); +} + +TEST(CornerGaps, LongWrapStillMatchesPoint) +{ + expect_wrapping_interval(200, 10, 3, 0x33333333u, 17); + + auto [k0, k1] = dpf::make_dpf(uint8_t{3}, uint32_t{0x33333333u}); + using key_t = std::decay_t; + auto memo0 = dpf::make_full_tree_interval_memoizer(uint8_t{200}, uint8_t{10}); + auto memo1 = dpf::make_full_tree_interval_memoizer(uint8_t{200}, uint8_t{10}); + auto [buf0, it0] = dpf::eval_interval(k0, uint8_t{200}, uint8_t{10}, std::move(memo0)); + auto [buf1, it1] = dpf::eval_interval(k1, uint8_t{200}, uint8_t{10}, std::move(memo1)); + auto p0 = *dpf::eval_point(k0, uint8_t{200}); + auto p1 = *dpf::eval_point(k1, uint8_t{200}); + expect_same(*std::begin(it0), p0); + expect_same(*std::begin(it1), p1); + EXPECT_EQ(dpf::reconstruct(*std::begin(it0), *std::begin(it1)), 0u); + auto back0 = std::begin(it0); + auto back1 = std::begin(it1); + for (int i = 0; i < 66; ++i, ++back0, ++back1) {} + auto q0 = *dpf::eval_point(k0, uint8_t{10}); + auto q1 = *dpf::eval_point(k1, uint8_t{10}); + expect_same(*back0, q0); + expect_same(*back1, q1); +} + +TEST(CornerGaps, OneOutputPerLeafWrap) +{ + expect_wrapping_interval(5, 4, 9, simde_uint128{7}, 65536); +} + +TEST(CornerGaps, SaturatedUint64LeafCount) +{ + using in_t = uint64_t; + using out_t = simde_uint128; + using key_t = dpf::utils::dpf_type_t; + EXPECT_EQ((dpf::utils::get_nodes_in_interval(in_t{1}, ~in_t{0})), + std::numeric_limits::max()); + EXPECT_THROW((dpf::utils::get_nodes_in_interval(in_t{0}, ~in_t{0})), + std::length_error); + EXPECT_THROW((dpf::utils::get_nodes_in_interval(in_t{5}, in_t{4})), + std::length_error); +} + +TEST(CornerGaps, MemoizerRejectsALargerInterval) +{ + auto [k0, k1] = dpf::make_dpf(uint8_t{4}, uint32_t{1}); + using key_t = std::decay_t; + auto memo = dpf::make_basic_interval_memoizer(uint8_t{0}, uint8_t{10}); + auto buf = dpf::make_output_buffer_for_interval(uint8_t{0}, uint8_t{100}); + EXPECT_THROW(dpf::eval_interval(k0, uint8_t{0}, uint8_t{100}, buf, memo), + std::length_error); + (void)k1; +} + +TEST(CornerGaps, OddStartInteriorTail) +{ + auto [k0, k1] = dpf::make_dpf(uint16_t{3}, simde_uint128{11}); + for (uint16_t to : {uint16_t{9}, uint16_t{10}}) + { + auto [b0, it0] = dpf::eval_interval(k0, uint16_t{1}, to); + auto [b1, it1] = dpf::eval_interval(k1, uint16_t{1}, to); + auto a = std::begin(it0); + auto b = std::begin(it1); + for (uint16_t q = 1; q <= to; ++q, ++a, ++b) + { + auto p0 = *dpf::eval_point(k0, q); + auto p1 = *dpf::eval_point(k1, q); + expect_same(*a, p0); + expect_same(*b, p1); + } + EXPECT_EQ(a, std::end(it0)); + EXPECT_EQ(b, std::end(it1)); + } +} + +TEST(CornerGaps, PathMemoizerExtremes) +{ + auto [k0, k1] = dpf::make_dpf(uint16_t{0x0102}, uint16_t{9}); + using key_t = std::decay_t; + dpf::basic_path_memoizer m0; + dpf::basic_path_memoizer m1; + const uint16_t queries[] = {0, 0x8000, 1, 0}; + for (uint16_t q : queries) + { + auto a = *dpf::eval_point(k0, q, m0); + auto b = *dpf::eval_point(k1, q, m1); + auto fa = *dpf::eval_point(k0, q); + auto fb = *dpf::eval_point(k1, q); + EXPECT_EQ(a, fa) << q; + EXPECT_EQ(b, fb) << q; + } +} + +TEST(CornerGaps, DepthOneBitInterval) +{ + for (uint8_t alpha : {uint8_t{0}, uint8_t{127}, uint8_t{128}, uint8_t{255}}) + { + auto [k0, k1] = dpf::make_dpf(alpha, dpf::bit::one); + using key_t = std::decay_t; + EXPECT_EQ(key_t::depth, 1u); + + auto check = [&](uint8_t from, uint8_t to) { + auto [b0, it0] = dpf::eval_interval(k0, from, to); + auto [b1, it1] = dpf::eval_interval(k1, from, to); + auto a = std::begin(it0); + auto b = std::begin(it1); + for (uint8_t q = from; ; ) + { + const bool on = q == alpha; + const bool bit = static_cast(*a) != static_cast(*b); + EXPECT_EQ(bit, on) << int(q); + ++a; + ++b; + if (q == to) + break; + ++q; + } + EXPECT_EQ(a, std::end(it0)); + EXPECT_EQ(std::end(it1), b); + }; + check(alpha, alpha); + check(127, 128); + } +} + +TEST(CornerGaps, EmptyAndDuplicateSequence) +{ + auto [k0, k1] = dpf::make_dpf(uint8_t{40}, uint8_t{7}); + std::vector empty; + auto [eb0, eit0] = dpf::eval_sequence(k0, empty.begin(), empty.end()); + auto [eb1, eit1] = dpf::eval_sequence(k1, empty.begin(), empty.end()); + EXPECT_EQ(std::begin(eit0), std::end(eit0)); + EXPECT_EQ(std::begin(eit1), std::end(eit1)); + + const std::vector seq{40, 40, 41}; + auto [b0, it0] = dpf::eval_sequence(k0, seq.begin(), seq.end()); + auto [b1, it1] = dpf::eval_sequence(k1, seq.begin(), seq.end()); + auto a = std::begin(it0); + auto b = std::begin(it1); + const uint8_t want[] = {7, 7, 0}; + for (int i = 0; i < 3; ++i, ++a, ++b) + EXPECT_EQ(dpf::reconstruct(*a, *b), want[i]) << i; + EXPECT_EQ(a, std::end(it0)); +} + +TEST(CornerGaps, IncrementalAdjacentLaneWrap) +{ + const uint16_t alpha = 0x00ab; + auto [k0, k1] = dpf::make_dpf(alpha, dpf::at<8>(uint32_t{0xabcdu})); + auto [b0, it0] = dpf::eval_interval(dpf::out<0, 8>, k0, uint8_t{8}, uint8_t{7}); + auto [b1, it1] = dpf::eval_interval(dpf::out<0, 8>, k1, uint8_t{8}, uint8_t{7}); + auto a = std::begin(it0); + auto b = std::begin(it1); + std::size_t n = 0; + for (int lane = 8; ; ) + { + const uint16_t query = static_cast(static_cast(lane) << 8); + auto p0 = *dpf::eval_point(dpf::out<0, 8>, k0, query); + auto p1 = *dpf::eval_point(dpf::out<0, 8>, k1, query); + EXPECT_EQ(*a, p0) << lane; + EXPECT_EQ(*b, p1) << lane; + ++a; + ++b; + ++n; + if (lane == 7) + break; + lane = (lane + 1) & 255; + } + EXPECT_EQ(n, 256u); + EXPECT_EQ(a, std::end(it0)); +} + +TEST(CornerGaps, ModintWideLiteralAndLimbShift) +{ + using namespace dpf::literals; + EXPECT_EQ(1_u129, dpf::modint<129>{1}); + const uint256_t bit128{1, 0}; + const auto wide = 340282366920938463463374607431768211456_u129; + EXPECT_EQ(wide, (dpf::modint<129>{bit128})); + + const bool shift10 = (dpf::modint<10>{1} << 10) == dpf::modint<10>{0}; + const bool shift16 = (dpf::modint<10>{1} << 16) == dpf::modint<10>{0}; + const bool shift64 = (dpf::modint<64>{1} << 64) == dpf::modint<64>{0}; + const bool shift128 = (dpf::modint<65>{1} << 128) == dpf::modint<65>{0}; + const bool rshift10 = (dpf::modint<10>{5} >> 10) == dpf::modint<10>{0}; + const bool rshift64 = (dpf::modint<64>{1} >> 64) == dpf::modint<64>{0}; + EXPECT_TRUE(shift10 && shift16 && shift64 && shift128 && rshift10 && rshift64); + + dpf::modint<10> assigned{1}; + assigned <<= 16; + EXPECT_TRUE(assigned == dpf::modint<10>{0}); + assigned = dpf::modint<10>{7}; + assigned >>= 10; + EXPECT_TRUE(assigned == dpf::modint<10>{0}); +} + +TEST(CornerGaps, SetbitEmptyAndSingleAndNarrowLeaf) +{ + dpf::dynamic_bit_array<> zeros(128); + using iter_t = decltype(zeros.begin()); + dpf::subinterval_iterable all(zeros.begin(), zeros.size(), + 0, zeros.size() - 1, 0, 0); + auto none = dpf::indices_set_in(all); + EXPECT_EQ(none.begin(), none.end()); + + zeros[0] = true; + dpf::subinterval_iterable one(zeros.begin(), zeros.size(), + 0, zeros.size() - 1, 0, 0); + auto set = dpf::indices_set_in(one); + auto it = set.begin(); + ASSERT_NE(it, set.end()); + EXPECT_EQ(*it, 0u); + ++it; + EXPECT_EQ(it, set.end()); + + dpf::dynamic_bit_array<> narrow(128); + narrow[0] = true; + dpf::subinterval_iterable clipped(narrow.begin(), narrow.size(), + 0, 1, 0, 2); + auto clipped_set = dpf::indices_set_in(clipped); + auto cit = clipped_set.begin(); + ASSERT_NE(cit, clipped_set.end()); + EXPECT_EQ(*cit, 0u); + ++cit; + EXPECT_EQ(cit, clipped_set.end()); +} + +TEST(CornerGaps, EmptyRotationIsEmptyAndZeroIsIdentity) +{ + std::vector empty; + dpf::rotation_iterable::iterator> none( + empty.begin(), empty.end(), 1); + EXPECT_EQ(none.begin(), none.end()); + + std::vector values{1, 2, 3}; + dpf::rotation_iterable::iterator> id( + values.begin(), values.end(), 0); + std::vector got; + for (auto it = id.begin(); it != id.end(); ++it) + got.push_back(*it); + EXPECT_EQ(got, values); + + dpf::rotation_iterable::iterator> rot( + values.begin(), values.end(), 1); + got.clear(); + for (auto it = rot.begin(); it != rot.end(); ++it) + got.push_back(*it); + EXPECT_EQ(got, (std::vector{2, 3, 1})); +} + +TEST(CornerGaps, Party1NegatesSignedMinimum) +{ + using sub = dpf::subtractive_share; + using add0 = dpf::additive_share; + const auto raw = std::numeric_limits::min(); + const auto party1 = sub::from_raw(raw).as_additive(); + EXPECT_EQ(party1.raw(), raw); + EXPECT_EQ(dpf::reconstruct(add0::from_raw(0), party1), raw); + EXPECT_EQ((-sub::from_raw(raw)).raw(), raw); +} + +TEST(CornerGaps, FixedMulFloorsAndPrecisionCastDiffersFromLogicalShift) +{ + using q4 = grotto::fixedpoint<4, std::int32_t>; + const auto prod = grotto::fixed_mul<8, 4>(q4::from_raw(-3), q4::from_raw(1)); + EXPECT_EQ(prod.integral_representation(), -1); + + const auto neg_pair = grotto::fixed_mul<8, 4>(q4::from_raw(-3), q4::from_raw(-2)); + EXPECT_EQ(neg_pair.integral_representation(), 0); + + auto quarter = q4::from_raw(-12); + const auto casted = grotto::precision_cast<0>(quarter); + EXPECT_EQ(casted.integral_representation(), -1); + quarter >>= 4; + EXPECT_EQ(quarter.integral_representation(), 268435455); +} + +TEST(CornerGaps, Int64MinFactorIsDefined) +{ + using grotto::principal_detail::w_from_i128; + using grotto::principal_detail::w_mul_i64; + using grotto::principal_detail::w_mul_u64; + using grotto::principal_detail::w_neg; + const auto value = w_from_i128(3); + const auto got = w_mul_i64(value, std::numeric_limits::min()); + const auto want = w_neg(w_mul_u64(value, std::uint64_t{1} << 63)); + EXPECT_EQ(got.lo, want.lo); + EXPECT_EQ(got.hi, want.hi); +} + +TEST(CornerGaps, PrgRejectsUint32Seam) +{ + alignas(64) simde__m128i seed = simde_mm_set_epi64x(1, 2); + alignas(64) simde__m128i out[4]; + const auto pos = static_cast(UINT32_MAX - 1u); + EXPECT_THROW(dpf::prg::aes128::eval(seed, out, 4, pos), std::invalid_argument); + EXPECT_THROW(dpf::prg::lowmc128::eval(seed, out, 4, pos), std::invalid_argument); + + const auto ok = static_cast(UINT32_MAX - 3u); + dpf::prg::aes128::eval(seed, out, 4, ok); + for (psnip_uint32_t i = 0; i < 4; ++i) + { + const auto one = dpf::prg::aes128::eval(seed, ok + i); + EXPECT_EQ(std::memcmp(&out[i], &one, sizeof(one)), 0) << i; + } + dpf::prg::lowmc128::eval(seed, out, 4, ok); + for (psnip_uint32_t i = 0; i < 4; ++i) + { + const auto one = dpf::prg::lowmc128::eval(seed, ok + i); + EXPECT_EQ(std::memcmp(&out[i], &one, sizeof(one)), 0) << i; + } + + EXPECT_THROW((dpf::randomness::detail::lane_codec::fill( + seed, static_cast(UINT32_MAX) - 1u, out, 4)), + std::invalid_argument); +} diff --git a/test/tests/dyadic_lut_test.cpp b/test/tests/dyadic_lut_test.cpp new file mode 100644 index 0000000..479e467 --- /dev/null +++ b/test/tests/dyadic_lut_test.cpp @@ -0,0 +1,215 @@ +#include + +#include "grotto/dyadic_lut.hpp" + +#include +#include + +namespace +{ + +int ref_clz(std::int16_t raw) +{ + const auto bits = static_cast(raw); + if (bits == 0) + return 16; + return __builtin_clz(static_cast(bits)) - (32 - 16); +} + +int ref_clrsb(std::int16_t raw) +{ + const auto bits = static_cast(raw); + const bool neg = (bits & 0x8000u) != 0; + int count = 0; + for (int b = 14; b >= 0; --b) + { + const bool bit = ((bits >> b) & 1u) != 0; + if (bit != neg) + break; + ++count; + } + return count; +} + +int ref_ilogb(std::int16_t raw, unsigned k) +{ + if (raw == 0) + return 0; + unsigned mag; + if (raw == std::numeric_limits::min()) + mag = 1u << 15; + else + mag = static_cast(raw < 0 ? -raw : raw); + int log = 0; + while (mag > 1) + { + mag >>= 1; + ++log; + } + return log - static_cast(k); +} + +int ref_ilog10(std::int64_t raw, unsigned k) +{ + if (raw == 0) + return 0; + using u128 = unsigned __int128; + u128 mag; + if (raw == std::numeric_limits::min()) + mag = u128{1} << 63; + else + mag = static_cast(raw < 0 ? -raw : raw); + int e = -static_cast(k) - 2; + for (;;) + { + bool ge = false; + if (e >= 0) + { + u128 decade = 1; + for (int i = 0; i < e; ++i) + decade *= 10; + ge = mag >= (decade << k); + } + else + { + u128 decade = 1; + for (int i = 0; i < -e; ++i) + decade *= 10; + const u128 scale = u128{1} << k; + u128 threshold = scale / decade; + if (scale % decade != 0) + ++threshold; + ge = mag >= threshold; + } + if (!ge) + return e - 1; + ++e; + if (e > 40) + return 40; + } +} + +std::int64_t enc(std::int64_t units, unsigned k) +{ + return units << k; +} + +} // namespace + +TEST(DyadicLut, SignBundleInt16) +{ + const auto positive = grotto::make_positive_lut(); + const auto negative = grotto::make_negative_lut(); + const auto nonnegative = grotto::make_nonnegative_lut(); + const auto nonpositive = grotto::make_nonpositive_lut(); + const auto zero = grotto::make_zero_lut(); + const auto nonzero = grotto::make_nonzero_lut(); + const auto signum = grotto::make_signum_lut(); + EXPECT_EQ(positive.parts(), 2u); + EXPECT_EQ(signum.parts(), 3u); + EXPECT_EQ(zero.parts(), 3u); + + for (std::int32_t raw = -32768; raw <= 32767; ++raw) + { + const auto x = static_cast(raw); + EXPECT_EQ(positive(x), x > 0 ? 1 : 0); + EXPECT_EQ(negative(x), x < 0 ? 1 : 0); + EXPECT_EQ(nonnegative(x), x >= 0 ? 1 : 0); + EXPECT_EQ(nonpositive(x), x <= 0 ? 1 : 0); + EXPECT_EQ(zero(x), x == 0 ? 1 : 0); + EXPECT_EQ(nonzero(x), x != 0 ? 1 : 0); + const int sgn = x < 0 ? -1 : (x > 0 ? 1 : 0); + EXPECT_EQ(signum(x), sgn); + } + + const auto scaled = grotto::make_positive_lut(4); + EXPECT_EQ(scaled(std::int16_t{1}), 16); + EXPECT_EQ(scaled(std::int16_t{0}), 0); + EXPECT_EQ(grotto::make_signum_lut(4)(std::int16_t{-3}), -16); +} + +TEST(DyadicLut, ClzAndClrsbInt16) +{ + const auto clz = grotto::make_clz_lut(); + const auto clrsb = grotto::make_clrsb_lut(); + EXPECT_EQ(clz.parts(), 17u); + for (std::int32_t raw = -32768; raw <= 32767; ++raw) + { + const auto x = static_cast(raw); + EXPECT_EQ(clz(x), ref_clz(x)) << raw; + EXPECT_EQ(clrsb(x), ref_clrsb(x)) << raw; + } + const auto scaled = grotto::make_clz_lut(3); + EXPECT_EQ(scaled(std::int16_t{1}), ref_clz(1) << 3); +} + +TEST(DyadicLut, IntegerLogsInt16) +{ + for (unsigned k : {0u, 4u}) + { + const auto lg = grotto::make_ilogb_lut(k); + const auto log10 = grotto::make_ilog10_lut(k); + EXPECT_EQ(lg(std::int16_t{0}), grotto::ilog_of_zero); + EXPECT_EQ(log10(std::int16_t{0}), grotto::ilog_of_zero); + for (std::int32_t raw = -32768; raw <= 32767; ++raw) + { + if (raw == 0) + continue; + const auto x = static_cast(raw); + EXPECT_EQ(lg(x), enc(ref_ilogb(x, k), k)) << raw << " k=" << k; + EXPECT_EQ(log10(x), enc(ref_ilog10(raw, k), k)) << raw << " k=" << k; + } + } + EXPECT_EQ(grotto::make_ilogb_lut()(std::int16_t{1}), 0); + EXPECT_EQ(grotto::make_ilogb_lut()(std::int16_t{2}), 1); + EXPECT_EQ(grotto::make_ilog10_lut()(std::int16_t{9}), 0); + EXPECT_EQ(grotto::make_ilog10_lut()(std::int16_t{10}), 1); + EXPECT_EQ(grotto::make_ilog10_lut(4)(std::int16_t{1}), enc(-2, 4)); +} + +TEST(DyadicLut, MostSignificantBits) +{ + for (unsigned i = 0; i < grotto::msb_bit_limit && i < 16; ++i) + { + const auto lut = grotto::make_msb_lut(i); + EXPECT_EQ(lut.parts(), std::size_t{1} << (i + 1)) << i; + const int shift = 15 - static_cast(i); + for (std::int32_t raw = -32768; raw <= 32767; ++raw) + { + const auto x = static_cast(raw); + const int bit = (static_cast(x) >> shift) & 1; + EXPECT_EQ(lut(x), bit) << raw << " i=" << i; + } + } + EXPECT_THROW(grotto::make_msb_lut(grotto::msb_bit_limit), + std::invalid_argument); + const auto scaled = grotto::make_msb_lut(0, 4); + EXPECT_EQ(scaled(std::int16_t{-1}), 16); + EXPECT_EQ(scaled(std::int16_t{1}), 0); +} + +TEST(DyadicLut, Int64Edges) +{ + const auto clz = grotto::make_clz_lut(); + EXPECT_EQ(clz(std::int64_t{0}), 64); + EXPECT_EQ(clz(std::int64_t{-1}), 0); + EXPECT_EQ(clz(std::int64_t{1}), 63); + EXPECT_EQ(clz(std::numeric_limits::min()), 0); + EXPECT_EQ(clz(std::int64_t{1} << 62), 1); + + const auto clrsb = grotto::make_clrsb_lut(); + EXPECT_EQ(clrsb(std::int64_t{0}), 63); + EXPECT_EQ(clrsb(std::int64_t{-1}), 63); + EXPECT_EQ(clrsb(std::int64_t{1}), 62); + EXPECT_EQ(clrsb(std::numeric_limits::min()), 0); + + const auto lg = grotto::make_ilogb_lut(16); + EXPECT_EQ(lg(std::int64_t{1} << 16), 0); + EXPECT_EQ(lg(std::int64_t{1} << 17), enc(1, 16)); + EXPECT_EQ(lg(std::numeric_limits::min()), enc(63 - 16, 16)); + + const auto bit = grotto::make_msb_lut(0); + EXPECT_EQ(bit.parts(), 2u); + EXPECT_EQ(bit(std::int64_t{-5}), 1); + EXPECT_EQ(bit(std::int64_t{5}), 0); +} diff --git a/test/tests/geneval_test.cpp b/test/tests/geneval_test.cpp index 6ce0734..0a9da4c 100644 --- a/test/tests/geneval_test.cpp +++ b/test/tests/geneval_test.cpp @@ -362,110 +362,87 @@ TEST(Geneval, EmptySequence) EXPECT_TRUE(g.correction_words.empty()); } -TEST(Geneval, WildcardPointMatchesShiftedEval) +TEST(Geneval, ArithPointMatchesDealerAtQuery) { using in_t = uint16_t; using out_t = uint16_t; in_t x = 0x1357; in_t x0 = 0x0100; in_t x1 = static_cast(x - x0); - in_t alpha = 0xabcd; in_t query = 0x2000; out_t y = 99; - const in_t delta = static_cast(alpha - x); - const in_t shifted = static_cast(query + delta); reset_roots(); - auto keys = dpf::make_dpf(alpha, dpf::root_sampler_t{take_root}, y); + auto keys = dpf::make_dpf(x, dpf::root_sampler_t{take_root}, y); reset_roots(); - auto g = dpf::geneval_point(dpf::wildcard_input, x0, x1, query, rng(), - [&] { return alpha; }, y); + auto g = dpf::geneval_point(dpf::arith_input, x0, x1, query, rng(), y); using key_t = std::decay_t; const auto live = live_through_lcp( - lcp_bits(leaf_of(alpha), leaf_of(shifted), key_t::depth), + lcp_bits(leaf_of(x), leaf_of(query), key_t::depth), key_t::depth); EXPECT_EQ(g.live_levels, live); EXPECT_LT(live, key_t::depth); expect_prefix_words(keys.first, g.correction_words, g.correction_advice, g.live_levels, g.leaf_live, &g.leaf, sizeof(g.leaf)); - auto e0 = ev(keys.first, shifted); - auto e1 = ev(keys.second, shifted); + auto e0 = ev(keys.first, query); + auto e1 = ev(keys.second, query); EXPECT_EQ(recon(g.party0[0], g.party1[0]), recon(e0, e1)); - - dpf::wildcard_value slot{alpha}; - reset_roots(); - auto wild = dpf::make_dpf(slot, dpf::root_sampler_t{take_root}, y); - auto s0 = wild.first.offset_x.compute_and_get_share(x0); - auto s1 = wild.second.offset_x.compute_and_get_share(x1); - wild.first.offset_x.reconstruct(s1); - wild.second.offset_x.reconstruct(s0); - auto w0 = ev(wild.first, query); - auto w1 = ev(wild.second, query); - EXPECT_EQ(recon(g.party0[0], g.party1[0]), recon(w0, w1)); - expect_prefix_words(wild.first, g.correction_words, g.correction_advice, - g.live_levels, false, nullptr, 0); } -TEST(Geneval, WildcardPointOnSecretIsFullKey) +TEST(Geneval, ArithPointOnSecretIsFullKey) { using in_t = uint16_t; using out_t = uint16_t; in_t x = 0x42; in_t x0 = 0x10; in_t x1 = static_cast(x - x0); - in_t alpha = 0x1111; out_t y = 8; reset_roots(); - auto keys = dpf::make_dpf(alpha, dpf::root_sampler_t{take_root}, y); + auto keys = dpf::make_dpf(x, dpf::root_sampler_t{take_root}, y); reset_roots(); - auto g = dpf::geneval_point(dpf::wildcard_input, x0, x1, x, rng(), - [&] { return alpha; }, y); + auto g = dpf::geneval_point(dpf::arith_input, x0, x1, x, rng(), y); using key_t = std::decay_t; EXPECT_EQ(g.live_levels, key_t::depth); EXPECT_TRUE(g.leaf_live); expect_prefix_words(keys.first, g.correction_words, g.correction_advice, g.live_levels, true, &g.leaf, sizeof(g.leaf)); - auto e0 = ev(keys.first, alpha); - auto e1 = ev(keys.second, alpha); + auto e0 = ev(keys.first, x); + auto e1 = ev(keys.second, x); EXPECT_EQ(g.party0[0], e0); EXPECT_EQ(g.party1[0], e1); EXPECT_EQ(recon(g.party0[0], g.party1[0]), y); } -TEST(Geneval, WildcardIntervalAndSequence) +TEST(Geneval, ArithIntervalAndSequence) { using in_t = uint8_t; using out_t = uint8_t; in_t x = 40; in_t x0 = 7; in_t x1 = static_cast(x - x0); - in_t alpha = 200; out_t y = 3; - const in_t delta = static_cast(alpha - x); reset_roots(); - auto keys = dpf::make_dpf(alpha, dpf::root_sampler_t{take_root}, y); + auto keys = dpf::make_dpf(x, dpf::root_sampler_t{take_root}, y); reset_roots(); - auto iv = dpf::geneval_interval(dpf::wildcard_input, x0, x1, in_t{10}, in_t{20}, - rng(), [&] { return alpha; }, y); + auto iv = dpf::geneval_interval(dpf::arith_input, x0, x1, in_t{10}, in_t{20}, + rng(), y); using key_t = std::decay_t; ASSERT_EQ(iv.party0.size(), 11u); for (in_t q = 10; q <= 20; ++q) { - const in_t shifted = static_cast(q + delta); const std::size_t i = static_cast(q - 10); - auto e0 = ev(keys.first, shifted); - auto e1 = ev(keys.second, shifted); + auto e0 = ev(keys.first, q); + auto e1 = ev(keys.second, q); EXPECT_EQ(recon(iv.party0[i], iv.party1[i]), recon(e0, e1)) << int(q); } - const in_t shifted_from = static_cast(10 + delta); const auto live = live_through_lcp( - lcp_bits(leaf_of(alpha), leaf_of(shifted_from), key_t::depth), + lcp_bits(leaf_of(x), leaf_of(in_t{10}), key_t::depth), key_t::depth); EXPECT_EQ(iv.live_levels, live); expect_prefix_words(keys.first, iv.correction_words, iv.correction_advice, @@ -473,8 +450,8 @@ TEST(Geneval, WildcardIntervalAndSequence) const in_t seq[] = {1, x, 255, 2}; reset_roots(); - auto sq = dpf::geneval_sequence(dpf::wildcard_input, x0, x1, - std::begin(seq), std::end(seq), rng(), [&] { return alpha; }, y); + auto sq = dpf::geneval_sequence(dpf::arith_input, x0, x1, + std::begin(seq), std::end(seq), rng(), y); ASSERT_EQ(sq.party0.size(), 4u); EXPECT_TRUE(sq.leaf_live); EXPECT_EQ(sq.live_levels, key_t::depth); @@ -482,31 +459,27 @@ TEST(Geneval, WildcardIntervalAndSequence) sq.live_levels, true, &sq.leaf, sizeof(sq.leaf)); for (std::size_t i = 0; i < 4; ++i) { - const in_t shifted = static_cast(seq[i] + delta); - auto e0 = ev(keys.first, shifted); - auto e1 = ev(keys.second, shifted); + auto e0 = ev(keys.first, seq[i]); + auto e1 = ev(keys.second, seq[i]); EXPECT_EQ(sq.party0[i], e0); EXPECT_EQ(sq.party1[i], e1); EXPECT_EQ(recon(sq.party0[i], sq.party1[i]), seq[i] == x ? y : out_t{0}); } } -TEST(Geneval, WildcardFullRotates) +TEST(Geneval, ArithFullMatchesDealer) { using in_t = uint8_t; using out_t = uint8_t; in_t x = 40; in_t x0 = 7; in_t x1 = static_cast(x - x0); - in_t alpha = 200; out_t y = 3; - const in_t delta = static_cast(alpha - x); reset_roots(); - auto keys = dpf::make_dpf(alpha, dpf::root_sampler_t{take_root}, y); + auto keys = dpf::make_dpf(x, dpf::root_sampler_t{take_root}, y); reset_roots(); - auto g = dpf::geneval_full(dpf::wildcard_input, x0, x1, rng(), - [&] { return alpha; }, y); + auto g = dpf::geneval_full(dpf::arith_input, x0, x1, rng(), y); using key_t = std::decay_t; EXPECT_EQ(g.party0.size(), 256u); @@ -517,9 +490,8 @@ TEST(Geneval, WildcardFullRotates) EXPECT_EQ(recon(g.party0[x], g.party1[x]), y); for (int q = 0; q < 256; ++q) { - const in_t shifted = static_cast(static_cast(q) + delta); - auto e0 = ev(keys.first, shifted); - auto e1 = ev(keys.second, shifted); + auto e0 = ev(keys.first, static_cast(q)); + auto e1 = ev(keys.second, static_cast(q)); EXPECT_EQ(g.party0[q], e0); EXPECT_EQ(g.party1[q], e1); } @@ -919,7 +891,7 @@ TEST(Geneval, SignedPointIntervalAndCrossZero) (void)near; } -TEST(Geneval, SignedFullAndWildcardFull) +TEST(Geneval, SignedFullAndArithFull) { using in_t = int8_t; using out_t = int8_t; @@ -945,13 +917,11 @@ TEST(Geneval, SignedFullAndWildcardFull) in_t secret = -20; in_t a0 = 100; in_t a1 = static_cast(secret - a0); - in_t target = 40; - const in_t delta = static_cast(target - secret); + ASSERT_EQ(static_cast(a0 + a1), secret); reset_roots(); - auto wkeys = dpf::make_dpf(target, dpf::root_sampler_t{take_root}, y); + auto wkeys = dpf::make_dpf(secret, dpf::root_sampler_t{take_root}, y); reset_roots(); - auto w = dpf::geneval_full(dpf::wildcard_input, a0, a1, rng(), - [&] { return target; }, y); + auto w = dpf::geneval_full(dpf::arith_input, a0, a1, rng(), y); ASSERT_EQ(w.party0.size(), 256u); EXPECT_EQ(w.live_levels, std::decay_t::depth); expect_prefix_words(wkeys.first, w.correction_words, w.correction_advice, @@ -959,16 +929,15 @@ TEST(Geneval, SignedFullAndWildcardFull) for (int q = -128; q <= 127; ++q) { in_t v = static_cast(q); - in_t shifted = static_cast(v + delta); const std::size_t i = static_cast(to_int(v)); - EXPECT_EQ(w.party0[i], ev(wkeys.first, shifted)) << q; - EXPECT_EQ(w.party1[i], ev(wkeys.second, shifted)) << q; + EXPECT_EQ(w.party0[i], ev(wkeys.first, v)) << q; + EXPECT_EQ(w.party1[i], ev(wkeys.second, v)) << q; } EXPECT_EQ(recon(w.party0[static_cast(to_int(secret))], w.party1[static_cast(to_int(secret))]), y); } -TEST(Geneval, WildcardShareOverflowAndWrappingInterval) +TEST(Geneval, ArithShareOverflowAndWrappingInterval) { using in_t = uint8_t; using out_t = uint8_t; @@ -976,15 +945,12 @@ TEST(Geneval, WildcardShareOverflowAndWrappingInterval) in_t x0 = 200; in_t x1 = 66; ASSERT_EQ(static_cast(x0 + x1), secret); - in_t alpha = 5; out_t y = 17; - const in_t delta = static_cast(alpha - secret); reset_roots(); - auto keys = dpf::make_dpf(alpha, dpf::root_sampler_t{take_root}, y); + auto keys = dpf::make_dpf(secret, dpf::root_sampler_t{take_root}, y); reset_roots(); - auto on = dpf::geneval_point(dpf::wildcard_input, x0, x1, secret, rng(), - [&] { return alpha; }, y); + auto on = dpf::geneval_point(dpf::arith_input, x0, x1, secret, rng(), y); using key_t = std::decay_t; EXPECT_EQ(on.live_levels, key_t::depth); EXPECT_TRUE(on.leaf_live); @@ -993,36 +959,33 @@ TEST(Geneval, WildcardShareOverflowAndWrappingInterval) EXPECT_EQ(recon(on.party0[0], on.party1[0]), y); in_t query = 250; - const in_t shifted = static_cast(query + delta); reset_roots(); - auto off = dpf::geneval_point(dpf::wildcard_input, x0, x1, query, rng(), - [&] { return alpha; }, y); + auto off = dpf::geneval_point(dpf::arith_input, x0, x1, query, rng(), y); const auto live = live_through_lcp( - lcp_bits(leaf_of(alpha), leaf_of(shifted), key_t::depth), + lcp_bits(leaf_of(secret), leaf_of(query), key_t::depth), key_t::depth); EXPECT_EQ(off.live_levels, live); expect_prefix_words(keys.first, off.correction_words, off.correction_advice, off.live_levels, off.leaf_live, &off.leaf, sizeof(off.leaf)); EXPECT_EQ(recon(off.party0[0], off.party1[0]), - recon(ev(keys.first, shifted), ev(keys.second, shifted))); + recon(ev(keys.first, query), ev(keys.second, query))); in_t from = 250; in_t to = 10; - EXPECT_THROW((dpf::geneval_interval(dpf::wildcard_input, x0, x1, from, to, - rng(), [&] { return alpha; }, y)), std::invalid_argument); + EXPECT_THROW((dpf::geneval_interval(dpf::arith_input, x0, x1, from, to, + rng(), y)), std::invalid_argument); from = 250; to = 255; reset_roots(); - auto iv = dpf::geneval_interval(dpf::wildcard_input, x0, x1, from, to, - rng(), [&] { return alpha; }, y); + auto iv = dpf::geneval_interval(dpf::arith_input, x0, x1, from, to, + rng(), y); ASSERT_EQ(iv.party0.size(), 6u); for (in_t q = from; ; ++q) { const std::size_t i = static_cast(static_cast(q - from)); - const in_t s = static_cast(q + delta); EXPECT_EQ(recon(iv.party0[i], iv.party1[i]), - recon(ev(keys.first, s), ev(keys.second, s))) << int(q); + recon(ev(keys.first, q), ev(keys.second, q))) << int(q); if (q == to) break; } @@ -1099,26 +1062,19 @@ TEST(Geneval, SignedRegressionsFromTheCornerPass) EXPECT_EQ(full.party0[bit], ev(keys.first, v)) << q; } - // Negative target, additive shares that wrap, bound through a real - // wildcard key so the raw offset bits are what geneval subtracts. + // Negative secret, additive shares that wrap through the signed MSB. in_t secret = -20; in_t a0 = 100; in_t a1 = static_cast(secret - a0); ASSERT_EQ(static_cast(a0 + a1), secret); - in_t target = -90; in_t query = 60; reset_roots(); - dpf::wildcard_value slot{target}; - auto wild = dpf::make_dpf(slot, dpf::root_sampler_t{take_root}, y); - auto s0 = wild.first.offset_x.compute_and_get_share(a0); - auto s1 = wild.second.offset_x.compute_and_get_share(a1); - wild.first.offset_x.reconstruct(s1); - wild.second.offset_x.reconstruct(s0); + auto akeys = dpf::make_dpf(secret, dpf::root_sampler_t{take_root}, y); reset_roots(); - auto g = dpf::geneval_point(dpf::wildcard_input, a0, a1, query, rng(), - [&] { return target; }, y); - EXPECT_EQ(recon(g.party0[0], g.party1[0]), recon(ev(wild.first, query), ev(wild.second, query))); - expect_prefix_words(wild.first, g.correction_words, g.correction_advice, + auto g = dpf::geneval_point(dpf::arith_input, a0, a1, query, rng(), y); + EXPECT_EQ(recon(g.party0[0], g.party1[0]), + recon(ev(akeys.first, query), ev(akeys.second, query))); + expect_prefix_words(akeys.first, g.correction_words, g.correction_advice, g.live_levels, g.leaf_live, &g.leaf, sizeof(g.leaf)); } @@ -1223,3 +1179,211 @@ TEST(Geneval, CmpLtIsTheComplementOfTheStrictUpperSet) EXPECT_EQ(opened, ends[i] < alpha ? 4u : 0u) << int(ends[i]); } } + +TEST(Geneval, DoernerShelatOnTargetSharesMatch) +{ + using in_t = uint16_t; + using out_t = uint16_t; + const in_t alpha = 0x55aa; + const in_t x0 = 0x1234; + const in_t x1 = static_cast(alpha ^ x0); + const out_t y = 0x9f3c; + + reset_roots(); + dpf::ds_randomness ds_rng{take_root, Pad{}}; + auto ds = dpf::make_dpf_doerner_shelat(x0, x1, ds_rng, y); + reset_roots(); + auto g = dpf::geneval_point(x0, x1, alpha, rng(), y); + + using key_t = std::decay_t; + EXPECT_EQ(g.live_levels, key_t::depth); + EXPECT_TRUE(g.leaf_live); + expect_prefix_words(ds.first, g.correction_words, g.correction_advice, + g.live_levels, true, &g.leaf, sizeof(g.leaf)); + EXPECT_EQ(g.party0[0], ev(ds.first, alpha)); + EXPECT_EQ(g.party1[0], ev(ds.second, alpha)); + EXPECT_EQ(recon(g.party0[0], g.party1[0]), y); +} + +TEST(Geneval, WideLiveFrontierMatchesDealer) +{ + using in_t = uint16_t; + using out_t = uint16_t; + const in_t alpha = 0x00ff; + const in_t x0 = 0x0f0f; + const in_t x1 = static_cast(alpha ^ x0); + const out_t y = 0xabcd; + const in_t from = 0; + const in_t to = 0x00ff; + + reset_roots(); + auto keys = dpf::make_dpf(alpha, dpf::root_sampler_t{take_root}, y); + reset_roots(); + auto g = dpf::geneval_interval(x0, x1, from, to, rng(), y); + + using key_t = std::decay_t; + EXPECT_EQ(g.live_levels, key_t::depth); + EXPECT_TRUE(g.leaf_live); + expect_prefix_words(keys.first, g.correction_words, g.correction_advice, + g.live_levels, true, &g.leaf, sizeof(g.leaf)); + ASSERT_EQ(g.party0.size(), static_cast(to - from) + 1); + for (in_t q = from; ; ++q) + { + const std::size_t i = static_cast(q - from); + EXPECT_EQ(g.party0[i], ev(keys.first, q)) << q; + EXPECT_EQ(g.party1[i], ev(keys.second, q)) << q; + const out_t opened = recon(g.party0[i], g.party1[i]); + EXPECT_EQ(opened, q == alpha ? y : out_t{0}) << q; + if (q == to) + break; + } +} + +TEST(Geneval, CmpLeqGeqNonzeroElseAndDomainMin) +{ + using in_t = uint8_t; + const in_t alpha = 10; + const in_t x0 = 3; + const in_t x1 = static_cast(alpha ^ x0); + const std::vector ends{0, 9, 10, 11, 255}; + + auto check = [&](auto spec, auto pred) { + auto spec_ds = spec; + auto spec_g = spec; + reset_roots(); + dpf::ds_randomness ds_rng{take_root, Pad{}}; + auto ds = dpf::make_dpf_doerner_shelat(x0, x1, ds_rng, spec_ds); + reset_roots(); + auto g = dpf::geneval_cmp(x0, x1, ends.begin(), ends.end(), rng(), spec_g); + using key_t = std::decay_t; + EXPECT_EQ(g.live_levels, key_t::depth); + EXPECT_EQ(g.correction_words.size(), key_t::depth); + for (std::size_t level = 0; level < key_t::depth; ++level) + { + EXPECT_EQ(std::memcmp(&g.correction_words[level], + &ds.first.correction_word(level), sizeof(simde__m128i)), 0) << level; + EXPECT_EQ(g.correction_advice[level], ds.first.correction_advice(level)); + } + for (std::size_t i = 0; i < ends.size(); ++i) + { + const uint64_t opened = (g.party0[i] + g.party1[i]) & g.mask; + const uint64_t from_key = + (dpf::eval_point(dpf::cmp, ds.first, ends[i]).raw() + + dpf::eval_point(dpf::cmp, ds.second, ends[i]).raw()) + & ds.first.cmp().mask; + EXPECT_EQ(opened, from_key) << int(ends[i]); + EXPECT_EQ(opened, pred(ends[i])) << int(ends[i]); + } + }; + + check(dpf::leq(uint64_t{5}, uint64_t{2}), [&](in_t e) { + return e <= alpha ? uint64_t{5} : uint64_t{2}; + }); + check(dpf::geq(uint64_t{5}, uint64_t{2}), [&](in_t e) { + return e >= alpha ? uint64_t{5} : uint64_t{2}; + }); + + using wide = int16_t; + const wide amin = std::numeric_limits::min(); + const wide w0 = 1; + const wide w1 = static_cast(amin ^ w0); + const std::vector wends{amin, static_cast(amin + 1), wide{-1}, wide{0}, + std::numeric_limits::max()}; + reset_roots(); + auto g = dpf::geneval_cmp(w0, w1, wends.begin(), wends.end(), rng(), + dpf::gt(uint64_t{3})); + for (std::size_t i = 0; i < wends.size(); ++i) + { + const uint64_t opened = (g.party0[i] + g.party1[i]) & g.mask; + EXPECT_EQ(opened, wends[i] > amin ? 3u : 0u) << wends[i]; + } +} + +TEST(Geneval, ArithSignedMsbAndCarryAcrossPowerOfTwo) +{ + using in_t = int8_t; + using out_t = int8_t; + const in_t secret = -20; + const in_t a0 = 100; + const in_t a1 = static_cast(secret - a0); + ASSERT_EQ(static_cast(a0 + a1), secret); + const out_t y = -7; + + reset_roots(); + auto keys = dpf::make_dpf(secret, dpf::root_sampler_t{take_root}, y); + reset_roots(); + auto g = dpf::geneval_point(dpf::arith_input, a0, a1, secret, rng(), y); + using key_t = std::decay_t; + EXPECT_EQ(g.live_levels, key_t::depth); + EXPECT_TRUE(g.leaf_live); + expect_prefix_words(keys.first, g.correction_words, g.correction_advice, + g.live_levels, true, &g.leaf, sizeof(g.leaf)); + EXPECT_EQ(g.party0[0], ev(keys.first, secret)); + EXPECT_EQ(g.party1[0], ev(keys.second, secret)); + EXPECT_EQ(recon(g.party0[0], g.party1[0]), y); + + // Carry across 2^k: shares that wrap the unsigned modulus. + using u8 = uint8_t; + const u8 usecret = 7; + const u8 x0 = 200; + const u8 x1 = static_cast(usecret - x0); + ASSERT_EQ(static_cast(x0 + x1), usecret); + const u8 uy = 9; + + reset_roots(); + auto ukeys = dpf::make_dpf(usecret, dpf::root_sampler_t{take_root}, uy); + reset_roots(); + auto iv = dpf::geneval_interval(dpf::arith_input, x0, x1, u8{0}, u8{3}, + rng(), uy); + ASSERT_EQ(iv.party0.size(), 4u); + for (u8 q = 0; q <= 3; ++q) + { + const std::size_t i = static_cast(q); + EXPECT_EQ(iv.party0[i], ev(ukeys.first, q)) << int(q); + EXPECT_EQ(iv.party1[i], ev(ukeys.second, q)) << int(q); + EXPECT_EQ(recon(iv.party0[i], iv.party1[i]), + recon(ev(ukeys.first, q), ev(ukeys.second, q))) << int(q); + } + using ukey_t = std::decay_t; + EXPECT_EQ(iv.live_levels, + live_through_lcp( + lcp_bits(leaf_of(usecret), leaf_of(u8{0}), ukey_t::depth), + ukey_t::depth)); +} + +TEST(Geneval, ArithDoernerShelatAndCmpMatchDealer) +{ + using in_t = uint8_t; + const in_t secret = 40; + const in_t a0 = 250; + const in_t a1 = static_cast(secret - a0); + ASSERT_EQ(static_cast(a0 + a1), secret); + const uint64_t beta = 7; + const std::vector ends{0, 1, 10, 40, 200, 255}; + + reset_roots(); + auto dealer = dpf::make_dpf(secret, dpf::root_sampler_t{take_root}, + dpf::gt(beta)); + reset_roots(); + auto ds = dpf::make_dpf_doerner_shelat(dpf::arith_input, a0, a1, rng(), + dpf::gt(beta)); + using key_t = std::decay_t; + for (std::size_t level = 0; level < key_t::depth; ++level) + { + EXPECT_EQ(std::memcmp(&ds.first.correction_word(level), + &dealer.first.correction_word(level), sizeof(simde__m128i)), 0) << level; + EXPECT_EQ(ds.first.correction_advice(level), + dealer.first.correction_advice(level)) << level; + EXPECT_EQ(ds.first.value_cw(level), dealer.first.value_cw(level)) << level; + } + + reset_roots(); + auto g = dpf::geneval_cmp(dpf::arith_input, a0, a1, ends.begin(), ends.end(), + rng(), beta); + EXPECT_EQ(g.live_levels, key_t::depth); + for (std::size_t i = 0; i < ends.size(); ++i) + { + const uint64_t opened = (g.party0[i] + g.party1[i]) & g.mask; + EXPECT_EQ(opened, ends[i] > secret ? beta : 0u) << int(ends[i]); + } +} diff --git a/test/tests/ic_test.cpp b/test/tests/ic_test.cpp new file mode 100644 index 0000000..fd37bce --- /dev/null +++ b/test/tests/ic_test.cpp @@ -0,0 +1,224 @@ +#include + +#include "dpf.hpp" + +#include +#include +#include + +namespace +{ + +uint64_t oracle(uint64_t x, uint64_t r, uint64_t p, uint64_t q, + uint64_t nmask, uint64_t if_true, uint64_t if_false, uint64_t gmask) +{ + const uint64_t w = (x - r) & nmask; + const bool inside = w >= p && w <= q; + return (inside ? if_true : if_false) & gmask; +} + +template +void expect_domain(Input r, Input p, Input q, Beta if_true, Beta if_false, + uint64_t gmask) +{ + auto keys = dpf::make_dpf(r, dpf::ic(p, q, if_true, if_false)); + const uint64_t nmask = keys.first.input_mask; + const uint64_t rb = static_cast(r); + const uint64_t pb = static_cast(p); + const uint64_t qb = static_cast(q); + for (uint64_t x = 0; x <= nmask; ++x) + { + const auto y0 = dpf::eval_point(dpf::ic, keys.first, + static_cast(x)); + const auto y1 = dpf::eval_point(dpf::ic, keys.second, + static_cast(x)); + const uint64_t got = static_cast(dpf::reconstruct(y0, y1)) & gmask; + const uint64_t want = oracle(x, rb, pb, qb, nmask, + static_cast(if_true), static_cast(if_false), gmask); + ASSERT_EQ(got, want) << "r=" << rb << " x=" << x + << " p=" << pb << " q=" << qb; + } +} + +} // namespace + +TEST(Ic, Uint8FullDomainCorners) +{ + const uint32_t betas[] = {1u, 7u, 255u}; + const uint32_t falses[] = {0u, 9u}; + const uint8_t intervals[][2] = { + {0, 0}, {0, 255}, {5, 5}, {1, 20}, {200, 250}, {0, 1}, {254, 255}, {10, 40} + }; + for (uint32_t beta : betas) + { + for (uint32_t f : falses) + { + for (const auto & iv : intervals) + { + for (int r = 0; r < 256; r += 17) + { + expect_domain(static_cast(r), iv[0], iv[1], + beta, f, 0xffffffffu); + } + } + } + } +} + +TEST(Ic, Uint8AllMasksOneInterval) +{ + expect_domain(uint8_t{0}, uint8_t{10}, uint8_t{20}, + uint32_t{3}, uint32_t{0}, 0xffffffffu); + expect_domain(uint8_t{200}, uint8_t{10}, uint8_t{100}, + uint32_t{3}, uint32_t{1}, 0xffffffffu); + expect_domain(uint8_t{255}, uint8_t{0}, uint8_t{255}, + uint32_t{1}, uint32_t{0}, 0xffffffffu); +} + +TEST(Ic, MemoizerAgrees) +{ + const uint8_t r = 40, p = 7, q = 90; + auto keys = dpf::make_dpf(r, dpf::ic(p, q, uint32_t{11}, uint32_t{2})); + dpf::basic_path_memoizer memo0; + dpf::basic_path_memoizer memo1; + for (int x = 0; x < 256; ++x) + { + const auto a = dpf::eval_point(dpf::ic, keys.first, static_cast(x), memo0); + const auto b = dpf::eval_point(dpf::ic, keys.second, static_cast(x), memo1); + const auto c = dpf::eval_point(dpf::ic, keys.first, static_cast(x)); + const auto d = dpf::eval_point(dpf::ic, keys.second, static_cast(x)); + EXPECT_EQ(dpf::reconstruct(a, b), dpf::reconstruct(c, d)); + } +} + +TEST(Ic, IntervalAndSequenceBuffers) +{ + const uint8_t r = 15, p = 4, q = 12; + auto keys = dpf::make_dpf(r, dpf::ic(p, q, uint16_t{9})); + auto buf0 = dpf::make_output_buffer(dpf::ic, keys.first, uint8_t{3}, uint8_t{18}); + auto buf1 = dpf::make_output_buffer(dpf::ic, keys.second, uint8_t{3}, uint8_t{18}); + dpf::basic_path_memoizer memo; + dpf::eval_interval(dpf::ic, keys.first, uint8_t{3}, uint8_t{18}, buf0, memo); + dpf::eval_interval(dpf::ic, keys.second, uint8_t{3}, uint8_t{18}, buf1); + for (std::size_t i = 0; i < buf0.size(); ++i) + { + const auto point = dpf::reconstruct( + dpf::eval_point(dpf::ic, keys.first, static_cast(3 + i)), + dpf::eval_point(dpf::ic, keys.second, static_cast(3 + i))); + EXPECT_EQ(dpf::reconstruct(buf0[i], buf1[i]), point); + } + + const uint8_t pts[] = {0, 9, 15, 255, 4}; + auto s0 = dpf::make_output_buffer(dpf::ic, keys.first, 5); + auto s1 = dpf::make_output_buffer(dpf::ic, keys.second, 5); + dpf::eval_sequence(dpf::ic, keys.first, std::begin(pts), std::end(pts), s0); + dpf::eval_sequence(dpf::ic, keys.second, std::begin(pts), std::end(pts), s1); + for (std::size_t i = 0; i < 5; ++i) + { + const auto point = dpf::reconstruct( + dpf::eval_point(dpf::ic, keys.first, pts[i]), + dpf::eval_point(dpf::ic, keys.second, pts[i])); + EXPECT_EQ(dpf::reconstruct(s0[i], s1[i]), point); + } +} + +TEST(Ic, WildcardAssign) +{ + auto keys = dpf::make_dpf(uint8_t{33}, + dpf::ic(uint8_t{2}, uint8_t{8}, dpf::wildcard)); + EXPECT_THROW(dpf::eval_point(dpf::ic, keys.first, uint8_t{0}), std::invalid_argument); + dpf::assign_cmp(keys.first, keys.second, uint32_t{6}, uint32_t{1}); + expect_domain(uint8_t{33}, uint8_t{2}, uint8_t{8}, + uint32_t{6}, uint32_t{1}, 0xffffffffu); + // The keys just assigned are a different generation; check those directly. + for (int x = 0; x < 256; ++x) + { + const uint64_t got = static_cast(dpf::reconstruct( + dpf::eval_point(dpf::ic, keys.first, static_cast(x)), + dpf::eval_point(dpf::ic, keys.second, static_cast(x)))); + const uint64_t w = static_cast(static_cast(x - 33)); + const uint64_t want = (w >= 2 && w <= 8) ? 6u : 1u; + EXPECT_EQ(got, want) << x; + } +} + +TEST(Ic, DoernerShelatMatchesDealer) +{ + std::mt19937 rng{7}; + std::uniform_int_distribution d(0, 255); + for (int n = 0; n < 30; ++n) + { + const uint8_t r0 = static_cast(d(rng)); + const uint8_t r1 = static_cast(d(rng)); + const uint8_t p = static_cast(d(rng)); + const uint8_t q = static_cast(p + static_cast(d(rng) % (256 - p))); + const uint32_t beta = 1u + static_cast(d(rng)); + const uint8_t r = static_cast(r0 ^ r1); + auto dealer = dpf::make_dpf(r, dpf::ic(p, q, beta)); + struct Pad + { + simde__m128i block() { return dpf::uniform_sample(); } + uint8_t bit() { return static_cast(dpf::uniform_sample() & 1u); } + }; + dpf::ds_randomness), Pad> rngs{ + &dpf::uniform_sample, {}}; + auto ds = dpf::make_dpf_doerner_shelat(r0, r1, rngs, dpf::ic(p, q, beta)); + for (int x = 0; x < 256; x += 5) + { + const auto dealer_y = dpf::reconstruct( + dpf::eval_point(dpf::ic, dealer.first, static_cast(x)), + dpf::eval_point(dpf::ic, dealer.second, static_cast(x))); + const auto ds_y = dpf::reconstruct( + dpf::eval_point(dpf::ic, ds.first, static_cast(x)), + dpf::eval_point(dpf::ic, ds.second, static_cast(x))); + EXPECT_EQ(dealer_y, ds_y) << "x=" << x; + } + } +} + +TEST(Ic, Geneval) +{ + struct Pad + { + simde__m128i block() { return dpf::uniform_sample(); } + uint8_t bit() { return static_cast(dpf::uniform_sample() & 1u); } + }; + const uint8_t r0 = 9, r1 = 100, p = 3, q = 50; + const uint32_t beta = 4; + const uint8_t queries[] = {0, 3, 12, 49, 50, 51, 255}; + dpf::ds_randomness), Pad> rngs{ + &dpf::uniform_sample, {}}; + auto opened = dpf::geneval_ic(r0, r1, std::begin(queries), std::end(queries), + rngs, dpf::ic(p, q, beta)); + ASSERT_EQ(opened.party0.size(), 7u); + ASSERT_EQ(opened.live_levels, 8u); + const uint8_t r = static_cast(r0 ^ r1); + for (std::size_t i = 0; i < 7; ++i) + { + const uint64_t got = (opened.party0[i] + opened.party1[i]) & 0xffffffffu; + const uint64_t w = static_cast( + static_cast(queries[i] - r)); + const uint64_t want = (w >= p && w <= q) ? beta : 0u; + EXPECT_EQ(got, want) << i; + } +} + +TEST(Ic, RejectsWrappedBounds) +{ + EXPECT_THROW(dpf::make_dpf(uint8_t{1}, dpf::ic(uint8_t{9}, uint8_t{2}, uint32_t{1})), + std::invalid_argument); +} + +TEST(Ic, BitPayload) +{ + auto keys = dpf::make_dpf(uint8_t{4}, + dpf::ic(uint8_t{1}, uint8_t{3}, dpf::bit::one, dpf::bit::zero)); + for (int x = 0; x < 256; ++x) + { + const auto y = dpf::reconstruct( + dpf::eval_point(dpf::ic, keys.first, static_cast(x)), + dpf::eval_point(dpf::ic, keys.second, static_cast(x))); + const uint64_t w = static_cast(static_cast(x - 4)); + EXPECT_EQ(static_cast(y), w >= 1 && w <= 3) << x; + } +} diff --git a/test/tests/lane_blast_test.cpp b/test/tests/lane_blast_test.cpp index e9110ec..b572cae 100644 --- a/test/tests/lane_blast_test.cpp +++ b/test/tests/lane_blast_test.cpp @@ -448,7 +448,7 @@ TEST(LaneBlast, GenevalMatchesKeyOnPackedLanes) EXPECT_EQ(opened(sq.party0[i], sq.party1[i]), seq[i] == alpha ? y : out_t{}); } -TEST(LaneBlast, GenevalNybleWildcardAndSigned) +TEST(LaneBlast, GenevalNybleArithAndSigned) { using out_t = dpf::nyble; const out_t y{0x0c}; @@ -459,20 +459,19 @@ TEST(LaneBlast, GenevalNybleWildcardAndSigned) const in_t a0 = 200; const in_t a1 = 66; ASSERT_EQ(static_cast(a0 + a1), secret); - const in_t target = 5; reset_roots(); - auto keys = dpf::make_dpf(target, + auto keys = dpf::make_dpf(secret, dpf::root_sampler_t{take_root}, y); reset_roots(); - auto g = dpf::geneval_point(dpf::wildcard_input, a0, a1, secret, - dpf::ds_randomness{take_root, Pad{}}, [&] { return target; }, y); + auto g = dpf::geneval_point(dpf::arith_input, a0, a1, secret, + dpf::ds_randomness{take_root, Pad{}}, y); EXPECT_TRUE(g.leaf_live); expect_live_words(keys.first, g); EXPECT_EQ(opened(g.party0[0], g.party1[0]), y); reset_roots(); - auto miss = dpf::geneval_point(dpf::wildcard_input, a0, a1, in_t{250}, - dpf::ds_randomness{take_root, Pad{}}, [&] { return target; }, y); + auto miss = dpf::geneval_point(dpf::arith_input, a0, a1, in_t{250}, + dpf::ds_randomness{take_root, Pad{}}, y); EXPECT_EQ(opened(miss.party0[0], miss.party1[0]), out_t{}); expect_live_words(keys.first, miss); } diff --git a/test/tests/nmod_test.cpp b/test/tests/nmod_test.cpp new file mode 100644 index 0000000..8961c94 --- /dev/null +++ b/test/tests/nmod_test.cpp @@ -0,0 +1,176 @@ +#include + +#include "grotto/nmod.hpp" + +#include +#include + +using u128 = unsigned __int128; + +namespace +{ + +grotto::nmod_result oracle(std::int64_t x_raw, unsigned x_bits, std::uint64_t recip, + unsigned recip_bits, unsigned residue_bits) +{ + const unsigned scale = x_bits + recip_bits; + const __int128 prod = static_cast<__int128>(x_raw) * static_cast<__int128>(recip); + const bool neg = prod < 0; + const auto mag = static_cast(neg ? -prod : prod); + const u128 mask = scale >= 128 ? ~u128{0} : (u128{1} << scale) - 1; + const u128 quot_mag = scale >= 128 ? 0 : mag >> scale; + const u128 rem = scale >= 128 ? mag : mag & mask; + grotto::nmod_result out; + if (!neg) + out.quotient = static_cast(quot_mag); + else if (rem == 0) + out.quotient = -static_cast(quot_mag); + else + out.quotient = -static_cast(quot_mag) - 1; + if (residue_bits == 0 || rem == 0) + return out; + u128 field = rem; + if (neg) + field = (u128{1} << scale) - rem; + if (scale >= residue_bits) + field >>= scale - residue_bits; + else + field <<= residue_bits - scale; + const u128 unit = u128{1} << residue_bits; + out.residue = static_cast(field & (unit - 1)); + return out; +} + +} // namespace + +TEST(Nmod, SplitsAnIntegerModulusOnTheFractionalBoundary) +{ + // 3.25 = 13/4, modulo 1. + const auto split = grotto::nmod(13, 2, 1, 0, 2); + EXPECT_EQ(split.quotient, 3); + EXPECT_EQ(split.residue, 1); + + // -1.25 = -5/4. floor is -2 and the residue is 0.75. + const auto neg = grotto::nmod(-5, 2, 1, 0, 2); + EXPECT_EQ(neg.quotient, -2); + EXPECT_EQ(neg.residue, 3); + + // Coarser residue truncates toward -infinity: floor(0.75 * 2) = 1. + EXPECT_EQ(grotto::nmod(-5, 2, 1, 0, 1).residue, 1); +} + +TEST(Nmod, ExactNegativeIntegersHaveAZeroResidue) +{ + const auto split = grotto::nmod(-8, 2, 1, 0, 2); + EXPECT_EQ(split.quotient, -2); + EXPECT_EQ(split.residue, 0); + EXPECT_EQ(grotto::nmod(0, 8, 1, 0, 8).quotient, 0); + EXPECT_EQ(grotto::nmod(0, 8, 1, 0, 8).residue, 0); +} + +TEST(Nmod, FloorDoesNotRoundUpToTheNextQuotient) +{ + // 1 - 2^{-16}. A round-to-nearest reciprocal product would become 1. + const auto split = grotto::nmod(1, 0, (std::uint64_t{1} << 16) - 1, 16, 8); + EXPECT_EQ(split.quotient, 0); + EXPECT_EQ(split.residue, 255); +} + +TEST(Nmod, PowerOfTwoModulusIsAnExactShift) +{ + // 1.5 / 2^{-1} = 3 exactly. + const auto half = grotto::nmod_pow2(6, 2, 1, 2); + EXPECT_EQ(half.quotient, 3); + EXPECT_EQ(half.residue, 0); + + // 1.5 / 2 = 0.75. + const auto two = grotto::nmod_pow2(6, 2, -1, 2); + EXPECT_EQ(two.quotient, 0); + EXPECT_EQ(two.residue, 3); + + // 2^x splits at the integer. + const auto unit = grotto::nmod_pow2(6, 2, 0, 2); + EXPECT_EQ(unit.quotient, 1); + EXPECT_EQ(unit.residue, 2); + + const auto via_recip = grotto::nmod(6, 2, 2, 0, 2); + EXPECT_EQ(half.quotient, via_recip.quotient); + EXPECT_EQ(half.residue, via_recip.residue); +} + +TEST(Nmod, IntegerReciprocalAgreesWithFloorDivision) +{ + // round(2^100 / 3) == floor(2^100 / 3) for this width. + const u128 recip = (u128{1} << 100) / 3; + const auto split = grotto::nmod(10, 0, recip, 100, 16); + EXPECT_EQ(split.quotient, 3); + EXPECT_EQ(split.residue, 21845); +} + +TEST(Nmod, WideQuarterTurnReciprocal) +{ + // RN(4/π · 2^80). x = 2.5 at 8 fractional bits, residue at 16 bits. + const u128 recip = (u128{83443} << 64) | 494442167743545356ULL; + const auto split = grotto::nmod(640, 8, recip, 80, 16); + EXPECT_EQ(split.quotient, 3); + EXPECT_EQ(split.residue, 11999); +} + +TEST(Nmod, MatchesFloorOnRandomReciprocals) +{ + std::mt19937 rng(0x6d6f64u); + std::uniform_int_distribution values(-4000, 4000); + std::uniform_int_distribution bits(0, 20); + std::uniform_int_distribution recip_dist(1, 100000); + for (int i = 0; i < 4000; ++i) + { + const unsigned x_bits = static_cast(bits(rng)); + const unsigned recip_bits = static_cast(bits(rng)); + const unsigned residue_bits = static_cast(bits(rng) % 16); + const std::int64_t x_raw = values(rng); + const std::uint64_t recip = recip_dist(rng); + const auto got = grotto::nmod(x_raw, x_bits, recip, recip_bits, residue_bits); + const auto want = oracle(x_raw, x_bits, recip, recip_bits, residue_bits); + EXPECT_EQ(got.quotient, want.quotient) << i; + EXPECT_EQ(got.residue, want.residue) << i; + if (got.quotient != want.quotient || got.residue != want.residue) + break; + } +} + +TEST(Nmod, PowerOfTwoAgreesWithTheGeneralSplit) +{ + std::mt19937 rng(13); + std::uniform_int_distribution values(-2000, 2000); + std::uniform_int_distribution exps(-12, 12); + for (int i = 0; i < 500; ++i) + { + const int exp = exps(rng); + const unsigned residue_bits = static_cast(i % 10); + const std::int64_t x_raw = values(rng); + const auto got = grotto::nmod_pow2(x_raw, 8, exp, residue_bits); + grotto::nmod_result want; + if (exp >= 0) + want = grotto::nmod(x_raw, 8, std::uint64_t{1} << exp, 0, residue_bits); + else + want = grotto::nmod(x_raw, 8, 1, static_cast(-exp), residue_bits); + EXPECT_EQ(got.quotient, want.quotient); + EXPECT_EQ(got.residue, want.residue); + } +} + +TEST(Nmod, RejectsAZeroReciprocalAndAHugeQuotient) +{ + EXPECT_THROW(grotto::nmod(1, 0, 0, 0, 4), std::invalid_argument); + EXPECT_THROW(grotto::nmod(1, 0, 1, 0, 64), std::invalid_argument); + EXPECT_THROW(grotto::nmod_pow2(1, 0, 128, 4), std::overflow_error); + EXPECT_THROW(grotto::nmod(INT64_MAX, 0, u128{1} << 80, 0, 4), + std::overflow_error); +} + +TEST(Nmod, MostNegativeInputModuloOne) +{ + const auto split = grotto::nmod(INT64_MIN, 0, 1, 0, 4); + EXPECT_EQ(split.quotient, INT64_MIN); + EXPECT_EQ(split.residue, 0); +} diff --git a/test/tests/offset_horner_test.cpp b/test/tests/offset_horner_test.cpp index 60dca09..09712df 100644 --- a/test/tests/offset_horner_test.cpp +++ b/test/tests/offset_horner_test.cpp @@ -835,6 +835,33 @@ TEST(OffsetHorner, HornerOfOpenedCoefficientsMatchesValue) EXPECT_EQ(y, gold(center, eta, knots, coeff)); } +TEST(OffsetHorner, OpenedSharesAreNotHornerInputs) +{ + constexpr std::size_t D = 2; + const std::vector knots{0, 50, 150}; + const auto coeff = take_degree(pad3({ + {1, 0, 0, 0}, + {0, 4, 1, 0}, + {8, 0, 0, 0}, + })); + const uint8_t center = 10; + const uint8_t eta = 60; + auto mat = grotto::make_offset_horner_keys(center); + const auto c = open_coeffs(mat, knots, coeff, eta); + uint64_t sum = 0; + for (uint64_t ck : c) + sum += ck; + const uint64_t value = gold(center, eta, knots, coeff); + EXPECT_EQ(sum, value); + EXPECT_EQ(value, 5180u); + + uint64_t horner = c[D]; + const uint64_t limb = lift(center); + for (std::size_t k = D; k-- > 0; ) + horner = horner * limb + c[k]; + EXPECT_NE(horner, value); +} + template void expect_wrapped(const grotto::offset_horner_keys & mat, const std::vector & knots, diff --git a/test/tests/path_recipe_test.cpp b/test/tests/path_recipe_test.cpp new file mode 100644 index 0000000..e55a8c0 --- /dev/null +++ b/test/tests/path_recipe_test.cpp @@ -0,0 +1,223 @@ +#include + +#include "dpf.hpp" + +#include +#include + +namespace +{ + +template +uint64_t recon_cmp(const A & a, const B & b, Target target, uint8_t x) +{ + return dpf::reconstruct(dpf::eval_point(target, a, x), + dpf::eval_point(target, b, x)); +} + +uint64_t lcp_len(uint8_t x, uint8_t alpha, std::size_t n = 8, std::size_t width = 8) +{ + for (std::size_t i = 0; i < n; ++i) + { + const uint8_t shift = static_cast(width - 1 - i); + if (((x >> shift) & 1) != ((alpha >> shift) & 1)) + return i; + } + return n; +} + +uint64_t high_prefix(uint8_t alpha, uint64_t d, std::size_t n = 8) +{ + if (d == 0) + return 0; + if (d >= n) + return alpha & ((1u << n) - 1u); + const unsigned drop = static_cast(n - d); + return (static_cast(alpha) >> drop) << drop; +} + +uint64_t low_prefix(uint8_t alpha, uint64_t d, std::size_t n = 8) +{ + if (d == 0) + return 0; + if (d >= n) + return alpha; + return static_cast(alpha) >> (n - d); +} + +template +void expect_domain(Make make, Unit unit_of) +{ + constexpr uint8_t alpha = 0xB4; + auto [k0, k1] = make(alpha); + for (int x = 0; x < 256; ++x) + { + const auto got = recon_cmp(k0, k1, dpf::cmp, static_cast(x)); + EXPECT_EQ(got, unit_of(static_cast(x), alpha)) + << "x=" << x; + } +} + +} // namespace + +TEST(PathRecipe, LengthMaskPrefixBreakAndPacked) +{ + constexpr uint8_t alpha = 0xB4; + expect_domain( + [](uint8_t a) { return dpf::make_dpf(a, dpf::lcp(uint64_t{1})); }, + [](uint8_t x, uint8_t a) { return lcp_len(x, a); }); + expect_domain( + [](uint8_t a) { return dpf::make_dpf(a, dpf::common_prefix(uint64_t{1})); }, + [](uint8_t x, uint8_t a) { return high_prefix(a, lcp_len(x, a)); }); + expect_domain( + [](uint8_t a) { return dpf::make_dpf(a, dpf::prefix_mask(uint64_t{1})); }, + [](uint8_t x, uint8_t a) { return high_prefix(0xFF, lcp_len(x, a)); }); + expect_domain( + [](uint8_t a) { return dpf::make_dpf(a, dpf::diverge_one_hot(uint64_t{1})); }, + [](uint8_t x, uint8_t a) { return 1ULL << lcp_len(x, a); }); + expect_domain( + [](uint8_t a) { return dpf::make_dpf(a, dpf::break_bit(uint64_t{3})); }, + [](uint8_t x, uint8_t a) { + const auto d = lcp_len(x, a); + if (d >= 8) + return 0ULL; + return 3ULL * ((a >> (7 - d)) & 1); + }); + expect_domain( + [](uint8_t a) { + return dpf::make_dpf(a, dpf::prefix_with_length<4>(uint64_t{1})); + }, + [](uint8_t x, uint8_t a) { + const auto d = lcp_len(x, a); + return (low_prefix(a, d) << 4) | d; + }); + expect_domain( + [](uint8_t a) { + return dpf::make_dpf(a, dpf::lcp(uint64_t{5}, uint64_t{2})); + }, + [](uint8_t x, uint8_t a) { return 2ULL + 3ULL * lcp_len(x, a); }); + expect_domain( + [](uint8_t a) { + return dpf::make_dpf(a, dpf::path_paint( + [](std::size_t matched, uint64_t, bool) { + return static_cast(matched); + })); + }, + [](uint8_t x, uint8_t a) { return lcp_len(x, a); }); + + auto [p0, p1] = dpf::make_dpf(alpha, dpf::lcp_at<4>(uint64_t{1})); + for (int x = 0; x < 256; ++x) + { + const auto got = recon_cmp(p0, p1, dpf::cmp, static_cast(x)); + EXPECT_EQ(got, lcp_len(static_cast(x), alpha, 4, 8)) << x; + } +} + +TEST(PathRecipe, WildcardScaleAssignsLength) +{ + constexpr uint8_t alpha = 0x3C; + auto [k0, k1] = dpf::make_dpf(alpha, dpf::lcp(dpf::wildcard)); + dpf::assign_cmp(k0, k1, uint64_t{4}); + for (int x = 0; x < 256; ++x) + { + EXPECT_EQ(recon_cmp(k0, k1, dpf::cmp, static_cast(x)), + 4ULL * lcp_len(static_cast(x), alpha)) + << x; + } +} + +TEST(PathRecipe, IdpfSlotsArePrefixPointFunctions) +{ + constexpr uint8_t alpha = 0xA6; + auto [k0, k1] = dpf::make_dpf(alpha, + dpf::idpf(uint64_t{11}, uint64_t{22}, uint64_t{33})); + auto slot = [&](auto target, uint8_t x) { + return dpf::reconstruct(*dpf::eval_point(target, k0, x), + *dpf::eval_point(target, k1, x)); + }; + for (int x = 0; x < 256; ++x) + { + const auto d = lcp_len(static_cast(x), alpha); + EXPECT_EQ(slot(dpf::out<0>, static_cast(x)), d >= 1 ? 11u : 0u); + EXPECT_EQ(slot(dpf::out<1>, static_cast(x)), d >= 2 ? 22u : 0u); + EXPECT_EQ(slot(dpf::out<2>, static_cast(x)), d >= 3 ? 33u : 0u); + } + + auto [s0, s1] = dpf::make_dpf(alpha, dpf::idpf_at<4, 7>(uint8_t{9}, uint8_t{8})); + for (int x = 0; x < 256; ++x) + { + const auto d = lcp_len(static_cast(x), alpha); + auto at = [&](auto target) { + return dpf::reconstruct(*dpf::eval_point(target, s0, static_cast(x)), + *dpf::eval_point(target, s1, static_cast(x))); + }; + EXPECT_EQ(at(dpf::out<0>), d >= 4 ? 9u : 0u); + EXPECT_EQ(at(dpf::out<1>), d >= 7 ? 8u : 0u); + } +} + +TEST(PathRecipe, IdcfMatchesComparisonAtEveryPrefix) +{ + constexpr uint8_t alpha = 0x6E; + auto check = [&](auto idcf_spec, auto at_spec, auto full_spec, std::size_t L, + auto target) { + auto [i0, i1] = dpf::make_dpf(alpha, idcf_spec); + auto [n0, n1] = dpf::make_dpf(alpha, at_spec); + auto [f0, f1] = dpf::make_dpf(alpha, full_spec); + for (int x = 0; x < 256; ++x) + { + const auto q = static_cast(x); + EXPECT_EQ(recon_cmp(i0, i1, target, q), recon_cmp(n0, n1, dpf::cmp, q)) + << "L=" << L << " x=" << x; + EXPECT_EQ(recon_cmp(i0, i1, dpf::cmp, q), recon_cmp(f0, f1, dpf::cmp, q)) + << "full x=" << x; + } + EXPECT_EQ(i0.prefix_cw(8), i0.cw_last()); + }; + check(dpf::idcf(dpf::lt(uint64_t{1})), dpf::lt_at<4>(uint64_t{1}), + dpf::lt(uint64_t{1}), 4, dpf::cmp_prefix<4>); + check(dpf::idcf(dpf::leq(uint64_t{1})), dpf::leq_at<3>(uint64_t{1}), + dpf::leq(uint64_t{1}), 3, dpf::cmp_prefix<3>); + check(dpf::idcf(dpf::gt(uint64_t{1})), dpf::gt_at<5>(uint64_t{1}), + dpf::gt(uint64_t{1}), 5, dpf::cmp_prefix<5>); + check(dpf::idcf(dpf::geq(uint64_t{1})), dpf::geq_at<1>(uint64_t{1}), + dpf::geq(uint64_t{1}), 1, dpf::cmp_prefix<1>); + + auto [z0, z1] = dpf::make_dpf(alpha, dpf::idcf(dpf::lt(uint64_t{1}))); + auto [e0, e1] = dpf::make_dpf(alpha, dpf::idcf(dpf::leq(uint64_t{1}))); + auto [g0, g1] = dpf::make_dpf(alpha, dpf::idcf(dpf::gt(uint64_t{1}))); + auto [q0, q1] = dpf::make_dpf(alpha, dpf::idcf(dpf::geq(uint64_t{1}))); + for (int x = 0; x < 256; ++x) + { + const auto q = static_cast(x); + EXPECT_EQ(recon_cmp(z0, z1, dpf::cmp_prefix<0>, q), 0u); + EXPECT_EQ(recon_cmp(e0, e1, dpf::cmp_prefix<0>, q), 1u); + EXPECT_EQ(recon_cmp(g0, g1, dpf::cmp_prefix<0>, q), 0u); + EXPECT_EQ(recon_cmp(q0, q1, dpf::cmp_prefix<0>, q), 1u); + } +} + +TEST(PathRecipe, DoernerShelatAndGenevalMatchLength) +{ + constexpr uint8_t alpha = 0x91; + const uint8_t x0 = 0x10; + const uint8_t x1 = static_cast(alpha ^ x0); + auto [k0, k1] = dpf::make_dpf_doerner_shelat(x0, x1, dpf::lcp(uint64_t{1})); + for (int x = 0; x < 256; ++x) + { + EXPECT_EQ(recon_cmp(k0, k1, dpf::cmp, static_cast(x)), + lcp_len(static_cast(x), alpha)); + } + + dpf::ds_randomness rng{ + dpf::uniform_sample, {}}; + const uint8_t ends[] = {0x00, 0x91, 0xFF}; + auto g = dpf::geneval_cmp(x0, x1, std::begin(ends), std::end(ends), rng, + dpf::lcp(uint64_t{1})); + ASSERT_EQ(g.party0.size(), 3u); + for (std::size_t i = 0; i < 3; ++i) + { + EXPECT_EQ((g.party0[i] + g.party1[i]) & g.mask, + lcp_len(ends[i], alpha)); + } +} diff --git a/test/tests/prg_chacha_test.cpp b/test/tests/prg_chacha_test.cpp new file mode 100644 index 0000000..7562097 --- /dev/null +++ b/test/tests/prg_chacha_test.cpp @@ -0,0 +1,245 @@ +#include + +#include +#include +#include +#include + +#include "dpf.hpp" +#include "simde/simde/x86/avx2.h" + +namespace +{ + +bool blocks_equal(simde__m128i a, simde__m128i b) +{ + return simde_mm_movemask_epi8(simde_mm_cmpeq_epi8(a, b)) == 0xFFFF; +} + +simde__m128i block_from_lanes(std::uint64_t lo, std::uint64_t hi) +{ + simde__m128i x; + std::uint64_t lane[2] = {lo, hi}; + std::memcpy(&x, lane, sizeof(x)); + return x; +} + +simde__m128i block_from_bytes(const std::uint8_t * p) +{ + simde__m128i x; + std::memcpy(&x, p, sizeof(x)); + return x; +} + +// RFC 8439 §2.3.2. Counter = 1, nonce = 00:00:00:09:00:00:00:4a:00:00:00:00. +constexpr std::uint8_t k_rfc_keystream[64] = { + 0x10, 0xf1, 0xe7, 0xe4, 0xd1, 0x3b, 0x59, 0x15, + 0x50, 0x0f, 0xdd, 0x1f, 0xa3, 0x20, 0x71, 0xc4, + 0xc7, 0xd1, 0xf4, 0xc7, 0x33, 0xc0, 0x68, 0x03, + 0x04, 0x22, 0xaa, 0x9a, 0xc3, 0xd4, 0x6c, 0x4e, + 0xd2, 0x82, 0x64, 0x46, 0x07, 0x9f, 0xaa, 0x09, + 0x14, 0xc2, 0xd7, 0x05, 0xd9, 0x8b, 0x02, 0xa2, + 0xb5, 0x12, 0x9c, 0xd1, 0xde, 0x16, 0x4e, 0xb9, + 0xcb, 0xd0, 0x83, 0xe8, 0xa2, 0x50, 0x3c, 0x4e +}; + +} // namespace + +TEST(ChachaPrg, Rfc8439Block) +{ + const std::uint32_t key[8] = { + 0x03020100u, 0x07060504u, 0x0b0a0908u, 0x0f0e0d0cu, + 0x13121110u, 0x17161514u, 0x1b1a1918u, 0x1f1e1d1cu + }; + const std::uint32_t nonce[3] = {0x09000000u, 0x4a000000u, 0x00000000u}; + std::uint8_t out[64]; + dpf::prg::chacha_detail::block<20>(key, 1, nonce, out); + EXPECT_EQ(0, std::memcmp(out, k_rfc_keystream, sizeof(out))); + + std::uint8_t fewer[64]; + dpf::prg::chacha_detail::block<8>(key, 1, nonce, fewer); + EXPECT_NE(0, std::memcmp(fewer, k_rfc_keystream, sizeof(fewer))); +} + +TEST(ChachaPrg, WideBlockMatchesScalar) +{ + std::uint32_t keys[4][8]; + std::uint32_t counters[4] = {0u, 1u, 5u, 0x00fffff0u}; + for (int lane = 0; lane < 4; ++lane) + { + for (int w = 0; w < 8; ++w) + { + keys[lane][w] = 0x9e3779b9u * static_cast(lane + 1) + + static_cast(w) * 0x01000193u; + } + } + std::uint8_t wide[4][64]; + dpf::prg::chacha_detail::block4<20>(keys, counters, wide); + for (int lane = 0; lane < 4; ++lane) + { + std::uint8_t scalar[64]; + dpf::prg::chacha_detail::block<20>(keys[lane], counters[lane], + dpf::prg::chacha_detail::zero_nonce, scalar); + EXPECT_EQ(0, std::memcmp(wide[lane], scalar, 64)) << "lane=" << lane; + } +} + +TEST(ChachaPrg, EvalIsKeystreamChunk) +{ + using prg = dpf::prg::chacha20; + simde__m128i seed = block_from_lanes(0x0123456789abcdefull, 0xfedcba9876543210ull); + std::uint32_t key[8]; + dpf::prg::chacha_detail::seed_key(seed, key); + + for (psnip_uint32_t pos = 0; pos < 8; ++pos) + { + std::uint8_t buf[64]; + dpf::prg::chacha_detail::block<20>(key, pos >> 2, + dpf::prg::chacha_detail::zero_nonce, buf); + EXPECT_TRUE(blocks_equal(prg::eval(seed, pos), + block_from_bytes(buf + 16 * (pos & 3u)))) << "pos=" << pos; + } + + auto kids = prg::eval01(seed); + EXPECT_TRUE(blocks_equal(kids[0], prg::eval(seed, 0))); + EXPECT_TRUE(blocks_equal(kids[1], prg::eval(seed, 1))); + EXPECT_FALSE(blocks_equal(kids[0], kids[1])); + EXPECT_FALSE(blocks_equal(kids[0], seed)); + EXPECT_FALSE(blocks_equal(dpf::prg::chacha12::eval(seed, 0), kids[0])); + EXPECT_FALSE(blocks_equal(dpf::prg::chacha8::eval(seed, 0), kids[0])); +} + +TEST(ChachaPrg, BulkAndWideAgree) +{ + using prg = dpf::prg::chacha20; + simde__m128i seed = block_from_lanes(0x0123456789abcdefull, 0xfedcba9876543210ull); + + auto check_bulk = [&](psnip_uint32_t pos, psnip_uint32_t count) + { + alignas(16) simde__m128i bulk[32]; + prg::eval(seed, bulk, count, pos); + for (psnip_uint32_t i = 0; i < count; ++i) + { + EXPECT_TRUE(blocks_equal(bulk[i], prg::eval(seed, pos + i))) + << "pos=" << pos << " i=" << i; + } + }; + check_bulk(0, 1); + check_bulk(0, 2); + check_bulk(0, 4); + check_bulk(0, 16); + check_bulk(1, 20); + check_bulk(3, 6); + check_bulk(4, 7); + check_bulk(0xfffffffeu, 2); + + alignas(16) simde__m128i seeds[8]; + alignas(16) simde__m128i out4[4]; + alignas(16) simde__m128i out8[8]; + alignas(16) simde__m128i left[4]; + alignas(16) simde__m128i right[4]; + for (int i = 0; i < 8; ++i) + { + seeds[i] = block_from_lanes(0x1000u + static_cast(i), 0x2000u); + } + prg::eval_x4(seeds, out4, 5); + prg::eval_x8(seeds, out8, 0); + prg::eval01_x4(seeds, left, right); + for (int i = 0; i < 4; ++i) + { + EXPECT_TRUE(blocks_equal(out4[i], prg::eval(seeds[i], 5))); + EXPECT_TRUE(blocks_equal(left[i], prg::eval(seeds[i], 0))); + EXPECT_TRUE(blocks_equal(right[i], prg::eval(seeds[i], 1))); + } + for (int i = 0; i < 8; ++i) + { + EXPECT_TRUE(blocks_equal(out8[i], prg::eval(seeds[i], 0))); + } + + alignas(16) simde__m128i one[1]; + EXPECT_NO_THROW(prg::eval(seed, one, 1, 0xffffffffu)); + EXPECT_TRUE(blocks_equal(one[0], prg::eval(seed, 0xffffffffu))); + EXPECT_THROW(prg::eval(seed, out4, 2, 0xffffffffu), std::invalid_argument); + prg::eval(seed, out4, 0, 0xffffffffu); +} + +TEST(ChachaPrg, DpfPointAndFull) +{ + using prg = dpf::prg::chacha20; + const std::uint8_t x = 0x2a; + const std::uint32_t y = 0x01020304; + auto [k0, k1] = dpf::make_dpf(x, y); + + for (int i = 0; i < 256; ++i) + { + auto y0 = *dpf::eval_point(k0, static_cast(i)); + auto y1 = *dpf::eval_point(k1, static_cast(i)); + auto sum = dpf::reconstruct(y0, y1); + EXPECT_EQ(sum, static_cast(i) == x ? y : 0u) << "i=" << i; + } + + auto [buf0, iter0] = dpf::eval_full(k0); + auto [buf1, iter1] = dpf::eval_full(k1); + (void)buf0; + (void)buf1; + std::size_t i = 0; + auto it0 = std::begin(iter0); + auto it1 = std::begin(iter1); + for (; it0 != std::end(iter0); ++it0, ++it1, ++i) + { + auto sum = dpf::reconstruct(*it0, *it1); + EXPECT_EQ(sum, static_cast(i) == x ? y : 0u) << "i=" << i; + } + EXPECT_EQ(i, std::size_t{256}); +} + +TEST(ChachaPrg, ReducedRoundsStillCorrect) +{ + using prg = dpf::prg::chacha8; + const std::uint8_t x = 0x11; + const std::uint32_t y = 0xabcdu; + auto [k0, k1] = dpf::make_dpf(x, y); + for (int i = 0; i < 256; ++i) + { + auto sum = dpf::reconstruct( + *dpf::eval_point(k0, static_cast(i)), + *dpf::eval_point(k1, static_cast(i))); + EXPECT_EQ(sum, static_cast(i) == x ? y : 0u) << "i=" << i; + } +} + +TEST(ChachaPrg, CounterWrapperCountsEval01) +{ + using prg = dpf::prg::counter_wrapper; + const auto before = prg::count(); + simde__m128i seed = block_from_lanes(0x1111ull, 0x2222ull); + auto kids = prg::eval01(seed); + EXPECT_FALSE(blocks_equal(kids[0], kids[1])); + EXPECT_EQ(prg::count(), before + 2u); + + alignas(16) simde__m128i bulk[4]; + prg::eval(seed, bulk, 4, 0); + EXPECT_EQ(prg::count(), before + 2u + 4u); +} + +TEST(ChachaPrg, ExpandAndBufferedReplay) +{ + using prg = dpf::prg::chacha20; + simde__m128i seed = block_from_lanes(0x1111ull, 0x2222ull); + auto t0 = prg::expand(seed, 3); + auto t1 = prg::expand(seed, 3); + EXPECT_EQ(dpf::reconstruct(t0, t1), 0u); + + auto u0 = prg::expand(seed, 0); + auto u1 = prg::expand(seed, 1); + EXPECT_NE(u0.raw(), u1.raw()); + + dpf::randomness::buffered_prg streamed(seed, 8); + auto s0 = streamed.get<0>(); + auto s1 = streamed.get<1>(); + dpf::randomness::buffered_prg replay(seed, 8); + EXPECT_EQ(replay.at<0>(0), s0); + EXPECT_EQ(replay.at<1>(0), s1); + EXPECT_EQ(replay.get<0>(), s0); + EXPECT_NE(s0, streamed.get<0>()); +} diff --git a/test/tests/range_lut_test.cpp b/test/tests/range_lut_test.cpp new file mode 100644 index 0000000..8014060 --- /dev/null +++ b/test/tests/range_lut_test.cpp @@ -0,0 +1,246 @@ +#include + +#include "grotto/range_lut.hpp" + +#include +#include + +namespace +{ + +long double truth_of(grotto::reduced which, long double x) +{ + switch (which) + { + case grotto::reduced::ln: return std::log(x); + case grotto::reduced::lg: return std::log2(x); + case grotto::reduced::log10: return std::log10(x); + case grotto::reduced::exp: return std::exp(x); + case grotto::reduced::exp2: return std::exp2(x); + case grotto::reduced::exp10: return std::exp(x * std::log(10.0L)); + case grotto::reduced::sin: return std::sin(x); + case grotto::reduced::cos: return std::cos(x); + case grotto::reduced::tan: return std::tan(x); + case grotto::reduced::cot: return 1.0L / std::tan(x); + case grotto::reduced::sec: return 1.0L / std::cos(x); + case grotto::reduced::csc: return 1.0L / std::sin(x); + case grotto::reduced::sinh: return std::sinh(x); + case grotto::reduced::cosh: return std::cosh(x); + case grotto::reduced::tanh: return std::tanh(x); + case grotto::reduced::coth: return 1.0L / std::tanh(x); + case grotto::reduced::sech: return 1.0L / std::cosh(x); + case grotto::reduced::csch: return 1.0L / std::sinh(x); + case grotto::reduced::sqrt: return std::sqrt(x); + case grotto::reduced::inv: return 1.0L / x; + case grotto::reduced::rsqrt: return 1.0L / std::sqrt(x); + case grotto::reduced::invsq: return 1.0L / (x * x); + } + return 0; +} + +const char * name_of(grotto::reduced which) +{ + switch (which) + { + case grotto::reduced::ln: return "ln"; + case grotto::reduced::lg: return "lg"; + case grotto::reduced::log10: return "log10"; + case grotto::reduced::exp: return "exp"; + case grotto::reduced::exp2: return "exp2"; + case grotto::reduced::exp10: return "exp10"; + case grotto::reduced::sin: return "sin"; + case grotto::reduced::cos: return "cos"; + case grotto::reduced::tan: return "tan"; + case grotto::reduced::cot: return "cot"; + case grotto::reduced::sec: return "sec"; + case grotto::reduced::csc: return "csc"; + case grotto::reduced::sinh: return "sinh"; + case grotto::reduced::cosh: return "cosh"; + case grotto::reduced::tanh: return "tanh"; + case grotto::reduced::coth: return "coth"; + case grotto::reduced::sech: return "sech"; + case grotto::reduced::csch: return "csch"; + case grotto::reduced::sqrt: return "sqrt"; + case grotto::reduced::inv: return "inv"; + case grotto::reduced::rsqrt: return "rsqrt"; + case grotto::reduced::invsq: return "invsq"; + } + return "?"; +} + +std::int64_t raw_of(long double x, unsigned k) +{ + const long double scaled = std::ldexp(x, static_cast(k)); + return std::llround(scaled); +} + +void expect_ulps(grotto::reduced which, unsigned k, long double x, long double ulps) +{ + const std::int64_t raw = raw_of(x, k); + std::int64_t got = 0; + ASSERT_NO_THROW(got = grotto::eval_reduced(which, k, raw)) + << name_of(which) << " k=" << k << " x=" << static_cast(x); + const long double truth = truth_of(which, std::ldexp(static_cast(raw), -static_cast(k))); + const long double want = truth * std::ldexp(1.0L, static_cast(k)); + EXPECT_LE(std::fabsl(static_cast(got) - want), ulps) + << name_of(which) << " k=" << k << " x=" << static_cast(x) + << " got=" << got << " want=" << static_cast(want); +} + +bool near_odd_multiple_of_half_pi(long double x) +{ + const long double turn = std::fmod(std::fabsl(x), 3.14159265358979323846L); + const long double dist = std::fmod(turn + 1.5707963267948966L, 3.14159265358979323846L); + const long double folded = dist > 1.5707963267948966L ? 3.14159265358979323846L - dist : dist; + return folded < 0.15L; +} + +} // namespace + +TEST(RangeLut, LogarithmsTrackLibmOnEveryPrecision) +{ + const grotto::reduced maps[] = { + grotto::reduced::ln, grotto::reduced::lg, grotto::reduced::log10, + }; + const long double samples[] = { + 0.125L, 0.3L, 0.5L, 0.75L, 1.0L, 1.5L, 2.0L, 3.0L, 7.5L, 16.0L, 24.0L, 100.0L, + }; + for (auto which : maps) + for (unsigned k : grotto::principal_precisions) + for (long double x : samples) + expect_ulps(which, k, x, 6.0L); +} + +TEST(RangeLut, ExponentialsTrackLibmOnEveryPrecision) +{ + const long double samples[] = { + -2.0L, -1.5L, -0.5L, -0.1L, 0.0L, 0.1L, 0.5L, 1.0L, 1.5L, 2.0L, + }; + for (auto which : {grotto::reduced::exp, grotto::reduced::exp2}) + for (unsigned k : grotto::principal_precisions) + for (long double x : samples) + expect_ulps(which, k, x, 16.0L); + for (unsigned k : grotto::principal_precisions) + { + for (long double x : {-0.9L, -0.25L, 0.0L, 0.25L, 0.9L}) + expect_ulps(grotto::reduced::exp10, k, x, 24.0L); + // 10^q scales the absolute error of exp(f ln 10) by the integer power. + for (long double x : {-1.5L, 1.5L}) + expect_ulps(grotto::reduced::exp10, k, x, 80.0L); + } +} + +TEST(RangeLut, QuarterTurnTrigTracksLibm) +{ + const long double samples[] = { + -3.5L, -2.2L, -1.2L, -0.7L, -0.3L, 0.0L, 0.2L, 0.4L, 0.7L, 1.0L, 1.2L, 2.5L, 3.5L, + }; + for (auto which : {grotto::reduced::sin, grotto::reduced::cos}) + for (unsigned k : grotto::principal_precisions) + for (long double x : samples) + expect_ulps(which, k, x, 3.0L); + for (auto which : {grotto::reduced::tan, grotto::reduced::sec}) + for (unsigned k : grotto::principal_precisions) + for (long double x : samples) + { + if (near_odd_multiple_of_half_pi(x)) + continue; + expect_ulps(which, k, x, 8.0L); + } + for (auto which : {grotto::reduced::cot, grotto::reduced::csc}) + for (unsigned k : grotto::principal_precisions) + for (long double x : samples) + { + if (x == 0.0L || near_odd_multiple_of_half_pi(x)) + continue; + expect_ulps(which, k, x, 8.0L); + } +} + +TEST(RangeLut, HyperbolicReductionsTrackLibm) +{ + const long double samples[] = { + -2.0L, -1.2L, -0.4L, -0.05L, 0.05L, 0.4L, 0.7L, 1.2L, 2.0L, + }; + for (auto which : {grotto::reduced::sinh, grotto::reduced::cosh}) + for (unsigned k : grotto::principal_precisions) + for (long double x : samples) + expect_ulps(which, k, x, 12.0L); + for (auto which : { + grotto::reduced::tanh, grotto::reduced::sech, grotto::reduced::coth, grotto::reduced::csch, + }) + { + for (unsigned k : grotto::principal_precisions) + for (long double x : samples) + { + if ((which == grotto::reduced::coth || which == grotto::reduced::csch) && std::fabsl(x) < 0.2L) + continue; + expect_ulps(which, k, x, 6.0L); + } + } +} + +TEST(RangeLut, DyadicLiftsCoverOddAndEvenExponents) +{ + const grotto::reduced maps[] = { + grotto::reduced::sqrt, grotto::reduced::inv, grotto::reduced::rsqrt, grotto::reduced::invsq, + }; + const long double samples[] = { + 0.125L, 0.3L, 0.5L, 0.75L, 1.0L, 1.5L, 2.0L, 3.0L, 6.0L, 7.5L, 16.0L, 24.0L, 48.0L, 100.0L, + }; + for (auto which : maps) + for (unsigned k : grotto::principal_precisions) + for (long double x : samples) + { + const long double budget = which == grotto::reduced::sqrt ? 12.0L : 4.0L; + expect_ulps(which, k, x, budget); + } +} + +TEST(RangeLut, TinyHyperbolicArgumentsUseThePrincipalTables) +{ + for (unsigned k : grotto::principal_precisions) + { + if (k < 13) + continue; + const long double tiny = std::ldexp(1.0L, -16); + expect_ulps(grotto::reduced::sinh, k, tiny, 2.0L); + expect_ulps(grotto::reduced::cosh, k, tiny, 2.0L); + expect_ulps(grotto::reduced::sinh, k, -tiny, 2.0L); + expect_ulps(grotto::reduced::cosh, k, -tiny, 2.0L); + } +} + +TEST(RangeLut, TanhAndCothSaturatePastBeta) +{ + for (unsigned k : grotto::principal_precisions) + { + const std::int64_t one = std::int64_t{1} << k; + const std::int64_t raw = raw_of(20.0L, k); + EXPECT_EQ(grotto::eval_reduced(grotto::reduced::tanh, k, raw), one); + EXPECT_EQ(grotto::eval_reduced(grotto::reduced::tanh, k, -raw), -one); + EXPECT_EQ(grotto::eval_reduced(grotto::reduced::coth, k, raw), one); + EXPECT_EQ(grotto::eval_reduced(grotto::reduced::coth, k, -raw), -one); + } +} + +TEST(RangeLut, SquareRootOfZeroIsZero) +{ + for (unsigned k : grotto::principal_precisions) + EXPECT_EQ(grotto::eval_reduced(grotto::reduced::sqrt, k, 0), 0); +} + +TEST(RangeLut, RejectsPolesAndNonPositiveLogarithms) +{ + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::ln, 16, 0), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::ln, 16, -4), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::lg, 16, -1), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::sqrt, 16, -4), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::inv, 16, 0), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::rsqrt, 16, -8), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::cot, 16, 0), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::csc, 16, 0), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::coth, 16, 0), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::csch, 16, 0), std::domain_error); + EXPECT_THROW(grotto::eval_reduced(grotto::reduced::ln, 7, 32), std::invalid_argument); +} diff --git a/test/tests/signed_prefix_test.cpp b/test/tests/signed_prefix_test.cpp index b17092b..81a5352 100644 --- a/test/tests/signed_prefix_test.cpp +++ b/test/tests/signed_prefix_test.cpp @@ -6,6 +6,7 @@ #include #include +#include #include namespace @@ -141,3 +142,54 @@ TEST(SignedPrefix, RejectsAKeyWithoutAComparison) EXPECT_THROW(grotto::signed_prefix_parities(k0, ends), std::invalid_argument); EXPECT_THROW(grotto::signed_segment_parities(k1, ends), std::invalid_argument); } + +template +void expect_ilogb_segments(const grotto::constant_lut & lut, int8_t alpha) +{ + if (lut.bounds.size() != N) + return; + std::array ends{}; + for (std::size_t i = 0; i < N; ++i) + ends[i] = lut.bounds[i]; + auto [k0, k1] = dpf::make_dpf(alpha, dpf::gt(uint64_t{1})); + const uint64_t mask = k0.cmp().mask; + const auto s0 = grotto::signed_segment_parities(k0, ends); + const auto s1 = grotto::signed_segment_parities(k1, ends); + uint64_t acc = 0; + for (std::size_t i = 0; i < N; ++i) + { + const uint64_t bit = opened(s0[i], s1[i], mask); + acc += bit * static_cast(lut.values[i]); + } + EXPECT_EQ(acc, static_cast(lut(alpha))) << int(alpha); +} + +template +void dispatch_ilogb(const grotto::constant_lut & lut, int8_t alpha, bool & matched) +{ + if (lut.bounds.size() == N) + { + expect_ilogb_segments(lut, alpha); + matched = true; + return; + } + if constexpr (N > 1) + dispatch_ilogb(lut, alpha, matched); +} + +TEST(SignedPrefix, SegmentsRecoverIlogbInt8) +{ + for (unsigned frac : {0u, 4u}) + { + const auto lut = grotto::make_exact_constant_lut( + grotto::exact_constant::ilogb, frac); + ASSERT_GE(lut.bounds.size(), 3u); + ASSERT_LE(lut.bounds.size(), 40u); + for (int v = -128; v <= 127; ++v) + { + bool matched = false; + dispatch_ilogb<40>(lut, static_cast(v), matched); + ASSERT_TRUE(matched) << lut.bounds.size(); + } + } +} diff --git a/test/tests/stress_scenarios_test.cpp b/test/tests/stress_scenarios_test.cpp index 31a53fe..a0409f6 100644 --- a/test/tests/stress_scenarios_test.cpp +++ b/test/tests/stress_scenarios_test.cpp @@ -145,7 +145,8 @@ bool same_cmp_channel(const Key & a, const Key & b) const auto & cb = b.cmp(); if (ca.nbits != cb.nbits || ca.mask != cb.mask || ca.kind != cb.kind || ca.trivial != cb.trivial || ca.eval_as_ge != cb.eval_as_ge - || ca.include_eq != cb.include_eq || ca.active != cb.active) + || ca.include_eq != cb.include_eq || ca.active != cb.active + || ca.incremental != cb.incremental) return false; using word = typename Key::value_cw_word; if (!same_bytes(a.value_cw().data(), b.value_cw().data(), @@ -874,6 +875,32 @@ TEST_F(StressScenariosTest, PrgDummyAesClassicPoint) } } +TEST_F(StressScenariosTest, PrgChachaInteriorAesExteriorClassicPoint) +{ + uint8_t x = 0x2a; + auto [k0, k1] = dpf::make_dpf(x, uint32_t{0x01020304}); + for (int i = 0; i < 256; ++i) + { + auto s = recon(*dpf::eval_point(k0, static_cast(i)), + *dpf::eval_point(k1, static_cast(i))); + EXPECT_EQ(static_cast(s), + static_cast(i) == x ? 0x01020304u : 0u); + } +} + +TEST_F(StressScenariosTest, PrgAesInteriorChachaExteriorClassicPoint) +{ + uint8_t x = 0x91; + auto [k0, k1] = dpf::make_dpf(x, uint32_t{0xdeadbeef}); + for (int i = 0; i < 256; ++i) + { + auto s = recon(*dpf::eval_point(k0, static_cast(i)), + *dpf::eval_point(k1, static_cast(i))); + EXPECT_EQ(static_cast(s), + static_cast(i) == x ? 0xdeadbeefu : 0u); + } +} + TEST_F(StressScenariosTest, PrgLowmcLowmcMultilevelPacking) { uint16_t x = 0x4c1d;