Record Grotto half-ulp tables and comparison geneval, and factor shared beaver terms before the quotient.
Horner and window evaluation need those tables in the tree. Comparison geneval opens the same value words as a Doerner–Shelat key. A factor common to every polynomial term is multiplied first so that preprocessing stays smaller. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
3f10e05176
commit
875f09fec1
14 changed files with 42668 additions and 184 deletions
|
|
@ -14,9 +14,12 @@
|
|||
/// A polynomial is a sum of monomials in several wires.
|
||||
/// `2 + 3*x + 4*y + 5*x*y + 6*pow(x, 2) + pow(x, 2)*y + x*y*z`
|
||||
/// is one round. `λ_x²` is stored once whether it appears as `x²`,
|
||||
/// inside `x² y`, or in a second polynomial. `sgn * (x*y + pow(x, 2))`
|
||||
/// is the sign-corrected form and reuses those powers. A product that
|
||||
/// uses an output of an earlier polynomial is a later round.
|
||||
/// inside `x² y`, or in a second polynomial. Wires that occur with the
|
||||
/// same exponents in every term, as in `a3*(x*z)^3 + a2*(x*z)^2 + a1*(x*z) + a0`,
|
||||
/// are multiplied first and the univariate polynomial is a later round.
|
||||
/// A factor shared by every term, such as a sign or a piecewise scale,
|
||||
/// is applied after the quotient when that uses fewer preprocessing
|
||||
/// values. A lone secret summand is added from its value share.
|
||||
///
|
||||
/// Doerner–Shelat's per-level AND is a `bit_mul` of a fresh bit and
|
||||
/// a fresh block. A wildcard leaf is a `scale` of one scalar by each
|
||||
|
|
@ -720,9 +723,23 @@ public:
|
|||
}
|
||||
else if (g.kind == gate_kind::poly)
|
||||
{
|
||||
for (const auto & term : g.terms)
|
||||
for (auto id : term.factors)
|
||||
for (const auto & step : g.steps)
|
||||
{
|
||||
if (step.value_wire >= 0)
|
||||
{
|
||||
const auto id = static_cast<std::uint32_t>(step.value_wire);
|
||||
if (!wires_[id].value_ready)
|
||||
throw std::logic_error("beaver wire is not ready to open");
|
||||
continue;
|
||||
}
|
||||
for (auto [id, exp] : step.delta)
|
||||
{
|
||||
(void)exp;
|
||||
ensure_delta(id);
|
||||
}
|
||||
if (step.mask_wire >= 0)
|
||||
ensure_delta(static_cast<std::uint32_t>(step.mask_wire));
|
||||
}
|
||||
val = eval_poly(g);
|
||||
}
|
||||
else
|
||||
|
|
@ -858,6 +875,7 @@ private:
|
|||
Ring scale{};
|
||||
int bundle = -1;
|
||||
int mask_wire = -1;
|
||||
int value_wire = -1;
|
||||
bool public_only = false;
|
||||
};
|
||||
|
||||
|
|
@ -1176,67 +1194,108 @@ private:
|
|||
for (std::uint32_t id = 0; id < wires_.size(); ++id)
|
||||
already[id] = needs_blind(id) ? 1 : 0;
|
||||
auto saved = bundles_;
|
||||
std::vector<std::vector<poly_step>> compiled;
|
||||
compiled.reserve(pieces.size());
|
||||
for (const auto & piece : pieces)
|
||||
(void)compile_poly(piece);
|
||||
std::size_t added = bundles_.size() - saved.size();
|
||||
bundles_ = std::move(saved);
|
||||
compiled.push_back(compile_poly(piece));
|
||||
const auto bundle_base = saved.size();
|
||||
std::size_t added = bundles_.size() - bundle_base;
|
||||
std::map<std::uint32_t, char> blinds;
|
||||
for (const auto & piece : pieces)
|
||||
auto note = [&](std::uint32_t id) {
|
||||
if (id < already.size() && already[id] != 0)
|
||||
return;
|
||||
blinds[id] = 1;
|
||||
};
|
||||
for (std::size_t i = bundle_base; i < bundles_.size(); ++i)
|
||||
{
|
||||
for (const auto & term : piece)
|
||||
for (const auto & part : bundles_[i].parts)
|
||||
{
|
||||
for (auto id : term.factors)
|
||||
for (auto [wid, exp] : part.lam)
|
||||
{
|
||||
if (id >= already.size() || already[id] == 0)
|
||||
blinds[id] = 1;
|
||||
(void)exp;
|
||||
note(wid);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const auto & steps : compiled)
|
||||
{
|
||||
for (const auto & step : steps)
|
||||
{
|
||||
if (step.mask_wire >= 0)
|
||||
note(static_cast<std::uint32_t>(step.mask_wire));
|
||||
for (auto [wid, exp] : step.delta)
|
||||
{
|
||||
(void)exp;
|
||||
note(wid);
|
||||
}
|
||||
}
|
||||
}
|
||||
bundles_ = std::move(saved);
|
||||
return added + blinds.size();
|
||||
}
|
||||
|
||||
static std::vector<poly_term> drop_one(std::vector<poly_term> terms, std::uint32_t wire_id)
|
||||
{
|
||||
for (auto & term : terms)
|
||||
{
|
||||
auto it = std::find(term.factors.begin(), term.factors.end(), wire_id);
|
||||
if (it != term.factors.end())
|
||||
term.factors.erase(it);
|
||||
}
|
||||
return terms;
|
||||
}
|
||||
|
||||
static bool wire_in_every(const std::vector<poly_term> & terms, std::uint32_t wire_id)
|
||||
{
|
||||
for (const auto & term : terms)
|
||||
{
|
||||
if (factor_exp(term, wire_id) == 0)
|
||||
return false;
|
||||
}
|
||||
return !terms.empty();
|
||||
}
|
||||
|
||||
wire schedule_terms(std::vector<poly_term> terms)
|
||||
{
|
||||
if (terms.size() < 2)
|
||||
return emit_terms(std::move(terms));
|
||||
|
||||
std::vector<std::vector<poly_term>> best{terms};
|
||||
std::size_t best_cost = estimate_pieces(best);
|
||||
|
||||
std::map<std::uint32_t, char> seen;
|
||||
for (const auto & term : terms)
|
||||
for (auto id : term.factors)
|
||||
seen[id] = 1;
|
||||
for (auto [wire_id, _] : seen)
|
||||
{
|
||||
bool common = true;
|
||||
for (const auto & term : terms)
|
||||
{
|
||||
if (factor_exp(term, wire_id) == 0)
|
||||
common = false;
|
||||
}
|
||||
if (!common)
|
||||
continue;
|
||||
std::vector<poly_term> quot = terms;
|
||||
for (auto & term : quot)
|
||||
{
|
||||
auto it = std::find(term.factors.begin(), term.factors.end(), wire_id);
|
||||
if (it != term.factors.end())
|
||||
term.factors.erase(it);
|
||||
}
|
||||
const std::uint32_t mid = static_cast<std::uint32_t>(wires_.size());
|
||||
poly_term mul;
|
||||
mul.coeff = traits::one();
|
||||
mul.factors = {mid, wire_id};
|
||||
std::vector<std::vector<poly_term>> seq{std::move(quot), {std::move(mul)}};
|
||||
std::size_t cost = estimate_pieces(seq);
|
||||
auto consider = [&](std::vector<std::vector<poly_term>> seq) {
|
||||
const std::size_t cost = estimate_pieces(seq);
|
||||
if (cost < best_cost)
|
||||
{
|
||||
best = std::move(seq);
|
||||
best_cost = cost;
|
||||
}
|
||||
};
|
||||
|
||||
std::map<std::uint32_t, char> seen;
|
||||
for (const auto & term : terms)
|
||||
for (auto id : term.factors)
|
||||
seen[id] = 1;
|
||||
|
||||
std::vector<std::uint32_t> common;
|
||||
for (auto [wire_id, present] : seen)
|
||||
{
|
||||
(void)present;
|
||||
if (wire_in_every(terms, wire_id))
|
||||
common.push_back(wire_id);
|
||||
}
|
||||
const auto fresh = static_cast<std::uint32_t>(wires_.size());
|
||||
for (auto wire_id : common)
|
||||
{
|
||||
poly_term mul;
|
||||
mul.coeff = traits::one();
|
||||
mul.factors = {fresh, wire_id};
|
||||
consider({drop_one(terms, wire_id), {std::move(mul)}});
|
||||
}
|
||||
|
||||
std::map<std::vector<std::uint8_t>, std::vector<std::uint32_t>> clusters;
|
||||
for (auto [wire_id, _] : seen)
|
||||
for (auto [wire_id, present] : seen)
|
||||
{
|
||||
(void)present;
|
||||
std::vector<std::uint8_t> shape;
|
||||
shape.reserve(terms.size());
|
||||
bool any = false;
|
||||
|
|
@ -1247,13 +1306,13 @@ private:
|
|||
any = any || e != 0;
|
||||
}
|
||||
if (any)
|
||||
clusters[shape].push_back(wire_id);
|
||||
clusters[std::move(shape)].push_back(wire_id);
|
||||
}
|
||||
for (auto & [shape, group] : clusters)
|
||||
{
|
||||
(void)shape;
|
||||
if (group.size() < 2)
|
||||
continue;
|
||||
const std::uint32_t mid = static_cast<std::uint32_t>(wires_.size());
|
||||
poly_term prod;
|
||||
prod.coeff = traits::one();
|
||||
prod.factors = group;
|
||||
|
|
@ -1270,26 +1329,28 @@ private:
|
|||
next.factors.push_back(f);
|
||||
}
|
||||
for (std::uint8_t i = 0; i < e; ++i)
|
||||
next.factors.push_back(mid);
|
||||
next.factors.push_back(fresh);
|
||||
rewritten.push_back(std::move(next));
|
||||
}
|
||||
std::vector<std::vector<poly_term>> seq{{std::move(prod)}, std::move(rewritten)};
|
||||
std::size_t cost = estimate_pieces(seq);
|
||||
if (cost < best_cost)
|
||||
consider({{prod}, rewritten});
|
||||
|
||||
std::map<std::uint32_t, char> rewritten_seen;
|
||||
for (const auto & term : rewritten)
|
||||
for (auto id : term.factors)
|
||||
rewritten_seen[id] = 1;
|
||||
const auto later = fresh + 1;
|
||||
for (auto [wire_id, present] : rewritten_seen)
|
||||
{
|
||||
best = std::move(seq);
|
||||
best_cost = cost;
|
||||
(void)present;
|
||||
if (!wire_in_every(rewritten, wire_id))
|
||||
continue;
|
||||
poly_term mul;
|
||||
mul.coeff = traits::one();
|
||||
mul.factors = {later, wire_id};
|
||||
consider({{prod}, drop_one(rewritten, wire_id), {std::move(mul)}});
|
||||
}
|
||||
}
|
||||
|
||||
if (best.size() != 1 && terms.size() == 1 && terms[0].factors.size() <= 4)
|
||||
{
|
||||
std::fprintf(stderr, "split factors=%zu pieces=%zu cost=%zu flat_factors=",
|
||||
terms[0].factors.size(), best.size(), best_cost);
|
||||
for (auto f : terms[0].factors)
|
||||
std::fprintf(stderr, "%u ", f);
|
||||
std::fprintf(stderr, "\n");
|
||||
}
|
||||
wire last{};
|
||||
for (auto & piece : best)
|
||||
last = emit_terms(std::move(piece));
|
||||
|
|
@ -1352,10 +1413,7 @@ private:
|
|||
if (b.parts[0].lam == key)
|
||||
return b.share;
|
||||
}
|
||||
throw std::logic_error(
|
||||
"beaver monomial was not prepared (key " + std::to_string(key.size())
|
||||
+ " bundles " + std::to_string(bundles_.size())
|
||||
+ " gates " + std::to_string(gates_.size()) + ")");
|
||||
throw std::logic_error("beaver monomial was not prepared");
|
||||
}
|
||||
|
||||
unsigned exponent_of(const exp_list & key, std::uint32_t id) const
|
||||
|
|
@ -1512,8 +1570,17 @@ private:
|
|||
std::map<exp_list, Ring> lams;
|
||||
};
|
||||
std::map<exp_list, bucket> buckets;
|
||||
std::vector<poly_step> steps;
|
||||
for (const auto & term : terms)
|
||||
{
|
||||
if (term.factors.size() == 1)
|
||||
{
|
||||
poly_step step;
|
||||
step.scale = term.coeff;
|
||||
step.value_wire = static_cast<int>(term.factors[0]);
|
||||
steps.push_back(std::move(step));
|
||||
continue;
|
||||
}
|
||||
auto groups = group_exponents(term.factors);
|
||||
(void)expansion_size(groups);
|
||||
for_each_term(groups, [&](const exp_list & key) {
|
||||
|
|
@ -1544,7 +1611,6 @@ private:
|
|||
});
|
||||
}
|
||||
|
||||
std::vector<poly_step> steps;
|
||||
for (auto & [delta, slot] : buckets)
|
||||
{
|
||||
if (!(slot.pub == traits::zero()))
|
||||
|
|
@ -1627,6 +1693,13 @@ private:
|
|||
Ring s1 = traits::zero();
|
||||
for (const auto & step : g.steps)
|
||||
{
|
||||
if (step.value_wire >= 0)
|
||||
{
|
||||
const auto & val = wires_[static_cast<std::size_t>(step.value_wire)].value;
|
||||
s0 = traits::add(s0, traits::mul(step.scale, val.p0));
|
||||
s1 = traits::add(s1, traits::mul(step.scale, val.p1));
|
||||
continue;
|
||||
}
|
||||
Ring pub = pow_delta(step.delta);
|
||||
if (step.public_only)
|
||||
{
|
||||
|
|
|
|||
|
|
@ -371,16 +371,19 @@ struct ds_gen_state
|
|||
/// When `cmp` is non-null and active for `level`, also opens `value_cw` via
|
||||
/// the protocol (no second PRG expand outside).
|
||||
template <typename InteriorPRG, typename CwProtocol, typename NodeT,
|
||||
typename InputT, typename AdviceT>
|
||||
typename InputT, typename MaskT, typename AdviceT>
|
||||
void ds_advance_level(ds_gen_state<NodeT> & st, InputT x0, InputT x1,
|
||||
InputT mask, std::size_t level, CwProtocol & proto, NodeT & cw_out,
|
||||
MaskT mask, std::size_t level, CwProtocol & proto, NodeT & cw_out,
|
||||
AdviceT & advice_out, uint64_t * value_cw_out = nullptr,
|
||||
ds_cmp_gen_state * cmp = nullptr)
|
||||
{
|
||||
// Integral bridge so bit extraction works for `keyword` / `modint` /
|
||||
// signed / bitstring the same way dealer gen does via `mask & x`.
|
||||
// `msb_mask` is the unsigned bit pattern; a signed input must not be
|
||||
// required to have that same type.
|
||||
constexpr auto to_int = utils::to_integral_type<InputT>{};
|
||||
const auto mi = to_int(mask);
|
||||
constexpr auto to_mask = utils::to_integral_type<MaskT>{};
|
||||
const auto mi = to_mask(mask);
|
||||
const uint8_t bit0 = static_cast<uint8_t>(!!(mi & to_int(x0)));
|
||||
const uint8_t bit1 = static_cast<uint8_t>(!!(mi & to_int(x1)));
|
||||
|
||||
|
|
|
|||
|
|
@ -14,6 +14,13 @@
|
|||
/// a public query. It samples a random target, runs geneval there,
|
||||
/// and shifts the query by `target - x`, which is what
|
||||
/// `offset_x` does after a wildcard key is bound to `x`.
|
||||
///
|
||||
/// `geneval_cmp` is the comparison-channel form. The value-correction
|
||||
/// word is a function of the secret path at every level, so the walk
|
||||
/// stays live for the whole depth and the opened words match a
|
||||
/// Doerner–Shelat comparison key. Prefix shares are
|
||||
/// `eval_point(cmp, ...)` at each endpoint. Piecewise-cubic evaluation
|
||||
/// on top of that is `grotto::geneval_offset_horner`.
|
||||
/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors)
|
||||
/// @license Released under a GNU General Public v2.0 (GPLv2) license;
|
||||
/// see [LICENSE.md](@ref license) for details.
|
||||
|
|
@ -37,6 +44,7 @@
|
|||
|
||||
#include "dpf/aligned_allocator.hpp"
|
||||
#include "dpf/doerner_shelat.hpp"
|
||||
#include "dpf/eval_target.hpp"
|
||||
#include "dpf/leaf_node.hpp"
|
||||
|
||||
namespace dpf
|
||||
|
|
@ -633,6 +641,83 @@ auto geneval_sequence(wildcard_input_t, InputT x0, InputT x1,
|
|||
begin, end, std::move(rng), [] { return dpf::uniform_sample<InputT>(); }, y);
|
||||
}
|
||||
|
||||
/// Opened comparison key material and one prefix share per endpoint.
|
||||
/// `live_levels` is the full depth: a comparison value word depends on the
|
||||
/// secret path at every level, so there is no early dummy-word tail.
|
||||
struct geneval_cmp_result
|
||||
{
|
||||
std::vector<uint64_t> party0;
|
||||
std::vector<uint64_t> party1;
|
||||
std::vector<simde__m128i, aligned_allocator<simde__m128i>> correction_words;
|
||||
std::vector<uint8_t> correction_advice;
|
||||
std::vector<uint64_t> value_cw;
|
||||
uint64_t cw_last = 0;
|
||||
uint64_t addend0 = 0;
|
||||
uint64_t addend1 = 0;
|
||||
uint64_t mask = 0;
|
||||
std::size_t live_levels = 0;
|
||||
};
|
||||
|
||||
/// Doerner–Shelat comparison geneval. `x0 XOR x1` is the secret point, in the
|
||||
/// same share convention as `geneval_point`. `spec` is an `lt` / `leq` / `gt`
|
||||
/// / `geq` pack. Each endpoint is returned in order as the two parties'
|
||||
/// `eval_point(cmp, ...)` shares. An empty range opens nothing.
|
||||
template <typename InputT,
|
||||
typename ForwardIterator,
|
||||
typename RootSampler,
|
||||
typename PadRng,
|
||||
typename Spec>
|
||||
HEDLEY_WARN_UNUSED_RESULT
|
||||
geneval_cmp_result geneval_cmp(InputT x0, InputT x1,
|
||||
ForwardIterator begin, ForwardIterator end,
|
||||
ds_randomness<RootSampler, PadRng> rng, Spec spec)
|
||||
{
|
||||
geneval_cmp_result out;
|
||||
if (begin == end)
|
||||
return out;
|
||||
|
||||
auto keys = make_dpf_doerner_shelat(std::move(x0), std::move(x1),
|
||||
std::move(rng), std::move(spec));
|
||||
const auto & k0 = keys.first;
|
||||
const auto & k1 = keys.second;
|
||||
using key_type = std::decay_t<decltype(k0)>;
|
||||
constexpr std::size_t depth = key_type::depth;
|
||||
out.live_levels = depth;
|
||||
out.mask = k0.cmp().mask;
|
||||
out.cw_last = k0.cw_last();
|
||||
out.addend0 = k0.cmp_addend().raw();
|
||||
out.addend1 = k1.cmp_addend().raw();
|
||||
out.correction_words.resize(depth);
|
||||
out.correction_advice.resize(depth);
|
||||
out.value_cw.resize(depth);
|
||||
for (std::size_t level = 0; level < depth; ++level)
|
||||
{
|
||||
out.correction_words[level] = k0.correction_word(level);
|
||||
out.correction_advice[level] = static_cast<uint8_t>(k0.correction_advice(level));
|
||||
out.value_cw[level] = k0.value_cw(level);
|
||||
}
|
||||
for (auto it = begin; it != end; ++it)
|
||||
{
|
||||
out.party0.push_back(eval_point(dpf::cmp, k0, *it).raw());
|
||||
out.party1.push_back(eval_point(dpf::cmp, k1, *it).raw());
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/// `gt(beta)` comparison geneval. `if_false` is 0.
|
||||
template <typename InputT,
|
||||
typename ForwardIterator,
|
||||
typename RootSampler,
|
||||
typename PadRng>
|
||||
HEDLEY_WARN_UNUSED_RESULT
|
||||
geneval_cmp_result geneval_cmp(InputT x0, InputT x1,
|
||||
ForwardIterator begin, ForwardIterator end,
|
||||
ds_randomness<RootSampler, PadRng> rng, uint64_t beta)
|
||||
{
|
||||
return geneval_cmp(std::move(x0), std::move(x1), begin, end,
|
||||
std::move(rng), dpf::gt(beta));
|
||||
}
|
||||
|
||||
} // namespace dpf
|
||||
|
||||
#endif // LIBDPF_INCLUDE_DPF_GENEVAL_HPP__
|
||||
|
|
|
|||
|
|
@ -2,19 +2,19 @@
|
|||
/// @brief Noninteractive cubic evaluation after the public offset is opened.
|
||||
/// @details The dealer keys one comparison at `center` per power
|
||||
/// `1, center, center^2, center^3` in Z/2^64. After the parties open
|
||||
/// `eta`, each party shifts the knots by `eta` and sorts them. The
|
||||
/// `eta`, each party shifts the knots by `eta`, inserts the domain
|
||||
/// minimum and the public carry threshold, and sorts. The
|
||||
/// sign-respecting segment walk then returns additive shares of
|
||||
/// `center^m` on the piece that contains the wrapped sum
|
||||
/// `center + eta`, and shares of 0 on the other pieces. A public
|
||||
/// binomial combination of those shares is a share of the coefficients
|
||||
/// of that piece as a polynomial in `eta`. Horner at the public `eta`
|
||||
/// needs no further round.
|
||||
/// `center^m` on the refined piece that contains `center`. On each
|
||||
/// refined piece the wrapped input is `center + kappa` for a public
|
||||
/// `kappa`: `eta` on the side that does not overflow, and
|
||||
/// `eta ∓ 2^n` on the side that does. A public binomial shift by that
|
||||
/// `kappa`, dotted with the segment shares, is a share of the
|
||||
/// polynomial at the wrapped group element. No further round.
|
||||
///
|
||||
/// The opened value is that polynomial at `lift(center) + lift(eta)`
|
||||
/// in Z/2^64. `lift` sign-extends a signed domain element and
|
||||
/// zero-extends an unsigned one. This equals the polynomial at the
|
||||
/// wrapped group element only when the domain addition does not
|
||||
/// overflow. Piece selection still follows the wrapped element.
|
||||
/// Domains of 63 bits or more are already the ring Z/2^64, so the
|
||||
/// carry adjustment is the identity there. `lift` sign-extends a
|
||||
/// signed domain element and zero-extends an unsigned one.
|
||||
///
|
||||
/// `offset_horner_at_x_plus_r` is the wiring from the reconstruction
|
||||
/// the parties already do: `eta = x - r` and `center = 2r`.
|
||||
|
|
@ -26,6 +26,8 @@
|
|||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <limits>
|
||||
#include <optional>
|
||||
#include <stdexcept>
|
||||
#include <tuple>
|
||||
#include <type_traits>
|
||||
|
|
@ -159,18 +161,13 @@ std::vector<shifted_piece<Degree, InputT>> shift_and_sort(
|
|||
return rows;
|
||||
}
|
||||
|
||||
template <typename Key, typename InputT>
|
||||
std::vector<uint64_t> segments_of(const Key & key, const std::vector<InputT> & knots,
|
||||
uint64_t wrap_share)
|
||||
inline std::vector<uint64_t> segments_from_prefixes(
|
||||
const std::vector<uint64_t> & prefix, uint64_t wrap_share, uint64_t mask)
|
||||
{
|
||||
using namespace dpf::detail::dcf_impl;
|
||||
const std::size_t n = knots.size();
|
||||
const uint64_t mask = key.cmp().mask;
|
||||
const std::size_t n = prefix.size();
|
||||
if (n == 1)
|
||||
return std::vector<uint64_t>{wrap_share & mask};
|
||||
|
||||
std::vector<uint64_t> prefix(n);
|
||||
signed_prefix_parities_into(key, knots.data(), n, prefix.data());
|
||||
std::vector<uint64_t> seg(n);
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
{
|
||||
|
|
@ -181,23 +178,87 @@ std::vector<uint64_t> segments_of(const Key & key, const std::vector<InputT> & k
|
|||
return seg;
|
||||
}
|
||||
|
||||
template <std::size_t Degree>
|
||||
void accumulate(std::array<uint64_t, Degree + 1> & out,
|
||||
const std::array<std::vector<uint64_t>, Degree + 1> & seg,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff)
|
||||
template <typename Key, typename InputT>
|
||||
std::vector<uint64_t> segments_of(const Key & key, const std::vector<InputT> & knots,
|
||||
uint64_t wrap_share)
|
||||
{
|
||||
const std::size_t n = coeff.size();
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
const std::size_t n = knots.size();
|
||||
if (n == 1)
|
||||
return std::vector<uint64_t>{wrap_share & key.cmp().mask};
|
||||
std::vector<uint64_t> prefix(n);
|
||||
signed_prefix_parities_into(key, knots.data(), n, prefix.data());
|
||||
return segments_from_prefixes(prefix, wrap_share, key.cmp().mask);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
int64_t math_lift(T value) noexcept
|
||||
{
|
||||
if constexpr (std::is_signed_v<T>)
|
||||
return static_cast<int64_t>(value);
|
||||
else
|
||||
return static_cast<int64_t>(lift(value));
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
T domain_min() noexcept
|
||||
{
|
||||
if constexpr (std::is_signed_v<T>)
|
||||
return std::numeric_limits<T>::min();
|
||||
else
|
||||
return T{0};
|
||||
}
|
||||
|
||||
/// Public center-space cut where `center + eta` crosses the domain end.
|
||||
/// Empty when that cut is outside the domain, including `eta == 0`.
|
||||
template <typename T>
|
||||
std::optional<T> carry_threshold(T eta) noexcept
|
||||
{
|
||||
constexpr unsigned bits = dpf::utils::bitlength_of_v<T>;
|
||||
if (bits > 62)
|
||||
return std::nullopt;
|
||||
const int64_t mod = int64_t{1} << bits;
|
||||
const int64_t half = mod >> 1;
|
||||
const int64_t ez = math_lift(eta);
|
||||
if constexpr (std::is_signed_v<T>)
|
||||
{
|
||||
for (std::size_t m = 0; m <= Degree; ++m)
|
||||
{
|
||||
for (std::size_t k = 0; k <= m; ++k)
|
||||
{
|
||||
// seg[m-k] opens to center^{m-k} on this piece.
|
||||
const uint64_t weight = seg[m - k][i];
|
||||
out[k] += weight * coeff[i][m] * binom[m][k];
|
||||
}
|
||||
}
|
||||
if (ez > 0)
|
||||
return static_cast<T>(half - ez);
|
||||
if (ez < 0)
|
||||
return static_cast<T>(-half - ez);
|
||||
return std::nullopt;
|
||||
}
|
||||
else
|
||||
{
|
||||
if (ez == 0)
|
||||
return std::nullopt;
|
||||
return static_cast<T>(mod - ez);
|
||||
}
|
||||
}
|
||||
|
||||
/// `center + kappa` is the wrapped representative, as a mathematical integer.
|
||||
template <typename T>
|
||||
int64_t kappa_for(T left, T eta) noexcept
|
||||
{
|
||||
constexpr unsigned bits = dpf::utils::bitlength_of_v<T>;
|
||||
const int64_t ez = math_lift(eta);
|
||||
if (bits > 62)
|
||||
return ez;
|
||||
const int64_t mod = int64_t{1} << bits;
|
||||
const int64_t left_i = math_lift(left);
|
||||
if constexpr (std::is_signed_v<T>)
|
||||
{
|
||||
const int64_t half = mod >> 1;
|
||||
if (ez > 0 && left_i >= half - ez)
|
||||
return ez - mod;
|
||||
if (ez < 0 && left_i < -half - ez)
|
||||
return ez + mod;
|
||||
return ez;
|
||||
}
|
||||
else
|
||||
{
|
||||
if (ez != 0 && left_i >= mod - ez)
|
||||
return ez - mod;
|
||||
return ez;
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -232,6 +293,85 @@ std::array<uint64_t, Degree + 1> binomial_coefficients(
|
|||
return c;
|
||||
}
|
||||
|
||||
template <std::size_t Degree, typename InputT>
|
||||
struct prepared_piece
|
||||
{
|
||||
InputT knot{};
|
||||
std::array<uint64_t, Degree + 1> coeff{};
|
||||
int64_t kappa = 0;
|
||||
};
|
||||
|
||||
template <std::size_t Degree, typename InputT>
|
||||
void insert_cut(std::vector<shifted_piece<Degree, InputT>> & rows, InputT point)
|
||||
{
|
||||
for (const auto & row : rows)
|
||||
{
|
||||
if (row.knot == point)
|
||||
return;
|
||||
}
|
||||
std::vector<InputT> knots;
|
||||
knots.reserve(rows.size());
|
||||
for (const auto & row : rows)
|
||||
knots.push_back(row.knot);
|
||||
const int hot = piece_index<Degree>(point, knots);
|
||||
shifted_piece<Degree, InputT> extra;
|
||||
extra.knot = point;
|
||||
extra.coeff = rows[static_cast<std::size_t>(hot)].coeff;
|
||||
rows.push_back(std::move(extra));
|
||||
std::sort(rows.begin(), rows.end(),
|
||||
[](const shifted_piece<Degree, InputT> & a, const shifted_piece<Degree, InputT> & b) {
|
||||
return a.knot < b.knot;
|
||||
});
|
||||
}
|
||||
|
||||
template <std::size_t Degree, typename InputT>
|
||||
std::vector<prepared_piece<Degree, InputT>> prepare_pieces(
|
||||
const std::vector<InputT> & knots,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
|
||||
InputT eta)
|
||||
{
|
||||
auto rows = shift_and_sort<Degree>(knots, coeff, eta);
|
||||
constexpr unsigned bits = dpf::utils::bitlength_of_v<InputT>;
|
||||
if (bits <= 62)
|
||||
{
|
||||
insert_cut<Degree>(rows, domain_min<InputT>());
|
||||
if (const auto cut = carry_threshold(eta))
|
||||
insert_cut<Degree>(rows, *cut);
|
||||
}
|
||||
std::vector<prepared_piece<Degree, InputT>> out;
|
||||
out.reserve(rows.size());
|
||||
for (const auto & row : rows)
|
||||
{
|
||||
prepared_piece<Degree, InputT> piece;
|
||||
piece.knot = row.knot;
|
||||
piece.coeff = row.coeff;
|
||||
piece.kappa = kappa_for(row.knot, eta);
|
||||
out.push_back(std::move(piece));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/// `out[k]` sums to the polynomial at the wrapped input. It is
|
||||
/// `center^k` times the public binomial coefficient of `kappa`, not a
|
||||
/// coefficient you Horner-evaluate at `eta`.
|
||||
template <std::size_t Degree>
|
||||
std::array<uint64_t, Degree + 1> contributions(
|
||||
const std::array<std::vector<uint64_t>, Degree + 1> & seg,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
|
||||
const std::vector<int64_t> & kappa)
|
||||
{
|
||||
std::array<uint64_t, Degree + 1> out{};
|
||||
const std::size_t n = coeff.size();
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
{
|
||||
const auto q = binomial_coefficients<Degree>(
|
||||
coeff[i], static_cast<uint64_t>(kappa[i]));
|
||||
for (std::size_t k = 0; k <= Degree; ++k)
|
||||
out[k] += seg[k][i] * q[k];
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
} // namespace offset_horner_detail
|
||||
|
||||
/// Both parties' comparison keys and wrap-piece shares for one center.
|
||||
|
|
@ -272,7 +412,9 @@ offset_horner_keys<InputT, Degree> make_offset_horner_keys(InputT center)
|
|||
return mat;
|
||||
}
|
||||
|
||||
/// Cleartext coefficients of the selected piece, shifted to `center`, in Z/2^64.
|
||||
/// Cleartext binomial coefficients of the selected refined piece in the
|
||||
/// variable `center`: Horner at `lift(center)` is the polynomial at the
|
||||
/// wrapped input.
|
||||
template <std::size_t Degree, typename InputT>
|
||||
std::array<uint64_t, Degree + 1> offset_horner_clear_coefficients(
|
||||
InputT center,
|
||||
|
|
@ -282,15 +424,18 @@ std::array<uint64_t, Degree + 1> offset_horner_clear_coefficients(
|
|||
{
|
||||
using namespace offset_horner_detail;
|
||||
check_knots(knots, coeff.size());
|
||||
const auto rows = shift_and_sort<Degree>(knots, coeff, eta);
|
||||
std::vector<InputT> shifted(rows.size());
|
||||
for (std::size_t i = 0; i < rows.size(); ++i)
|
||||
shifted[i] = rows[i].knot;
|
||||
const int hot = piece_index<Degree>(center, shifted);
|
||||
return binomial_coefficients<Degree>(rows[static_cast<std::size_t>(hot)].coeff, lift(center));
|
||||
const auto pieces = prepare_pieces<Degree>(knots, coeff, eta);
|
||||
std::vector<InputT> cuts;
|
||||
cuts.reserve(pieces.size());
|
||||
for (const auto & piece : pieces)
|
||||
cuts.push_back(piece.knot);
|
||||
const int hot = piece_index<Degree>(center, cuts);
|
||||
const auto & piece = pieces[static_cast<std::size_t>(hot)];
|
||||
return binomial_coefficients<Degree>(
|
||||
piece.coeff, static_cast<uint64_t>(piece.kappa));
|
||||
}
|
||||
|
||||
/// Cleartext value: selected piece at `lift(center) + lift(eta)` in Z/2^64.
|
||||
/// Cleartext value of the selected piece at the wrapped `center + eta`.
|
||||
template <std::size_t Degree, typename InputT>
|
||||
uint64_t offset_horner_clear(
|
||||
InputT center,
|
||||
|
|
@ -299,7 +444,40 @@ uint64_t offset_horner_clear(
|
|||
InputT eta)
|
||||
{
|
||||
const auto c = offset_horner_clear_coefficients<Degree>(center, knots, coeff, eta);
|
||||
return offset_horner_detail::horner_at<Degree>(c, offset_horner_detail::lift(eta));
|
||||
return offset_horner_detail::horner_at<Degree>(c, offset_horner_detail::lift(center));
|
||||
}
|
||||
|
||||
/// `Party` selects `.first` or `.second` of each key pair.
|
||||
template <std::size_t Party, std::size_t Degree, typename InputT, typename KeyPair>
|
||||
std::array<uint64_t, Degree + 1> offset_horner_coefficient_share(
|
||||
const std::vector<KeyPair> & keys,
|
||||
const std::array<std::array<uint64_t, 2>, Degree + 1> & wrap_share,
|
||||
const std::vector<InputT> & knots,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
|
||||
InputT eta)
|
||||
{
|
||||
static_assert(Party < 2, "offset horner party is 0 or 1");
|
||||
using namespace offset_horner_detail;
|
||||
check_knots(knots, coeff.size());
|
||||
if (keys.size() != Degree + 1)
|
||||
throw std::invalid_argument("offset horner: one comparison key per power");
|
||||
const auto pieces = prepare_pieces<Degree>(knots, coeff, eta);
|
||||
std::vector<InputT> shifted(pieces.size());
|
||||
std::vector<std::array<uint64_t, Degree + 1>> ordered(pieces.size());
|
||||
std::vector<int64_t> kappa(pieces.size());
|
||||
for (std::size_t i = 0; i < pieces.size(); ++i)
|
||||
{
|
||||
shifted[i] = pieces[i].knot;
|
||||
ordered[i] = pieces[i].coeff;
|
||||
kappa[i] = pieces[i].kappa;
|
||||
}
|
||||
|
||||
std::array<std::vector<uint64_t>, Degree + 1> seg;
|
||||
for (std::size_t m = 0; m <= Degree; ++m)
|
||||
{
|
||||
seg[m] = segments_of(std::get<Party>(keys[m]), shifted, wrap_share[m][Party]);
|
||||
}
|
||||
return contributions<Degree>(seg, ordered, kappa);
|
||||
}
|
||||
|
||||
/// One party's coefficient shares. `Party` is 0 or 1.
|
||||
|
|
@ -310,30 +488,164 @@ std::array<uint64_t, Degree + 1> offset_horner_coefficient_share(
|
|||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
|
||||
InputT eta)
|
||||
{
|
||||
static_assert(Party < 2, "offset horner party is 0 or 1");
|
||||
using namespace offset_horner_detail;
|
||||
std::vector<typename offset_horner_keys<InputT, Degree>::key_pair> keys(
|
||||
mat.keys.begin(), mat.keys.end());
|
||||
return offset_horner_coefficient_share<Party, Degree>(
|
||||
keys, mat.wrap_share, knots, coeff, eta);
|
||||
}
|
||||
|
||||
/// Both parties' Horner shares from one joint Doerner–Shelat generation.
|
||||
/// `center0 XOR center1` is the comparison point, in geneval's share
|
||||
/// convention (the signed MSB of `center0` is flipped before the XOR, and
|
||||
/// flipped back here). `eta` is already public.
|
||||
template <std::size_t Degree, typename InputT>
|
||||
struct geneval_offset_horner_result
|
||||
{
|
||||
static_assert(Degree <= offset_horner_max_degree, "offset horner degree is at most 3");
|
||||
|
||||
InputT center{};
|
||||
InputT eta{};
|
||||
std::array<uint64_t, Degree + 1> coeff0{};
|
||||
std::array<uint64_t, Degree + 1> coeff1{};
|
||||
uint64_t value0 = 0;
|
||||
uint64_t value1 = 0;
|
||||
};
|
||||
|
||||
/// Logical comparison point for geneval's XOR shares. Matches `make_dpf(P)`
|
||||
/// when `center1 = P XOR center0` or when `center0 = P` and `center1 = 0`.
|
||||
template <typename InputT>
|
||||
InputT geneval_offset_horner_center(InputT center0, InputT center1)
|
||||
{
|
||||
InputT flipped0 = center0;
|
||||
dpf::utils::flip_msb_if_signed_integral(flipped0);
|
||||
InputT mixed = dpf::utils::xor_input_shares(flipped0, center1);
|
||||
dpf::utils::flip_msb_if_signed_integral(mixed);
|
||||
return mixed;
|
||||
}
|
||||
|
||||
namespace offset_horner_detail
|
||||
{
|
||||
|
||||
template <std::size_t Degree, typename InputT, typename Rng>
|
||||
geneval_offset_horner_result<Degree, InputT> geneval_at(
|
||||
InputT center0, InputT center1, InputT center, InputT eta,
|
||||
const std::vector<InputT> & knots,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
|
||||
Rng rng)
|
||||
{
|
||||
check_knots(knots, coeff.size());
|
||||
const auto rows = shift_and_sort<Degree>(knots, coeff, eta);
|
||||
std::vector<InputT> shifted(rows.size());
|
||||
std::vector<std::array<uint64_t, Degree + 1>> ordered(rows.size());
|
||||
for (std::size_t i = 0; i < rows.size(); ++i)
|
||||
const auto pieces = prepare_pieces<Degree>(knots, coeff, eta);
|
||||
std::vector<InputT> shifted(pieces.size());
|
||||
std::vector<std::array<uint64_t, Degree + 1>> ordered(pieces.size());
|
||||
std::vector<int64_t> kappa(pieces.size());
|
||||
for (std::size_t i = 0; i < pieces.size(); ++i)
|
||||
{
|
||||
shifted[i] = rows[i].knot;
|
||||
ordered[i] = rows[i].coeff;
|
||||
shifted[i] = pieces[i].knot;
|
||||
ordered[i] = pieces[i].coeff;
|
||||
kappa[i] = pieces[i].kappa;
|
||||
}
|
||||
|
||||
std::array<std::vector<uint64_t>, Degree + 1> seg;
|
||||
uint64_t payload[Degree + 1];
|
||||
fill_payloads<Degree>(lift(center), payload);
|
||||
|
||||
std::array<std::array<uint64_t, 2>, Degree + 1> wrap{};
|
||||
std::array<std::vector<uint64_t>, Degree + 1> seg0;
|
||||
std::array<std::vector<uint64_t>, Degree + 1> seg1;
|
||||
for (std::size_t m = 0; m <= Degree; ++m)
|
||||
{
|
||||
seg[m] = segments_of(std::get<Party>(mat.keys[m]), shifted,
|
||||
mat.wrap_share[m][Party]);
|
||||
const auto opened = dpf::geneval_cmp(center0, center1,
|
||||
shifted.begin(), shifted.end(), rng, payload[m]);
|
||||
const uint64_t blind = dpf::uniform_sample<uint64_t>();
|
||||
wrap[m][0] = blind;
|
||||
wrap[m][1] = payload[m] - blind;
|
||||
seg0[m] = segments_from_prefixes(opened.party0, wrap[m][0], opened.mask);
|
||||
seg1[m] = segments_from_prefixes(opened.party1, wrap[m][1], opened.mask);
|
||||
}
|
||||
std::array<uint64_t, Degree + 1> out{};
|
||||
accumulate<Degree>(out, seg, ordered);
|
||||
|
||||
geneval_offset_horner_result<Degree, InputT> out;
|
||||
out.center = center;
|
||||
out.eta = eta;
|
||||
out.coeff0 = contributions<Degree>(seg0, ordered, kappa);
|
||||
out.coeff1 = contributions<Degree>(seg1, ordered, kappa);
|
||||
for (uint64_t term : out.coeff0)
|
||||
out.value0 += term;
|
||||
for (uint64_t term : out.coeff1)
|
||||
out.value1 += term;
|
||||
return out;
|
||||
}
|
||||
|
||||
/// One party's share of the cubic at `lift(center) + lift(eta)`.
|
||||
} // namespace offset_horner_detail
|
||||
|
||||
/// Geneval-style offset Horner. The center is XOR-shared as in `geneval_point`.
|
||||
/// Comparison keys are opened with the same local Doerner–Shelat protocol
|
||||
/// geneval uses for its correction words. The value dot uses the per-piece
|
||||
/// carry shift and is local.
|
||||
/// A value-correction word is required on every level of the secret path, so
|
||||
/// this does not stop early the way a leaf trie does.
|
||||
template <std::size_t Degree = offset_horner_max_degree,
|
||||
typename InputT,
|
||||
typename Rng>
|
||||
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
|
||||
InputT center0, InputT center1, InputT eta,
|
||||
const std::vector<InputT> & knots,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
|
||||
Rng rng)
|
||||
{
|
||||
const InputT center = geneval_offset_horner_center(center0, center1);
|
||||
return offset_horner_detail::geneval_at<Degree>(
|
||||
center0, center1, center, eta, knots, coeff, std::move(rng));
|
||||
}
|
||||
|
||||
template <std::size_t Degree = offset_horner_max_degree, typename InputT>
|
||||
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
|
||||
InputT center0, InputT center1, InputT eta,
|
||||
const std::vector<InputT> & knots,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff)
|
||||
{
|
||||
using block = typename dpf::prg::aes128::block_type;
|
||||
dpf::ds_randomness<block (*)(), dpf::detail::urandom_pad_rng> rng{
|
||||
dpf::uniform_sample<block>, {}};
|
||||
return geneval_offset_horner<Degree>(
|
||||
center0, center1, eta, knots, coeff, std::move(rng));
|
||||
}
|
||||
|
||||
/// Additive shares of the input `x` and the mask `r`. Reconstructs
|
||||
/// `eta = x - r` and `center = 2r`, XOR-shares that center as `(center, 0)`,
|
||||
/// and returns both parties' Horner shares of the cubic at `x + r`
|
||||
/// (the group element `x + r`).
|
||||
template <std::size_t Degree = offset_horner_max_degree,
|
||||
typename InputT,
|
||||
typename Rng>
|
||||
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
|
||||
InputT x0, InputT x1, InputT r0, InputT r1,
|
||||
const std::vector<InputT> & knots,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
|
||||
Rng rng)
|
||||
{
|
||||
const InputT x = offset_horner_group_add(x0, x1);
|
||||
const InputT r = offset_horner_group_add(r0, r1);
|
||||
const InputT eta = offset_horner_group_sub(x, r);
|
||||
const InputT center = offset_horner_group_add(r, r);
|
||||
InputT zero{};
|
||||
return offset_horner_detail::geneval_at<Degree>(
|
||||
center, zero, center, eta, knots, coeff, std::move(rng));
|
||||
}
|
||||
|
||||
template <std::size_t Degree = offset_horner_max_degree, typename InputT>
|
||||
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
|
||||
InputT x0, InputT x1, InputT r0, InputT r1,
|
||||
const std::vector<InputT> & knots,
|
||||
const std::vector<std::array<uint64_t, Degree + 1>> & coeff)
|
||||
{
|
||||
using block = typename dpf::prg::aes128::block_type;
|
||||
dpf::ds_randomness<block (*)(), dpf::detail::urandom_pad_rng> rng{
|
||||
dpf::uniform_sample<block>, {}};
|
||||
return geneval_offset_horner<Degree>(
|
||||
x0, x1, r0, r1, knots, coeff, std::move(rng));
|
||||
}
|
||||
|
||||
/// One party's share of the cubic at the wrapped `center + eta`.
|
||||
/// Sum the coefficient shares; they are already scaled by `center^k`.
|
||||
template <std::size_t Party, std::size_t Degree, typename InputT>
|
||||
uint64_t offset_horner_eval(
|
||||
const offset_horner_keys<InputT, Degree> & mat,
|
||||
|
|
@ -342,7 +654,10 @@ uint64_t offset_horner_eval(
|
|||
InputT eta)
|
||||
{
|
||||
const auto shares = offset_horner_coefficient_share<Party, Degree>(mat, knots, coeff, eta);
|
||||
return offset_horner_detail::horner_at<Degree>(shares, offset_horner_detail::lift(eta));
|
||||
uint64_t value = 0;
|
||||
for (uint64_t term : shares)
|
||||
value += term;
|
||||
return value;
|
||||
}
|
||||
|
||||
} // namespace grotto
|
||||
|
|
|
|||
|
|
@ -5,8 +5,12 @@
|
|||
/// Chebfun). A requested precision only rounds those coefficients
|
||||
/// down to `k + 16` fractional bits. `coth` is different: its
|
||||
/// principal function depends on `k` through
|
||||
/// `beta = ln(2^{k+1}+1)/2`, so each precision has its own uniform
|
||||
/// partition. The returned raw value is
|
||||
/// `beta = ln(2^{k+1}+1)/2`, so each precision has its own
|
||||
/// partition. `inv` (1/x), `rsqrt` (1/sqrt(x)), and `invsq`
|
||||
/// (1/x^2) are also per precision: each is a longest-feasible
|
||||
/// cubic march on the closed principal interval [1/2, 1], with
|
||||
/// absolute error at most half an ulp at that precision. They do
|
||||
/// not apply an exponent lift. The returned raw value is
|
||||
/// `round_half_away(p(x) * 2^k)`. On the closed principal domain,
|
||||
/// `p` stays within one unit in the last place of that precision.
|
||||
|
||||
|
|
@ -33,6 +37,9 @@ enum class principal : unsigned
|
|||
sec,
|
||||
gsec,
|
||||
csch,
|
||||
inv,
|
||||
rsqrt,
|
||||
invsq,
|
||||
};
|
||||
|
||||
inline constexpr unsigned principal_precisions[] = {8u, 12u, 16u, 20u, 24u, 28u, 32u};
|
||||
|
|
@ -71,6 +78,7 @@ struct table_ref
|
|||
};
|
||||
|
||||
#include "grotto/principal_tables.inc"
|
||||
#include "grotto/principal_recip_tables.inc"
|
||||
|
||||
struct w256
|
||||
{
|
||||
|
|
@ -288,6 +296,14 @@ inline const table_ref & table_for(principal which, unsigned fractional_bits)
|
|||
const unsigned slot = fractional_bits / 4u - 2u;
|
||||
return *COTH_BY_K[slot];
|
||||
}
|
||||
if (which == principal::inv || which == principal::rsqrt || which == principal::invsq)
|
||||
{
|
||||
const unsigned slot = fractional_bits / 4u - 2u;
|
||||
const table_ref * const * bank = which == principal::inv
|
||||
? INV_BY_K
|
||||
: which == principal::rsqrt ? RSQRT_BY_K : INVSQ_BY_K;
|
||||
return *bank[slot];
|
||||
}
|
||||
return *SHARED_TABLE[index];
|
||||
}
|
||||
|
||||
|
|
|
|||
1134
include/grotto/principal_recip_tables.inc
Normal file
1134
include/grotto/principal_recip_tables.inc
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -5,7 +5,10 @@
|
|||
/// cubic. `erfc`, `softminus`, `logsigmoid`, and `acos` are integer
|
||||
/// rewrites of `erf`, `softplus`, and `asin`. `asin` on `(1/2, 1]`
|
||||
/// uses `π/2 − 2 asin(sqrt((1−x)/2))` with the principal square-root
|
||||
/// table. `probit` is stored on `(0, 1/2]` and mirrored.
|
||||
/// table. `probit` is stored on `(0, 1/2]` and mirrored. The tail
|
||||
/// below 1/20 is a cubic in `ln(p)` (knots at scale `2^{k+10}`),
|
||||
/// because a cubic in `p` cannot meet half an ulp on the first
|
||||
/// input step once `k` is large.
|
||||
|
||||
#ifndef LIBDPF_INCLUDE_GROTTO_WINDOW_LUT_HPP__
|
||||
#define LIBDPF_INCLUDE_GROTTO_WINDOW_LUT_HPP__
|
||||
|
|
@ -128,38 +131,153 @@ inline std::int64_t round_half_away_i128(__int128 number, unsigned shift)
|
|||
return static_cast<std::int64_t>(neg ? -out : out);
|
||||
}
|
||||
|
||||
/// `round(sqrt(v / 2^{k+1}) * 2^k)`, `v > 0`.
|
||||
inline std::int64_t sqrt_half_scale(unsigned fractional_bits, std::int64_t magnitude)
|
||||
inline unsigned __int128 isqrt_floor(unsigned __int128 n)
|
||||
{
|
||||
const int log = 63 - __builtin_clzll(static_cast<unsigned long long>(magnitude));
|
||||
const std::int64_t mant = magnitude << (fractional_bits - static_cast<unsigned>(log + 1));
|
||||
const std::int64_t root = eval_principal(principal::sqrt, fractional_bits, mant);
|
||||
const int exp2 = log - static_cast<int>(fractional_bits);
|
||||
if ((exp2 & 1) == 0)
|
||||
return round_half_away_i128(root, static_cast<unsigned>(-exp2) / 2u);
|
||||
const unsigned t = static_cast<unsigned>(-exp2 - 1) / 2u;
|
||||
// sqrt(2) rounded onto 62 fractional bits.
|
||||
constexpr __int128 sqrt2_62 = 6521908912666391106LL;
|
||||
return round_half_away_i128(__int128(root) * sqrt2_62, 62u + t + 1u);
|
||||
if (n == 0)
|
||||
return 0;
|
||||
const unsigned bits = (n >> 64) != 0
|
||||
? 128u - static_cast<unsigned>(__builtin_clzll(static_cast<unsigned long long>(n >> 64)))
|
||||
: 64u - static_cast<unsigned>(__builtin_clzll(static_cast<unsigned long long>(n)));
|
||||
unsigned __int128 x = static_cast<unsigned __int128>(1) << ((bits + 1u) / 2u);
|
||||
for (;;)
|
||||
{
|
||||
const unsigned __int128 y = (x + n / x) >> 1;
|
||||
if (y >= x)
|
||||
break;
|
||||
x = y;
|
||||
}
|
||||
while (x > 0 && x > n / x)
|
||||
--x;
|
||||
return x;
|
||||
}
|
||||
|
||||
/// `round(sqrt(v / 2^{k+1}) * 2^{k+extra})`, `v > 0`. Eight extra bits so the
|
||||
/// half-angle identity can absorb the square root before the final rounding.
|
||||
inline std::int64_t sqrt_half_scale_fine(unsigned fractional_bits, std::int64_t magnitude)
|
||||
{
|
||||
constexpr unsigned extra = 8;
|
||||
const unsigned shift = fractional_bits + 2u * extra;
|
||||
const unsigned __int128 radicand =
|
||||
static_cast<unsigned __int128>(static_cast<std::uint64_t>(magnitude)) << shift;
|
||||
const unsigned __int128 root = isqrt_floor(radicand);
|
||||
// sqrt(gap << (k+2*extra)) / sqrt(2) = sqrt(gap / 2^{k+1}) * 2^{k+extra}
|
||||
static constexpr unsigned __int128 sqrt2_64 =
|
||||
(static_cast<unsigned __int128>(1) << 64) | static_cast<unsigned __int128>(7640891576956012809ULL);
|
||||
const unsigned __int128 scaled = (root * sqrt2_64 + (static_cast<unsigned __int128>(1) << 64)) >> 65;
|
||||
return static_cast<std::int64_t>(scaled);
|
||||
}
|
||||
|
||||
inline int piece_of_scaled(const window_table & table, std::int64_t raw, unsigned extra)
|
||||
{
|
||||
int lo = 0;
|
||||
int hi = static_cast<int>(table.nparts);
|
||||
while (hi - lo > 1)
|
||||
{
|
||||
const int mid = (lo + hi) / 2;
|
||||
if ((table.knots[mid] << extra) <= raw)
|
||||
lo = mid;
|
||||
else
|
||||
hi = mid;
|
||||
}
|
||||
return lo;
|
||||
}
|
||||
|
||||
inline std::int64_t eval_asin_abs(unsigned fractional_bits, std::int64_t magnitude)
|
||||
{
|
||||
constexpr unsigned extra = 8;
|
||||
const auto half = std::int64_t{1} << (fractional_bits - 1);
|
||||
const window_table & table = at(ASIN, fractional_bits);
|
||||
if (magnitude <= half)
|
||||
return eval_table(table, fractional_bits, magnitude);
|
||||
const std::int64_t one = std::int64_t{1} << fractional_bits;
|
||||
if (magnitude >= one)
|
||||
return HALF_PI_RAW[slot_of(fractional_bits)];
|
||||
const std::int64_t gap = one - magnitude;
|
||||
const auto pi = HALF_PI_RAW[slot_of(fractional_bits)];
|
||||
if (gap <= 0)
|
||||
return pi;
|
||||
std::int64_t reduced = sqrt_half_scale(fractional_bits, gap);
|
||||
if (reduced > half)
|
||||
reduced = half;
|
||||
const std::int64_t inner = eval_table(table, fractional_bits, reduced);
|
||||
const std::int64_t lifted = pi - 2 * inner;
|
||||
return lifted < 0 ? 0 : lifted;
|
||||
std::int64_t reduced = sqrt_half_scale_fine(fractional_bits, gap);
|
||||
const std::int64_t half_fine = half << extra;
|
||||
if (reduced > half_fine)
|
||||
reduced = half_fine;
|
||||
const unsigned scale = fractional_bits + extra;
|
||||
const std::int64_t inner = horner(
|
||||
table.pieces[piece_of_scaled(table, reduced, extra)], table.q, reduced, scale);
|
||||
// pi/2 at 64 fractional bits, then onto scale k+extra in one rounding.
|
||||
static constexpr unsigned __int128 half_pi_64 =
|
||||
(static_cast<unsigned __int128>(1) << 64) | static_cast<unsigned __int128>(10529333758598939754ULL);
|
||||
const __int128 pi_fine = round_half_away_i128(
|
||||
static_cast<__int128>(half_pi_64), 64u - scale);
|
||||
const __int128 lifted = pi_fine - 2 * static_cast<__int128>(inner);
|
||||
const auto out = round_half_away_i128(lifted, extra);
|
||||
return out < 0 ? 0 : out;
|
||||
}
|
||||
|
||||
/// Surplus fractional bits on probit-tail knots. `u = ln(p)` is stored as
|
||||
/// `round(u * 2^{k+probit_tail_extra})`.
|
||||
inline constexpr unsigned probit_tail_extra = 10;
|
||||
|
||||
// ln((32+i)/64) * 2^64, stored as a positive magnitude. Every anchor is in (0, ln 2].
|
||||
static constexpr std::uint64_t probit_ln2_64 = 12786308645202655660ull;
|
||||
static constexpr std::uint64_t probit_ln_anchor_mag[32] = {
|
||||
12786308645202655660ull, 12218671733053503897ull, 11667981761989453435ull, 11133256087961349648ull,
|
||||
10613595130224743362ull, 10108173265494422292ull, 9616230936675340827ull, 9137067786804269247ull,
|
||||
8670036662410753619ull, 8214538357444912273ull, 7770016990662967709ull, 7335955927010031419ull,
|
||||
6911874167941132216ull, 6497323147432841322ull, 6091883880171659064ull, 5695164416463867605ull,
|
||||
5306797565112371681ull, 4926438851101192057ull, 4553764679618851579ull, 4188470681899169456ull,
|
||||
3830270221691897566ull, 3478893044001375095ull, 3134084050134459383ull, 2795602185149230175ull,
|
||||
2463219425550596028ull, 2136719856585056848ull, 1815898829783402670ull, 1500562192519310430ull,
|
||||
1190525582320469641ull, 885613779509420443ull, 585660112482476600ull, 290505910572683730ull,
|
||||
};
|
||||
|
||||
/// `round_half_away(ln(probability / 2^k) * 2^{k+10})`.
|
||||
inline std::int64_t probit_ln_argument(unsigned fractional_bits, std::int64_t probability)
|
||||
{
|
||||
const auto bits = static_cast<unsigned long long>(probability);
|
||||
const int e = 63 - __builtin_clzll(bits);
|
||||
const unsigned shift_in = static_cast<unsigned>(e + 1);
|
||||
const auto wide = static_cast<unsigned __int128>(bits) << (64u - shift_in);
|
||||
const auto m64 = static_cast<std::uint64_t>(wide);
|
||||
const unsigned idx = static_cast<unsigned>((m64 - (1ull << 63)) >> 58);
|
||||
const unsigned b_num = 32u + idx;
|
||||
const unsigned __int128 t_scaled = (static_cast<unsigned __int128>(m64) * 64u) / b_num;
|
||||
__int128 t = static_cast<__int128>(t_scaled - (static_cast<unsigned __int128>(1) << 64));
|
||||
__int128 p = t;
|
||||
__int128 acc = 0;
|
||||
for (int n = 1; n <= 14; ++n)
|
||||
{
|
||||
const __int128 term = p / n;
|
||||
acc += (n & 1) ? term : -term;
|
||||
p = (p * t) >> 64;
|
||||
}
|
||||
const __int128 ln_m = -static_cast<__int128>(probit_ln_anchor_mag[idx]) + acc;
|
||||
const int exp_fix = e + 1 - static_cast<int>(fractional_bits);
|
||||
const __int128 ln_x = ln_m + static_cast<__int128>(exp_fix) * static_cast<__int128>(probit_ln2_64);
|
||||
return round_half_away_i128(ln_x, 54u - fractional_bits);
|
||||
}
|
||||
|
||||
/// Horner, then one extra right shift so a tail argument at scale `k+10` rounds onto scale `k`.
|
||||
inline std::int64_t eval_cubic_extra(
|
||||
const cubic_bits & piece, unsigned q, std::int64_t raw,
|
||||
unsigned fractional_bits, unsigned extra_shift)
|
||||
{
|
||||
using namespace principal_detail;
|
||||
__int128 coeff[4];
|
||||
for (int i = 0; i < 4; ++i)
|
||||
coeff[i] = unpack_coeff(piece.hi[i], piece.lo[i]);
|
||||
const int q_use = static_cast<int>(q) < static_cast<int>(fractional_bits) + 16
|
||||
? static_cast<int>(q)
|
||||
: static_cast<int>(fractional_bits) + 16;
|
||||
const int drop = static_cast<int>(q) - q_use;
|
||||
for (int i = 0; i < 4; ++i)
|
||||
coeff[i] = rshift_ties_even(coeff[i], drop);
|
||||
|
||||
w256 acc = w_from_i128(coeff[3]);
|
||||
for (int i = 2; i >= 0; --i)
|
||||
{
|
||||
acc = w_mul_i64(acc, raw);
|
||||
w256 term = w_shl(w_from_i128(coeff[i]), fractional_bits * static_cast<unsigned>(3 - i));
|
||||
acc = w_add(acc, term);
|
||||
}
|
||||
const unsigned denom_shift = static_cast<unsigned>(q_use) + 2u * fractional_bits + extra_shift;
|
||||
return round_half_away_pow2(acc, denom_shift);
|
||||
}
|
||||
|
||||
inline std::int64_t eval_probit_abs(unsigned fractional_bits, std::int64_t probability)
|
||||
|
|
@ -167,7 +285,14 @@ inline std::int64_t eval_probit_abs(unsigned fractional_bits, std::int64_t proba
|
|||
const window_table & mid = at(PROBIT_MID, fractional_bits);
|
||||
if (probability >= mid.knots[0])
|
||||
return eval_table(mid, fractional_bits, probability);
|
||||
return eval_table(at(PROBIT_TAIL, fractional_bits), fractional_bits, probability);
|
||||
const window_table & tail = at(PROBIT_TAIL, fractional_bits);
|
||||
std::int64_t u = probit_ln_argument(fractional_bits, probability);
|
||||
if (u < tail.knots[0])
|
||||
u = tail.knots[0];
|
||||
if (u > tail.knots[tail.nparts])
|
||||
u = tail.knots[tail.nparts];
|
||||
const unsigned scale = fractional_bits + probit_tail_extra;
|
||||
return eval_cubic_extra(tail.pieces[piece_of(tail, u)], tail.q, u, scale, probit_tail_extra);
|
||||
}
|
||||
|
||||
inline std::int64_t eval_smoothstep(unsigned fractional_bits, std::int64_t raw)
|
||||
|
|
|
|||
36725
include/grotto/window_tables.inc
Normal file
36725
include/grotto/window_tables.inc
Normal file
File diff suppressed because one or more lines are too long
Loading…
Add table
Add a link
Reference in a new issue