libdpf/include/grotto/offset_horner.hpp
Ryan Henry 0d22946a0e Checkpoint the party/runtime stack before share-program and malicious-mode work.
Ship the TLS mesh, composer, Beaver/Yao/leaf MPC, prep/online paths, apps, and docs so the tree is pushable before elevating share_expr, security_mode, and prep resume.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-28 05:59:19 -06:00

1173 lines
47 KiB
C++
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

/// @file grotto/offset_horner.hpp
/// @brief Noninteractive cubic evaluation after the public offset is opened.
/// @details The dealer keys one comparison at `center` whose payload is the
/// vector `1, center, center^2, center^3` in Z/2^64. The seed spine
/// is stored once; the value words grow with the degree. After the parties open
/// `eta`, each party shifts the knots by `eta`, inserts the domain
/// minimum and the public carry threshold, and sorts. The
/// sign-respecting segment walk then returns additive shares of
/// `center^m` on the refined piece that contains `center`. On each
/// refined piece the wrapped input is `center + kappa` for a public
/// `kappa`: `eta` on the side that does not overflow, and
/// `eta ∓ 2^n` on the side that does. A public binomial shift by that
/// `kappa`, dotted with the segment shares, is a share of the
/// polynomial at the wrapped group element. No further round.
///
/// Domains of 63 bits or more are already the ring Z/2^64, so the
/// carry adjustment is the identity there. `lift` sign-extends a
/// signed domain element and zero-extends an unsigned one.
///
/// `offset_horner_at_x_plus_r` is the wiring from the reconstruction
/// the parties already do: `eta = x - r` and `center = 2r`.
/// @note Storrier, Vadapalli, Lyons, and Henry (ePrint 2023/108) evaluate a public piecewise polynomial from one point key by prefix parity. This dealer keys one comparison of a secret center, with a `dpf::vec` of the powers as the payload.
#ifndef LIBDPF_INCLUDE_GROTTO_OFFSET_HORNER_HPP__
#define LIBDPF_INCLUDE_GROTTO_OFFSET_HORNER_HPP__
#include "hedley/hedley.h"
#include <algorithm>
#include <array>
#include <cstddef>
#include <cstdint>
#include <limits>
#include <optional>
#include <stdexcept>
#include <tuple>
#include <type_traits>
#include <utility>
#include <variant>
#include <vector>
#include "dpf.hpp"
#include "grotto/prefix_parity.hpp"
namespace grotto
{
/// \complexity One modular add in the input group (cast to the unsigned width). `Θ(1)`.
/// @see grotto::offset_horner_at_x_plus_r
inline constexpr std::size_t offset_horner_max_degree = 3;
template <typename T>
HEDLEY_CONST
HEDLEY_NO_THROW
constexpr T offset_horner_group_add(T a, T b) noexcept
{
using u = std::make_unsigned_t<T>;
return static_cast<T>(static_cast<u>(static_cast<u>(a) + static_cast<u>(b)));
}
/// \complexity One modular subtract in the input group. `Θ(1)`.
/// @see grotto::offset_horner_group_add
template <typename T>
HEDLEY_CONST
HEDLEY_NO_THROW
constexpr T offset_horner_group_sub(T a, T b) noexcept
{
using u = std::make_unsigned_t<T>;
return static_cast<T>(static_cast<u>(static_cast<u>(a) - static_cast<u>(b)));
}
/// @brief `eta = x - r` and `center = 2r`, both in the input group.
/// @tparam T value type
template <typename T>
struct offset_horner_x_plus_r
{
T eta{};
T center{};
};
/// \complexity Two group operations: `eta = x - r`, `center = 2r`. `Θ(1)`.
/// @param x secret input share or value, in the input group
/// @param r mask, in the input group
/// @return `eta` and `center`
/// @see grotto::make_offset_horner_keys
template <typename T>
HEDLEY_CONST
HEDLEY_NO_THROW
constexpr offset_horner_x_plus_r<T> offset_horner_at_x_plus_r(T x, T r) noexcept
{
return offset_horner_x_plus_r<T>{
offset_horner_group_sub(x, r),
offset_horner_group_add(r, r)};
}
template <typename InputT, std::size_t Degree = offset_horner_max_degree,
bool Verifiable = false>
struct offset_horner_keys;
namespace offset_horner_detail
{
inline constexpr uint64_t binom[4][4] = {
{1, 0, 0, 0},
{1, 1, 0, 0},
{1, 2, 1, 0},
{1, 3, 3, 1},
};
template <typename T>
HEDLEY_CONST
HEDLEY_NO_THROW
constexpr uint64_t lift(T v) noexcept
{
if constexpr (std::is_signed_v<T>)
return static_cast<uint64_t>(static_cast<std::int64_t>(v));
else
return static_cast<uint64_t>(v);
}
template <std::size_t Degree>
HEDLEY_PURE
HEDLEY_NO_THROW
constexpr uint64_t horner_at(const std::array<uint64_t, Degree + 1> & coeff, uint64_t point) noexcept
{
uint64_t acc = coeff[Degree];
for (std::size_t k = Degree; k-- > 0; )
acc = acc * point + coeff[k];
return acc;
}
template <std::size_t Degree>
HEDLEY_NO_THROW
void fill_payloads(uint64_t base, uint64_t (&payload)[Degree + 1]) noexcept
{
uint64_t pow = 1;
for (std::size_t m = 0; m <= Degree; ++m)
{
payload[m] = pow;
pow *= base;
}
}
template <typename InputT, std::size_t Degree>
dpf::vec<uint64_t, Degree + 1> payload_vec(const uint64_t (&payload)[Degree + 1])
{
dpf::vec<uint64_t, Degree + 1> v;
for (std::size_t m = 0; m <= Degree; ++m)
v[m] = payload[m];
return v;
}
template <typename InputT, std::size_t Degree>
auto make_power_key(InputT center, const uint64_t (&payload)[Degree + 1], std::false_type)
{
return dpf::make_dpf(center, dpf::gt(payload_vec<InputT, Degree>(payload)));
}
template <typename InputT, std::size_t Degree>
auto make_power_key(InputT center, const uint64_t (&payload)[Degree + 1], std::true_type)
{
return dpf::make_dpf(center, dpf::gt(payload_vec<InputT, Degree>(payload)),
dpf::verifiable{});
}
template <typename InputT>
void check_knots(const std::vector<InputT> & knots, std::size_t coeff_rows)
{
if (knots.empty() || knots.size() != coeff_rows)
throw std::invalid_argument("offset horner: knots and coefficient rows differ");
for (std::size_t i = 1; i < knots.size(); ++i)
{
if (!(knots[i - 1] < knots[i]))
throw std::invalid_argument("offset horner: knots must be strictly increasing");
}
}
template <std::size_t Degree, typename InputT>
struct shifted_piece
{
InputT knot{};
std::array<uint64_t, Degree + 1> coeff{};
};
template <std::size_t Degree, typename InputT>
std::vector<shifted_piece<Degree, InputT>> shift_and_sort(
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
InputT eta)
{
std::vector<shifted_piece<Degree, InputT>> rows(knots.size());
for (std::size_t i = 0; i < knots.size(); ++i)
{
rows[i].knot = offset_horner_group_sub(knots[i], eta);
rows[i].coeff = coeff[i];
}
std::sort(rows.begin(), rows.end(),
[](const shifted_piece<Degree, InputT> & a, const shifted_piece<Degree, InputT> & b) {
return a.knot < b.knot;
});
return rows;
}
inline std::vector<uint64_t> segments_from_prefixes(
const std::vector<uint64_t> & prefix, uint64_t wrap_share, uint64_t mask)
{
using namespace dpf::detail::dcf_impl;
const std::size_t n = prefix.size();
if (n == 1)
return std::vector<uint64_t>{wrap_share & mask};
std::vector<uint64_t> seg(n);
for (std::size_t i = 0; i < n; ++i)
{
const uint64_t nxt = prefix[(i + 1) % n];
seg[i] = (nxt + neg_m(prefix[i], mask)) & mask;
}
seg[n - 1] = (seg[n - 1] + wrap_share) & mask;
return seg;
}
/// @brief One segment walk whose comparison payload is `N` lanes.
/// Lane `m` matches a per-power `segments_of` on `gt(payload[m])`.
template <std::size_t N, typename Key, typename InputT>
std::array<std::vector<uint64_t>, N> segments_lanes(
const Key & key, const std::vector<InputT> & knots,
const std::array<uint64_t, N> & wrap_party, dpf::proof_token * pi = nullptr)
{
using Vec = dpf::vec<std::uint64_t, N>;
const std::size_t n = knots.size();
std::array<std::vector<uint64_t>, N> out;
for (auto & row : out)
row.assign(n, 0);
if (n == 1)
{
if (pi != nullptr)
{
dpf::detail::vdpf::init_proof(*pi, key);
dpf::detail::vdpf::fold_output_binding(*pi, key);
}
const uint64_t mask = key.cmp().mask;
for (std::size_t m = 0; m < N; ++m)
out[m][0] = wrap_party[m] & mask;
return out;
}
if (pi != nullptr)
{
if constexpr (!Key::is_verifiable)
throw std::invalid_argument(
"offset horner: proof token requires a verifiable key");
dpf::detail::vdpf::init_proof(*pi, key);
}
const uint64_t mask = key.cmp().mask;
auto path = dpf::make_basic_path_memoizer(key);
std::vector<Vec> prefixes(n);
for (std::size_t which = 0; which < n; ++which)
{
auto tx = key.offset_x(knots[which]);
dpf::utils::flip_msb_if_signed_integral(tx);
prefixes[which] = dpf::detail::incr::eval_payload_path_sum<Vec>(
key, tx, path, false, 0, pi);
}
if (pi != nullptr)
dpf::detail::vdpf::fold_output_binding(*pi, key);
for (std::size_t m = 0; m < N; ++m)
{
std::vector<uint64_t> prefix(n);
for (std::size_t i = 0; i < n; ++i)
prefix[i] = prefixes[i][m];
out[m] = segments_from_prefixes(prefix, wrap_party[m], mask);
}
return out;
}
/// @brief Widest lane count stored in one offset comparison (degree 16).
inline constexpr std::size_t lane_key_max = 17;
template <typename InputT, std::size_t N, bool Verifiable>
struct lane_key_slot
{
static constexpr std::size_t lanes = N;
using key_pair = std::conditional_t<Verifiable,
decltype(dpf::make_dpf(std::declval<InputT>(),
dpf::idcf(dpf::gt(dpf::vec<uint64_t, N>{})), dpf::verifiable{})),
decltype(dpf::make_dpf(std::declval<InputT>(),
dpf::idcf(dpf::gt(dpf::vec<uint64_t, N>{}))))>;
key_pair keys;
explicit lane_key_slot(key_pair k)
: keys(std::move(k)) { }
};
template <typename InputT, bool Verifiable, typename Seq>
struct lane_key_variant;
template <typename InputT, bool Verifiable, std::size_t... I>
struct lane_key_variant<InputT, Verifiable, std::index_sequence<I...>>
{
using type = std::variant<std::monostate,
lane_key_slot<InputT, I + 1, Verifiable>...>;
};
template <typename InputT, bool Verifiable>
using lane_keys = typename lane_key_variant<InputT, Verifiable,
std::make_index_sequence<lane_key_max>>::type;
template <typename InputT, std::size_t N, bool Verifiable>
lane_key_slot<InputT, N, Verifiable> make_lane_slot(
InputT center, const uint64_t * payload)
{
dpf::vec<uint64_t, N> v;
for (std::size_t i = 0; i < N; ++i)
v[i] = payload[i];
if constexpr (Verifiable)
return lane_key_slot<InputT, N, Verifiable>{
dpf::make_dpf(center, dpf::idcf(dpf::gt(v)), dpf::verifiable{})};
else
return lane_key_slot<InputT, N, Verifiable>{
dpf::make_dpf(center, dpf::idcf(dpf::gt(v)))};
}
template <typename InputT, bool Verifiable, std::size_t N>
void emplace_lane_keys(lane_keys<InputT, Verifiable> & out,
InputT center, const uint64_t * payload, std::size_t n)
{
if (n == N)
out.template emplace<lane_key_slot<InputT, N, Verifiable>>(
make_lane_slot<InputT, N, Verifiable>(center, payload));
else if constexpr (N > 1)
emplace_lane_keys<InputT, Verifiable, N - 1>(out, center, payload, n);
}
template <typename InputT, bool Verifiable>
lane_keys<InputT, Verifiable> make_lane_keys(
InputT center, const uint64_t * payload, std::size_t n)
{
if (n == 0 || n > lane_key_max)
throw std::invalid_argument("offset comparison: lane count out of range");
lane_keys<InputT, Verifiable> out;
emplace_lane_keys<InputT, Verifiable, lane_key_max>(out, center, payload, n);
return out;
}
template <std::size_t Party, typename Slot, typename InputT>
std::vector<std::vector<uint64_t>> segments_of_slot(
const Slot & slot, const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, 2>> & wrap, dpf::proof_token * pi)
{
constexpr std::size_t N = Slot::lanes;
if (wrap.size() < N)
throw std::invalid_argument("offset comparison: wrap shares shorter than the payload");
std::array<uint64_t, N> wrap_party{};
for (std::size_t m = 0; m < N; ++m)
wrap_party[m] = wrap[m][Party];
const auto seg = segments_lanes<N>(std::get<Party>(slot.keys), knots, wrap_party, pi);
std::vector<std::vector<uint64_t>> out(knots.size(), std::vector<uint64_t>(N, 0));
for (std::size_t m = 0; m < N; ++m)
for (std::size_t i = 0; i < knots.size(); ++i)
out[i][m] = seg[m][i];
return out;
}
template <std::size_t Party, typename Keys, typename InputT>
std::vector<std::vector<uint64_t>> lane_segments(
const Keys & keys,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, 2>> & wrap, dpf::proof_token * pi)
{
return std::visit([&](const auto & slot) -> std::vector<std::vector<uint64_t>> {
using Slot = std::decay_t<decltype(slot)>;
if constexpr (std::is_same_v<Slot, std::monostate>)
throw std::invalid_argument("offset comparison: missing key");
else
return segments_of_slot<Party>(slot, knots, wrap, pi);
}, keys);
}
inline void replicate_proof(dpf::proof_token * tokens, std::size_t n)
{
if (tokens == nullptr || n == 0)
return;
for (std::size_t m = 1; m < n; ++m)
tokens[m] = tokens[0];
}
template <typename Key, typename InputT>
std::vector<uint64_t> segments_of(const Key & key, const std::vector<InputT> & knots,
uint64_t wrap_share, dpf::proof_token * pi = nullptr)
{
const std::size_t n = knots.size();
if (n == 1)
{
// One-segment short circuit (domains ≥ 63 bits): no prefix walk.
// Bind leaf / value words so tokens are non-zero and verify.
if (pi != nullptr)
{
dpf::detail::vdpf::init_proof(*pi, key);
dpf::detail::vdpf::fold_output_binding(*pi, key);
}
return std::vector<uint64_t>{wrap_share & key.cmp().mask};
}
std::vector<uint64_t> prefix(n);
signed_prefix_parities_into(key, knots.data(), n, prefix.data(), pi);
return segments_from_prefixes(prefix, wrap_share, key.cmp().mask);
}
template <typename T>
HEDLEY_NO_THROW
int64_t math_lift(T value) noexcept
{
if constexpr (std::is_signed_v<T>)
return static_cast<int64_t>(value);
else
return static_cast<int64_t>(lift(value));
}
template <typename T>
HEDLEY_NO_THROW
T domain_min() noexcept
{
if constexpr (std::is_signed_v<T>)
return std::numeric_limits<T>::min();
else
return T{0};
}
/// @brief Public center-space cut where `center + eta` crosses the domain end.
/// @details Empty when that cut is outside the domain, including `eta == 0`.
/// @tparam T value type
/// @param eta public offset `eta = x - r`
/// @return Public center-space cut where `center + eta` crosses the domain end
template <typename T>
HEDLEY_NO_THROW
std::optional<T> carry_threshold(T eta) noexcept
{
constexpr unsigned bits = dpf::utils::bitlength_of_v<T>;
if constexpr (bits > 62)
return std::nullopt;
else
{
const int64_t mod = int64_t{1} << bits;
const int64_t half = mod >> 1;
const int64_t ez = math_lift(eta);
if constexpr (std::is_signed_v<T>)
{
if (ez > 0)
return static_cast<T>(half - ez);
if (ez < 0)
return static_cast<T>(-half - ez);
return std::nullopt;
}
else
{
if (ez == 0)
return std::nullopt;
return static_cast<T>(mod - ez);
}
}
}
/// @brief `center + kappa` is the wrapped representative, as a mathematical integer.
/// @tparam T value type
/// @param left left endpoint of the piece, in the input group
/// @param eta public offset `eta = x - r`
/// @return `center + kappa` is the wrapped representative, as a mathematical integer
template <typename T>
HEDLEY_NO_THROW
int64_t kappa_for(T left, T eta) noexcept
{
constexpr unsigned bits = dpf::utils::bitlength_of_v<T>;
const int64_t ez = math_lift(eta);
if constexpr (bits > 62)
return ez;
else
{
const int64_t mod = int64_t{1} << bits;
const int64_t left_i = math_lift(left);
if constexpr (std::is_signed_v<T>)
{
const int64_t half = mod >> 1;
if (ez > 0 && left_i >= half - ez)
return ez - mod;
if (ez < 0 && left_i < -half - ez)
return ez + mod;
return ez;
}
else
{
if (ez != 0 && left_i >= mod - ez)
return ez - mod;
return ez;
}
}
}
template <std::size_t Degree, typename InputT>
int piece_index(InputT point, const std::vector<InputT> & sorted_knots)
{
const std::size_t n = sorted_knots.size();
if (n <= 1)
return 0;
for (std::size_t i = 0; i + 1 < n; ++i)
{
if (point >= sorted_knots[i] && point < sorted_knots[i + 1])
return static_cast<int>(i);
}
return static_cast<int>(n - 1);
}
template <std::size_t Degree>
std::array<uint64_t, Degree + 1> binomial_coefficients(
const std::array<uint64_t, Degree + 1> & a, uint64_t center_limb)
{
std::array<uint64_t, Degree + 1> c{};
uint64_t center_pow[Degree + 1];
center_pow[0] = 1;
for (std::size_t m = 1; m <= Degree; ++m)
center_pow[m] = center_pow[m - 1] * center_limb;
for (std::size_t m = 0; m <= Degree; ++m)
{
for (std::size_t k = 0; k <= m; ++k)
c[k] += a[m] * binom[m][k] * center_pow[m - k];
}
return c;
}
template <std::size_t Degree, typename InputT>
struct prepared_piece
{
InputT knot{};
std::array<uint64_t, Degree + 1> coeff{};
int64_t kappa = 0;
};
template <std::size_t Degree, typename InputT>
void insert_cut(std::vector<shifted_piece<Degree, InputT>> & rows, InputT point)
{
for (const auto & row : rows)
{
if (row.knot == point)
return;
}
std::vector<InputT> knots;
knots.reserve(rows.size());
for (const auto & row : rows)
knots.push_back(row.knot);
const int hot = piece_index<Degree>(point, knots);
shifted_piece<Degree, InputT> extra;
extra.knot = point;
extra.coeff = rows[static_cast<std::size_t>(hot)].coeff;
rows.push_back(std::move(extra));
std::sort(rows.begin(), rows.end(),
[](const shifted_piece<Degree, InputT> & a, const shifted_piece<Degree, InputT> & b) {
return a.knot < b.knot;
});
}
template <std::size_t Degree, typename InputT>
std::vector<prepared_piece<Degree, InputT>> prepare_pieces(
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
InputT eta)
{
auto rows = shift_and_sort<Degree>(knots, coeff, eta);
constexpr unsigned bits = dpf::utils::bitlength_of_v<InputT>;
if (bits <= 62)
{
insert_cut<Degree>(rows, domain_min<InputT>());
if (const auto cut = carry_threshold(eta))
insert_cut<Degree>(rows, *cut);
}
std::vector<prepared_piece<Degree, InputT>> out;
out.reserve(rows.size());
for (const auto & row : rows)
{
prepared_piece<Degree, InputT> piece;
piece.knot = row.knot;
piece.coeff = row.coeff;
piece.kappa = kappa_for(row.knot, eta);
out.push_back(std::move(piece));
}
return out;
}
/// @brief `out[k]` sums to the polynomial at the wrapped input. It is
/// `center^k` times the public binomial coefficient of `kappa`, not a
/// coefficient you Horner-evaluate at `eta`.
/// @tparam Degree degree
/// @param seg per-power segment shares on the refined pieces
/// @param coeff the public coefficient
/// @param kappa public carry of each refined piece
/// @return `out[k]` sums to the polynomial at the wrapped input
template <std::size_t Degree>
std::array<uint64_t, Degree + 1> contributions(
const std::array<std::vector<uint64_t>, Degree + 1> & seg,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
const std::vector<int64_t> & kappa)
{
std::array<uint64_t, Degree + 1> out{};
const std::size_t n = coeff.size();
for (std::size_t i = 0; i < n; ++i)
{
const auto q = binomial_coefficients<Degree>(
coeff[i], static_cast<uint64_t>(kappa[i]));
for (std::size_t k = 0; k <= Degree; ++k)
out[k] += seg[k][i] * q[k];
}
return out;
}
} // namespace offset_horner_detail
/// @brief Both parties' comparison keys and wrap-piece shares for one center.
/// @tparam InputT input domain type
/// @tparam Degree degree
/// @tparam Verifiable when true, keys carry `dpf::verifiable`
template <typename InputT, std::size_t Degree, bool Verifiable>
struct offset_horner_keys
{
static_assert(Degree <= offset_horner_max_degree, "offset horner degree is at most 3");
static_assert(std::is_integral_v<InputT>, "offset horner domain must be an integer group");
static constexpr std::size_t degree = Degree;
static constexpr bool is_verifiable = Verifiable;
using input_type = InputT;
using key_pair = std::conditional_t<Verifiable,
decltype(dpf::make_dpf(std::declval<InputT>(),
dpf::gt(dpf::vec<uint64_t, Degree + 1>{}), dpf::verifiable{})),
decltype(dpf::make_dpf(std::declval<InputT>(),
dpf::gt(dpf::vec<uint64_t, Degree + 1>{})))>;
/// @brief One `gt` at the hidden center. Lane `m` of the payload is `center^m`.
/// `.first` is party 0. The seed spine is stored once.
key_pair keys;
/// @brief Random additive split of `center^m`, indexed `[power][party]`.
std::array<std::array<uint64_t, 2>, Degree + 1> wrap_share{};
};
/// \complexity One `dpf::make_dpf` of a `Degree + 1` lane comparison. `Degree` is at most 3.
/// The seed spine is one key. Value words are `Degree + 1` lanes.
/// \rounds No party interaction.
/// \communication None inside this function. Shipping the returned keys is outside it.
/// \preprocessing One `gt` key and `wrap_share[Degree + 1][2]` words of `uint64_t`.
/// @see grotto::offset_horner_eval
/// @see grotto::offset_poly_eval
template <typename InputT, std::size_t Degree = offset_horner_max_degree>
offset_horner_keys<InputT, Degree, false> make_offset_horner_keys(InputT center)
{
using namespace offset_horner_detail;
uint64_t payload[Degree + 1];
fill_payloads<Degree>(lift(center), payload);
offset_horner_keys<InputT, Degree, false> mat{
make_power_key<InputT, Degree>(center, payload, std::false_type{}),
{}};
for (std::size_t m = 0; m <= Degree; ++m)
{
const uint64_t blind = dpf::uniform_sample<uint64_t>();
mat.wrap_share[m][0] = blind;
mat.wrap_share[m][1] = payload[m] - blind;
}
return mat;
}
/// \complexity One `dpf::make_dpf` of a `Degree + 1` lane comparison, with proof tokens. `Degree` is at most 3.
/// \rounds No party interaction.
/// \communication None inside this function.
/// \preprocessing One verifiable `gt` key and `wrap_share[Degree + 1][2]` words of `uint64_t`.
/// @see grotto::offset_horner_eval
template <typename InputT, std::size_t Degree = offset_horner_max_degree>
offset_horner_keys<InputT, Degree, true> make_offset_horner_keys(InputT center,
dpf::verifiable)
{
using namespace offset_horner_detail;
uint64_t payload[Degree + 1];
fill_payloads<Degree>(lift(center), payload);
offset_horner_keys<InputT, Degree, true> mat{
make_power_key<InputT, Degree>(center, payload, std::true_type{}),
{}};
for (std::size_t m = 0; m <= Degree; ++m)
{
const uint64_t blind = dpf::uniform_sample<uint64_t>();
mat.wrap_share[m][0] = blind;
mat.wrap_share[m][1] = payload[m] - blind;
}
return mat;
}
/// @brief Cleartext binomial coefficients of the selected refined piece in the
/// variable `center`: Horner at `lift(center)` is the polynomial at the
/// wrapped input.
/// @tparam Degree degree
/// @tparam InputT input domain type
/// @param center hidden comparison point in the input group
/// @param knots public breakpoints, strictly increasing
/// @param coeff the public coefficient
/// @param eta public offset `eta = x - r`
/// @return Cleartext binomial coefficients of the selected refined piece in the variable `center`:
/// Horner at `lift(center)` is the polynomial at the wrapped input
/// \complexity Same piece preparation as the online eval (`Θ(P log P)` sort plus the two optional cuts), then `Θ(Degree)` binomial coefficients on the hot piece. No keys.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Degree, typename InputT>
std::array<uint64_t, Degree + 1> offset_horner_clear_coefficients(
InputT center,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
InputT eta)
{
using namespace offset_horner_detail;
check_knots(knots, coeff.size());
const auto pieces = prepare_pieces<Degree>(knots, coeff, eta);
std::vector<InputT> cuts;
cuts.reserve(pieces.size());
for (const auto & piece : pieces)
cuts.push_back(piece.knot);
const int hot = piece_index<Degree>(center, cuts);
const auto & piece = pieces[static_cast<std::size_t>(hot)];
return binomial_coefficients<Degree>(
piece.coeff, static_cast<uint64_t>(piece.kappa));
}
/// @brief Cleartext value of the selected piece at the wrapped `center + eta`.
/// @tparam Degree degree
/// @tparam InputT input domain type
/// @param center hidden comparison point in the input group
/// @param knots public breakpoints, strictly increasing
/// @param coeff the public coefficient
/// @param eta public offset `eta = x - r`
/// @return Cleartext value of the selected piece at the wrapped `center + eta`
/// \complexity One `offset_horner_clear_coefficients` plus a Horner loop of `Degree + 1` terms (`Degree ≤ 3`).
/// @see grotto::offset_horner_eval
template <std::size_t Degree, typename InputT>
uint64_t offset_horner_clear(
InputT center,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
InputT eta)
{
const auto c = offset_horner_clear_coefficients<Degree>(center, knots, coeff, eta);
return offset_horner_detail::horner_at<Degree>(c, offset_horner_detail::lift(center));
}
/// @brief `Party` selects `.first` or `.second` of each key pair.
/// @tparam Party party index, `0` or `1`
/// @tparam Degree degree
/// @tparam InputT input domain type
/// @tparam KeyPair key pair
/// @param keys the party keys
/// @param wrap_share additive split of the keyed payload, indexed by party
/// @param knots public breakpoints, strictly increasing
/// @param coeff the public coefficient
/// @param eta public offset `eta = x - r`
/// @param tokens proof token folded by the segment walk, or null
/// @return `Party` selects `.first` or `.second` of each key pair
/// @throws std::invalid_argument if the key vector is not the single shared comparison
/// \complexity `prepare_pieces` sorts the knots and may insert two cuts. Then one segment walk of the `Degree + 1` lane payload.
/// Each walk is `signed_prefix_parities` over those `P` endpoints (a DPF path of `bitlength(InputT)` levels, reusing the common prefix).
/// Refined pieces: the knot vector, plus the cuts `prepare` / `prepare_pieces` inserts (domain minimum, and the carry threshold when the input width is at most 62 bits). Call that count `P`.
/// Time is one segment walk. Extra space is the piece vectors, `Θ(P · Degree)` words, with `Degree ≤ 3`.
/// \rounds None. `eta` is an argument; this function does not open it.
/// \communication None.
/// \preprocessing None created here. It reads the one comparison from `make_offset_horner_keys`.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Party, std::size_t Degree, typename InputT, typename KeyPair>
std::array<uint64_t, Degree + 1> offset_horner_coefficient_share(
const std::vector<KeyPair> & keys,
const std::array<std::array<uint64_t, 2>, Degree + 1> & wrap_share,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
InputT eta, dpf::proof_token * tokens = nullptr)
{
static_assert(Party < 2, "offset horner party is 0 or 1");
using namespace offset_horner_detail;
check_knots(knots, coeff.size());
if (keys.size() != 1)
throw std::invalid_argument("offset horner: one comparison key");
const auto pieces = prepare_pieces<Degree>(knots, coeff, eta);
std::vector<InputT> shifted(pieces.size());
std::vector<std::array<uint64_t, Degree + 1>> ordered(pieces.size());
std::vector<int64_t> kappa(pieces.size());
for (std::size_t i = 0; i < pieces.size(); ++i)
{
shifted[i] = pieces[i].knot;
ordered[i] = pieces[i].coeff;
kappa[i] = pieces[i].kappa;
}
std::array<uint64_t, Degree + 1> wrap_party{};
for (std::size_t m = 0; m <= Degree; ++m)
wrap_party[m] = wrap_share[m][Party];
dpf::proof_token * pi = (tokens != nullptr) ? &tokens[0] : nullptr;
const auto & key = std::get<Party>(keys.front());
auto seg = segments_lanes<Degree + 1>(key, shifted, wrap_party, pi);
if (tokens != nullptr)
{
for (std::size_t m = 1; m <= Degree; ++m)
tokens[m] = tokens[0];
}
return contributions<Degree>(seg, ordered, kappa);
}
/// @brief One party's coefficient shares. `Party` is 0 or 1.
/// \complexity `prepare_pieces` sorts the knots and may insert two cuts. Then one segment walk of the `Degree + 1` lane payload.
/// Each walk is `signed_prefix_parities` over those `P` endpoints (a DPF path of `bitlength(InputT)` levels, reusing the common prefix).
/// Refined pieces: the knot vector, plus the cuts `prepare` / `prepare_pieces` inserts (domain minimum, and the carry threshold when the input width is at most 62 bits). Call that count `P`.
/// Time is one segment walk. Extra space is the piece vectors, `Θ(P · Degree)` words, with `Degree ≤ 3`.
/// \rounds None. `eta` is an argument; this function does not open it.
/// \communication None.
/// \preprocessing None created here. It reads the one comparison from `make_offset_horner_keys`.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Party, std::size_t Degree, typename InputT,
bool Verifiable = false>
std::array<uint64_t, Degree + 1> offset_horner_coefficient_share(
const offset_horner_keys<InputT, Degree, Verifiable> & mat,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
InputT eta, dpf::proof_token * tokens = nullptr)
{
std::vector<typename offset_horner_keys<InputT, Degree, Verifiable>::key_pair> keys{
mat.keys};
return offset_horner_coefficient_share<Party, Degree>(
keys, mat.wrap_share, knots, coeff, eta, tokens);
}
/// @brief Both parties' Horner shares from one joint Doerner–Shelat generation.
/// @details `center0 XOR center1` is the comparison point, in geneval's share
/// convention (the signed MSB of `center0` is flipped before the XOR, and
/// flipped back here). `eta` is already public.
/// @tparam Degree degree
/// @tparam InputT input domain type
template <std::size_t Degree, typename InputT>
struct geneval_offset_horner_result
{
static_assert(Degree <= offset_horner_max_degree, "offset horner degree is at most 3");
/// @brief Public offset `eta = x - r`. F_Horner leaks only this.
InputT eta{};
std::array<uint64_t, Degree + 1> coeff0{};
std::array<uint64_t, Degree + 1> coeff1{};
uint64_t value0 = 0;
uint64_t value1 = 0;
std::array<dpf::proof_token, Degree + 1> proof0{};
std::array<dpf::proof_token, Degree + 1> proof1{};
};
/// @brief Logical comparison point for geneval's XOR shares. Matches `make_dpf(P)`
/// when `center1 = P XOR center0` or when `center0 = P` and `center1 = 0`.
/// @details Dealer / joint-simulator helper only. F_Horner does not return the
/// center to the parties; `center^m` stays a shared payload.
/// @tparam InputT input domain type
/// @param center0 party 0 share of the center
/// @param center1 party 1 share of the center
/// @return Logical comparison point for geneval's XOR shares
/// \complexity Two `flip_msb_if_signed_integral` calls and one XOR of the input shares. `Θ(1)`.
/// @note Dealer-side helper. The signed MSB of `center0` is flipped before the XOR and flipped back, matching `make_dpf`.
/// @see grotto::geneval_offset_horner
template <typename InputT>
InputT geneval_offset_horner_center(InputT center0, InputT center1)
{
InputT flipped0 = center0;
dpf::utils::flip_msb_if_signed_integral(flipped0);
InputT mixed = dpf::utils::xor_input_shares(flipped0, center1);
dpf::utils::flip_msb_if_signed_integral(mixed);
return mixed;
}
namespace offset_horner_detail
{
template <std::size_t Degree, typename InputT, typename Rng>
geneval_offset_horner_result<Degree, InputT> geneval_at(
bool arith, InputT center0, InputT center1, InputT center, InputT eta,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
Rng rng)
{
check_knots(knots, coeff.size());
const auto pieces = prepare_pieces<Degree>(knots, coeff, eta);
std::vector<InputT> shifted(pieces.size());
std::vector<std::array<uint64_t, Degree + 1>> ordered(pieces.size());
std::vector<int64_t> kappa(pieces.size());
for (std::size_t i = 0; i < pieces.size(); ++i)
{
shifted[i] = pieces[i].knot;
ordered[i] = pieces[i].coeff;
kappa[i] = pieces[i].kappa;
}
uint64_t payload[Degree + 1];
fill_payloads<Degree>(lift(center), payload);
std::array<std::array<uint64_t, 2>, Degree + 1> wrap{};
std::array<std::vector<uint64_t>, Degree + 1> seg0;
std::array<std::vector<uint64_t>, Degree + 1> seg1;
std::array<dpf::proof_token, Degree + 1> out_proofs0{};
std::array<dpf::proof_token, Degree + 1> out_proofs1{};
using Vec = dpf::vec<uint64_t, Degree + 1>;
const auto beta = payload_vec<InputT, Degree>(payload);
auto open_lanes = [&](auto && keys) {
const auto & k0 = keys.first;
const auto & k1 = keys.second;
const uint64_t mask = k0.cmp().mask;
dpf::detail::vdpf::init_proof(out_proofs0[0], k0);
dpf::detail::vdpf::init_proof(out_proofs1[0], k1);
auto path0 = dpf::make_basic_path_memoizer(k0);
auto path1 = dpf::make_basic_path_memoizer(k1);
std::vector<Vec> pref0(shifted.size());
std::vector<Vec> pref1(shifted.size());
for (std::size_t i = 0; i < shifted.size(); ++i)
{
auto tx0 = k0.offset_x(shifted[i]);
auto tx1 = k1.offset_x(shifted[i]);
dpf::utils::flip_msb_if_signed_integral(tx0);
dpf::utils::flip_msb_if_signed_integral(tx1);
pref0[i] = dpf::detail::incr::eval_payload_path_sum<Vec>(
k0, tx0, path0, false, 0, &out_proofs0[0]);
pref1[i] = dpf::detail::incr::eval_payload_path_sum<Vec>(
k1, tx1, path1, false, 0, &out_proofs1[0]);
}
dpf::detail::vdpf::fold_output_binding(out_proofs0[0], k0);
dpf::detail::vdpf::fold_output_binding(out_proofs1[0], k1);
for (std::size_t m = 1; m <= Degree; ++m)
{
out_proofs0[m] = out_proofs0[0];
out_proofs1[m] = out_proofs1[0];
}
for (std::size_t m = 0; m <= Degree; ++m)
{
const uint64_t blind = dpf::uniform_sample<uint64_t>();
wrap[m][0] = blind;
wrap[m][1] = payload[m] - blind;
std::vector<uint64_t> lane0(shifted.size());
std::vector<uint64_t> lane1(shifted.size());
for (std::size_t i = 0; i < shifted.size(); ++i)
{
lane0[i] = pref0[i][m];
lane1[i] = pref1[i][m];
}
seg0[m] = segments_from_prefixes(lane0, wrap[m][0], mask);
seg1[m] = segments_from_prefixes(lane1, wrap[m][1], mask);
}
};
if (arith)
open_lanes(dpf::make_dpf_doerner_shelat(dpf::arith_input, center0, center1,
rng, dpf::gt(beta), dpf::verifiable{}));
else
open_lanes(dpf::make_dpf_doerner_shelat(center0, center1,
rng, dpf::gt(beta), dpf::verifiable{}));
geneval_offset_horner_result<Degree, InputT> out;
out.eta = eta;
out.coeff0 = contributions<Degree>(seg0, ordered, kappa);
out.coeff1 = contributions<Degree>(seg1, ordered, kappa);
out.proof0 = out_proofs0;
out.proof1 = out_proofs1;
for (uint64_t term : out.coeff0)
out.value0 += term;
for (uint64_t term : out.coeff1)
out.value1 += term;
return out;
}
template <std::size_t Degree, typename InputT, typename Rng>
geneval_offset_horner_result<Degree, InputT> geneval_at(
InputT center0, InputT center1, InputT center, InputT eta,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
Rng rng)
{
return geneval_at<Degree>(false, center0, center1, center, eta, knots, coeff,
std::move(rng));
}
} // namespace offset_horner_detail
/// @brief Geneval-style offset Horner. The center is XOR-shared as in `geneval_point`.
/// @details Comparison keys are opened with the same local Doerner–Shelat protocol
/// geneval uses for its correction words. The value dot uses the per-piece
/// carry shift and is local.
/// A comparison value word is written on every level of the secret path.
/// The seed spine is generated once for the whole power vector.
/// @tparam Degree degree
/// @tparam InputT input domain type
/// @tparam Rng rng
/// @param center0 party 0 share of the center
/// @param center1 party 1 share of the center
/// @param eta public offset `eta = x - r`
/// @param knots public breakpoints, strictly increasing
/// @param coeff the public coefficient
/// @param rng the Doerner–Shelat randomness tapes
/// @return Geneval-style offset Horner
/// \complexity One Doerner–Shelat generation of a `Degree + 1` lane comparison (`Degree ≤ 3`), then the same local piece dot as `offset_horner_eval`.
/// The dot is `Θ(P · Degree)` after those calls. `P` is the refined piece count.
/// @note Rounds and bandwidth of each `geneval_cmp` live in the DPF headers, not in this function, so they are not stated here.
/// \preprocessing The randomness object the caller passes (`Rng`). This function also samples one `uint64_t` blind per power.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Degree = offset_horner_max_degree,
typename InputT,
typename Rng>
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
InputT center0, InputT center1, InputT eta,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
Rng rng)
{
const InputT center = geneval_offset_horner_center(center0, center1);
return offset_horner_detail::geneval_at<Degree>(
center0, center1, center, eta, knots, coeff, std::move(rng));
}
/// \complexity One Doerner–Shelat generation of a `Degree + 1` lane comparison (`Degree ≤ 3`), then the same local piece dot as `offset_horner_eval`.
/// The dot is `Θ(P · Degree)` after those calls. `P` is the refined piece count.
/// @note Rounds and bandwidth of each `geneval_cmp` live in the DPF headers, not in this function, so they are not stated here.
/// \preprocessing The randomness object the caller passes (`Rng`). This function also samples one `uint64_t` blind per power.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Degree = offset_horner_max_degree, typename InputT>
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
InputT center0, InputT center1, InputT eta,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff)
{
using block = typename dpf::prg::aes128::block_type;
HEDLEY_PRAGMA(GCC diagnostic push)
HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes")
dpf::ds_randomness<block (*)(), dpf::detail::urandom_pad_rng> rng{
dpf::uniform_sample<block>, {}};
HEDLEY_PRAGMA(GCC diagnostic pop)
return geneval_offset_horner<Degree>(
center0, center1, eta, knots, coeff, std::move(rng));
}
/// @brief Additive shares of the center: `center0 + center1` is the comparison point.
/// @tparam Degree degree
/// @tparam InputT input domain type
/// @tparam Rng rng
/// @param center0 party 0 share of the center
/// @param center1 party 1 share of the center
/// @param eta public offset `eta = x - r`
/// @param knots public breakpoints, strictly increasing
/// @param coeff the public coefficient
/// @param rng the Doerner–Shelat randomness tapes
/// @return Additive shares of the center: `center0 + center1` is the comparison point
/// \complexity One Doerner–Shelat generation of a `Degree + 1` lane comparison (`Degree ≤ 3`), then the same local piece dot as `offset_horner_eval`.
/// The dot is `Θ(P · Degree)` after those calls. `P` is the refined piece count.
/// @note Rounds and bandwidth of each `geneval_cmp` live in the DPF headers, not in this function, so they are not stated here.
/// \preprocessing The randomness object the caller passes (`Rng`). This function also samples one `uint64_t` blind per power.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Degree = offset_horner_max_degree,
typename InputT,
typename Rng>
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
dpf::arith_input_t, InputT center0, InputT center1, InputT eta,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
Rng rng)
{
const InputT center = offset_horner_group_add(center0, center1);
return offset_horner_detail::geneval_at<Degree>(
true, center0, center1, center, eta, knots, coeff, std::move(rng));
}
/// @brief Additive shares of the input `x` and the mask `r`.
/// @details Opens only `eta = (x0 - r0) + (x1 - r1)`. Passes additive shares
/// of `center = 2r` (`2·r0`, `2·r1`) to arithmetic `geneval_cmp`.
/// The clear center is used only to plant shared `center^m` payloads;
/// it is not returned. F_Horner leaks only `eta`.
/// @tparam Degree degree
/// @tparam InputT input domain type
/// @tparam Rng rng
/// @param x0 party 0 share of the input
/// @param x1 party 1 share of the input
/// @param r0 the party 0's share of the mask
/// @param r1 the party 1's share of the mask
/// @param knots public breakpoints, strictly increasing
/// @param coeff the public coefficient
/// @param rng the Doerner–Shelat randomness tapes
/// @return Both parties' Horner shares and the public `eta`
/// \complexity One Doerner–Shelat generation of a `Degree + 1` lane comparison (`Degree ≤ 3`), then the same local piece dot as `offset_horner_eval`.
/// The dot is `Θ(P · Degree)` after those calls. `P` is the refined piece count.
/// @note Rounds and bandwidth of each `geneval_cmp` live in the DPF headers, not in this function, so they are not stated here.
/// \preprocessing The randomness object the caller passes (`Rng`). This function also samples one `uint64_t` blind per power.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Degree = offset_horner_max_degree,
typename InputT,
typename Rng>
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
InputT x0, InputT x1, InputT r0, InputT r1,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
Rng rng)
{
// Open only eta; do not form clear x or r.
const InputT eta = offset_horner_group_add(
offset_horner_group_sub(x0, r0),
offset_horner_group_sub(x1, r1));
const InputT center0 = offset_horner_group_add(r0, r0);
const InputT center1 = offset_horner_group_add(r1, r1);
// Payload planting for the joint simulator; not returned to parties.
const InputT center = offset_horner_group_add(center0, center1);
return offset_horner_detail::geneval_at<Degree>(
true, center0, center1, center, eta, knots, coeff, std::move(rng));
}
/// \complexity One Doerner–Shelat generation of a `Degree + 1` lane comparison (`Degree ≤ 3`), then the same local piece dot as `offset_horner_eval`.
/// The dot is `Θ(P · Degree)` after those calls. `P` is the refined piece count.
/// @note Rounds and bandwidth of each `geneval_cmp` live in the DPF headers, not in this function, so they are not stated here.
/// \preprocessing The randomness object the caller passes (`Rng`). This function also samples one `uint64_t` blind per power.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Degree = offset_horner_max_degree, typename InputT>
geneval_offset_horner_result<Degree, InputT> geneval_offset_horner(
InputT x0, InputT x1, InputT r0, InputT r1,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff)
{
using block = typename dpf::prg::aes128::block_type;
HEDLEY_PRAGMA(GCC diagnostic push)
HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes")
dpf::ds_randomness<block (*)(), dpf::detail::urandom_pad_rng> rng{
dpf::uniform_sample<block>, {}};
HEDLEY_PRAGMA(GCC diagnostic pop)
return geneval_offset_horner<Degree>(
x0, x1, r0, r1, knots, coeff, std::move(rng));
}
/// @brief One party's share of the cubic at the wrapped `center + eta`.
/// @details Sum the coefficient shares; they are already scaled by `center^k`.
/// @tparam Party party index, `0` or `1`
/// @tparam Degree degree
/// @tparam InputT input domain type
/// @param mat dealer keys
/// @param knots public breakpoints, strictly increasing
/// @param coeff the public coefficient
/// @param eta public offset `eta = x - r`
/// @param tokens proof token folded by the segment walk, or null
/// @return One party's share of the cubic at the wrapped `center + eta`
/// \complexity `prepare_pieces` sorts the knots and may insert two cuts. Then one segment walk of the `Degree + 1` lane payload.
/// Each walk is `signed_prefix_parities` over those `P` endpoints (a DPF path of `bitlength(InputT)` levels, reusing the common prefix).
/// Refined pieces: the knot vector, plus the cuts `prepare` / `prepare_pieces` inserts (domain minimum, and the carry threshold when the input width is at most 62 bits). Call that count `P`.
/// Time is one segment walk. Extra space is the piece vectors, `Θ(P · Degree)` words, with `Degree ≤ 3`.
/// \rounds None. `eta` is an argument; this function does not open it.
/// \communication None.
/// \preprocessing None created here. It reads the one comparison from `make_offset_horner_keys`.
/// @see grotto::offset_poly_eval
/// @see grotto::offset_jet_eval
template <std::size_t Party, std::size_t Degree, typename InputT,
bool Verifiable = false>
uint64_t offset_horner_eval(
const offset_horner_keys<InputT, Degree, Verifiable> & mat,
const std::vector<InputT> & knots,
const std::vector<std::array<uint64_t, Degree + 1>> & coeff,
InputT eta, dpf::proof_token * tokens = nullptr)
{
const auto shares = offset_horner_coefficient_share<Party, Degree>(
mat, knots, coeff, eta, tokens);
uint64_t value = 0;
for (uint64_t term : shares)
value += term;
return value;
}
} // namespace grotto
#endif // LIBDPF_INCLUDE_GROTTO_OFFSET_HORNER_HPP__