Annotate noexcept and constexpr with HEDLEY, and add interval containment, ChaCha, and the dyadic range tables.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Ryan Henry 2026-09-24 20:44:07 -06:00
parent 875f09fec1
commit 0d8a5a8131
97 changed files with 9212 additions and 1159 deletions

View file

@ -1,11 +1,12 @@
/// @file dpf/doerner_shelat.hpp
/// @brief Doerner–Shelat generation of a dealer DPF key.
/// @details Two XOR shares of the point are walked level by level. Correction
/// words, advice bits, seeds, and leaves are the ones `make_dpf`
/// would emit for the XOR of those shares, the same roots, and the
/// same beaver coins. Beaver pads used to hide the path bit cancel
/// and are not part of the key. Pad randomness must not come from
/// `uniform_fill` if the beaver tape is being matched.
/// @details Two shares of the point are walked level by level — XOR shares by
/// default, or additive shares when tagged with `arith_input`.
/// Correction words, advice bits, seeds, and leaves are the ones
/// `make_dpf` would emit for the reconstructed point, the same roots,
/// and the same beaver coins. Beaver pads used to hide the path bit
/// cancel and are not part of the key. Pad randomness must not come
/// from `uniform_fill` if the beaver tape is being matched.
/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors)
/// @license Released under a GNU General Public v2.0 (GPLv2) license;
/// see [LICENSE.md](@ref license) for details.
@ -28,6 +29,14 @@
namespace dpf
{
/// Tag: Doerner–Shelat / geneval takes additive shares of the point
/// (`x0 + x1` in the input ring). Default calls take XOR shares.
struct arith_input_t
{
};
inline constexpr arith_input_t arith_input{};
/// Roots and the Beaver-pad stream for one Doerner–Shelat generation.
/// `root` is called twice, same as `make_dpf`: party 0 clears the low bit of
/// the first sample, party 1 sets the low bit of the second.
@ -86,12 +95,14 @@ struct ds_and_shares
simde__m128i z1;
};
HEDLEY_NO_THROW
HEDLEY_ALWAYS_INLINE
simde__m128i ds_xor(simde__m128i a, simde__m128i b) noexcept
{
return simde_mm_xor_si128(a, b);
}
HEDLEY_NO_THROW
HEDLEY_ALWAYS_INLINE
simde__m128i ds_gate(uint8_t bit, simde__m128i block) noexcept
{
@ -128,6 +139,7 @@ ds_and_pads ds_sample_and(PadRng & pad)
return p;
}
HEDLEY_NO_THROW
HEDLEY_ALWAYS_INLINE
simde__m128i ds_cw_share(simde__m128i L, simde__m128i R, uint8_t my_bit,
const ds_cw_party & mine, const ds_blind & their) noexcept
@ -144,6 +156,7 @@ simde__m128i ds_cw_share(simde__m128i L, simde__m128i R, uint8_t my_bit,
return out;
}
HEDLEY_NO_THROW
inline void ds_cw_blinds(const ds_cw_pads & p,
simde__m128i L0, simde__m128i R0, uint8_t bit0,
simde__m128i L1, simde__m128i R1, uint8_t bit1,
@ -155,6 +168,7 @@ inline void ds_cw_blinds(const ds_cw_pads & p,
b1.msg = ds_xor(ds_xor(L1, R1), p.p1.rand);
}
HEDLEY_NO_THROW
inline simde__m128i ds_cw_outs(const ds_cw_pads & p,
simde__m128i L0, simde__m128i R0, uint8_t bit0,
simde__m128i L1, simde__m128i R1, uint8_t bit1,
@ -165,6 +179,7 @@ inline simde__m128i ds_cw_outs(const ds_cw_pads & p,
ds_cw_share(L1, R1, bit1, p.p1, b0));
}
HEDLEY_NO_THROW
inline uint8_t ds_open_advice(simde__m128i L0, simde__m128i R0, uint8_t bit0,
simde__m128i L1, simde__m128i R1, uint8_t bit1) noexcept
{
@ -177,6 +192,7 @@ inline uint8_t ds_open_advice(simde__m128i L0, simde__m128i R0, uint8_t bit0,
return static_cast<uint8_t>((t1 << 1) | (t0 & 1u));
}
HEDLEY_NO_THROW
inline void ds_next_terms(simde__m128i L, simde__m128i R, uint8_t advice,
simde__m128i cw, uint8_t tpack, simde__m128i & M, simde__m128i & base) noexcept
{
@ -190,6 +206,7 @@ inline void ds_next_terms(simde__m128i L, simde__m128i R, uint8_t advice,
base = (advice & 1u) ? ds_xor(L, cw_base) : L;
}
HEDLEY_NO_THROW
inline ds_and_shares ds_and_open(const ds_and_pads & p, simde__m128i M,
uint8_t b_recv) noexcept
{
@ -202,6 +219,7 @@ inline ds_and_shares ds_and_open(const ds_and_pads & p, simde__m128i M,
return z;
}
HEDLEY_NO_THROW
inline simde__m128i ds_deliver(uint8_t b_exp, simde__m128i base, simde__m128i M,
const ds_and_shares & z) noexcept
{
@ -249,6 +267,11 @@ struct ds_cmp_gen_state
bool track_coeff = false;
uint64_t Va1 = 0;
uint64_t last_vcw_coeff = 0;
cmp_kind kind = cmp_kind::lt;
bool paint = false;
std::size_t length_bits = 0;
paint_callback paint_cb = nullptr;
const void * paint_ctx = nullptr;
};
/// Local joint simulation: today's `ds_cw_outs` / `ds_open_advice` / `ds_and_open`.
@ -288,6 +311,7 @@ struct local_cw_protocol
}
/// Open CW + advice only (AND pads stay in `blinds` for a later open).
HEDLEY_NO_THROW
std::pair<simde__m128i, uint8_t> open_cw(const ds_level_blinds & b) noexcept
{
return {ds_cw_outs(b.cwp, b.L0, b.R0, b.bit0, b.L1, b.R1, b.bit1,
@ -297,6 +321,7 @@ struct local_cw_protocol
/// Open the public value CW for this level (local: clear convert+make_value_cw).
/// MPC backends open additive shares of the same word.
HEDLEY_NO_THROW
uint64_t open_value_cw(const ds_level_blinds & b, uint8_t adv0, uint8_t adv1,
int ai, uint64_t & Va, uint64_t beta, uint64_t mask) noexcept
{
@ -304,6 +329,16 @@ struct local_cw_protocol
adv1, ai, Va, beta, mask);
}
/// Open a path-paint value CW. `plant` is the scaled lose-subtree constant.
HEDLEY_NO_THROW
uint64_t open_planted_cw(const ds_level_blinds & b, uint8_t adv0, uint8_t adv1,
int ai, uint64_t & Va, uint64_t plant, uint64_t mask) noexcept
{
return dcf_impl::make_value_cw_planted(b.L0, b.R0, b.L1, b.R1, adv0,
adv1, ai, Va, plant, mask);
}
HEDLEY_NO_THROW
ds_and_shares open_and(const ds_and_pads & p, simde__m128i M,
uint8_t b_recv) noexcept
{
@ -313,6 +348,7 @@ struct local_cw_protocol
/// Open the final comparison leaf CW. Wraps `make_final_cw` so the
/// Doerner–Shelat gen does not call it directly on reconstructed seeds;
/// an MPC backend would open additive shares of the same word.
HEDLEY_NO_THROW
uint64_t open_final_cw(simde__m128i s0, simde__m128i s1, uint8_t t1,
uint64_t Va, uint64_t mask, uint64_t on_path) noexcept
{
@ -323,18 +359,81 @@ struct local_cw_protocol
/// the shared root sampler so the blind matches the dealer's; an MPC
/// backend would instead pull a group-width element from the pad stream.
template <typename BlockSampler>
HEDLEY_NO_THROW
uint64_t sample_addend_blind(uint64_t mask, BlockSampler && sample) noexcept
{
return dcf_impl::sample_addend_blind(mask,
std::forward<BlockSampler>(sample));
}
/// Majority of three bits (next carry of a full adder).
HEDLEY_NO_THROW
static constexpr uint8_t majority(uint8_t a, uint8_t b, uint8_t c) noexcept
{
return static_cast<uint8_t>((a & b) | (a & c) | (b & c));
}
/// One additive digit: sum bit `a XOR b XOR cin`, carry out = majority.
HEDLEY_NO_THROW
static constexpr uint8_t open_sum_bit(uint8_t a, uint8_t b, uint8_t cin,
uint8_t & cout) noexcept
{
cout = majority(a, b, cin);
return static_cast<uint8_t>(a ^ b ^ cin);
}
/// Reconstruct `a0 + a1` via a LSB→MSB carry chain, then flip the MSB
/// when the domain is signed — matching `make_dpf` on the sum. The call
/// site never forms the sum; an MPC backend would open the same bits.
template <typename InputT>
InputT open_arith_point(InputT a0, InputT a1) const
{
constexpr auto to_int = utils::to_integral_type<InputT>{};
using FromI = typename utils::make_from_integral_value<InputT>::integral_type;
using U = std::make_unsigned_t<FromI>;
const U u0 = static_cast<U>(to_int(a0));
const U u1 = static_cast<U>(to_int(a1));
U sum = 0;
uint8_t carry = 0;
constexpr std::size_t nbits = utils::bitlength_of_v<InputT>;
for (std::size_t i = 0; i < nbits; ++i)
{
const uint8_t b0 = static_cast<uint8_t>((u0 >> i) & U{1});
const uint8_t b1 = static_cast<uint8_t>((u1 >> i) & U{1});
const uint8_t s = open_sum_bit(b0, b1, carry, carry);
sum = static_cast<U>(sum | (static_cast<U>(s) << i));
}
InputT out = utils::make_from_integral_value<InputT>{}(
static_cast<FromI>(sum));
utils::flip_msb_if_signed_integral(out);
return out;
}
/// Encode shares for the XOR-style CW walk. XOR mode flips party 0's MSB
/// (linear over XOR). Arithmetic mode opens the sum (carry + signed MSB)
/// and returns `(alpha, 0)` so the walk matches `make_dpf(alpha)`.
template <typename InputT>
void encode_walk_shares(InputT & x0, InputT & x1, bool arith) const
{
if (arith)
{
const InputT alpha = open_arith_point(x0, x1);
x0 = alpha;
x1 = InputT{};
}
else
{
utils::flip_msb_if_signed_integral(x0);
}
}
/// Open a group of leaf correction words for one prefix group. In this
/// local joint simulation both XOR shares of the point are present, so the
/// point is reconstructed *inside* the protocol and handed to `leaf_fn`
/// (which runs `make_leaves` for the group). The Doerner–Shelat gen never
/// forms `x = x0 ^ x1` at its own call site; an MPC backend would instead
/// run a per-group leaf CW exchange that never reveals `x`.
/// run a per-group leaf CW exchange that never reveals `x`. After
/// `encode_walk_shares`, arithmetic inputs are already `(alpha, 0)`.
template <typename InputT, typename LeafFn>
void open_leaf_group(InputT x0, InputT x1, LeafFn && leaf_fn)
{
@ -351,6 +450,7 @@ struct ds_gen_state
NodeT root0;
NodeT root1;
HEDLEY_NO_THROW
void init(NodeT r0, NodeT r1) noexcept
{
root0 = r0;
@ -361,9 +461,13 @@ struct ds_gen_state
home[1] = 1;
}
HEDLEY_NO_THROW
NodeT & seed0() noexcept { return inbox[home[0]]; }
HEDLEY_NO_THROW
NodeT & seed1() noexcept { return inbox[home[1]]; }
HEDLEY_NO_THROW
const NodeT & seed0() const noexcept { return inbox[home[0]]; }
HEDLEY_NO_THROW
const NodeT & seed1() const noexcept { return inbox[home[1]]; }
};
@ -403,15 +507,37 @@ void ds_advance_level(ds_gen_state<NodeT> & st, InputT x0, InputT x1,
{
const int ai = static_cast<int>(
(cmp->thresh >> (cmp->nbits - 1 - level)) & 1);
*value_cw_out = proto.open_value_cw(blinds, adv0, adv1, ai, cmp->Va,
cmp->beta, cmp->mask);
if (cmp->track_coeff)
if (cmp->paint)
{
// Affine coefficient: same level with β = 1 on a parallel Va.
const uint64_t v1 = proto.open_value_cw(blinds, adv0, adv1, ai,
cmp->Va1, 1ULL, cmp->mask);
cmp->last_vcw_coeff =
(v1 + dcf_impl::neg_m(*value_cw_out, cmp->mask)) & cmp->mask;
const uint64_t unit = dcf_impl::paint_unit(cmp->kind, level,
cmp->thresh, cmp->nbits, cmp->length_bits, false,
cmp->paint_cb, cmp->paint_ctx);
const uint64_t plant = dcf_impl::scale_plant(unit, cmp->beta,
cmp->mask);
*value_cw_out = proto.open_planted_cw(blinds, adv0, adv1, ai,
cmp->Va, plant, cmp->mask);
if (cmp->track_coeff)
{
const uint64_t plant1 = dcf_impl::scale_plant(unit, 1ULL,
cmp->mask);
const uint64_t v1 = proto.open_planted_cw(blinds, adv0, adv1,
ai, cmp->Va1, plant1, cmp->mask);
cmp->last_vcw_coeff =
(v1 + dcf_impl::neg_m(*value_cw_out, cmp->mask)) & cmp->mask;
}
}
else
{
*value_cw_out = proto.open_value_cw(blinds, adv0, adv1, ai, cmp->Va,
cmp->beta, cmp->mask);
if (cmp->track_coeff)
{
// Affine coefficient: same level with β = 1 on a parallel Va.
const uint64_t v1 = proto.open_value_cw(blinds, adv0, adv1, ai,
cmp->Va1, 1ULL, cmp->mask);
cmp->last_vcw_coeff =
(v1 + dcf_impl::neg_m(*value_cw_out, cmp->mask)) & cmp->mask;
}
}
}
@ -475,14 +601,14 @@ template <typename InteriorPRG,
typename ...OutputTs,
typename RootSampler,
typename CwProtocol>
auto make_dpf_doerner_shelat_impl(InputT x0, InputT x1,
auto make_dpf_doerner_shelat_impl(bool arith, InputT x0, InputT x1,
RootSampler & root_sampler, CwProtocol & proto, OutputT && y,
OutputTs && ...ys)
{
static_assert(!dpf::is_wildcard_v<InputT>,
"Doerner–Shelat gen takes XOR shares of a concrete point");
"Doerner–Shelat gen takes shares of a concrete point");
static_assert(!dpf::is_secret_share_v<InputT>,
"Doerner–Shelat: pass additive_share of xor_wrapper, or raw XOR shares");
"Doerner–Shelat: pass additive_share of xor_wrapper, or raw shares");
static_assert(sizeof(typename InteriorPRG::block_type) == sizeof(simde__m128i),
"Doerner–Shelat gen uses the AES-block interior node");
@ -492,7 +618,7 @@ auto make_dpf_doerner_shelat_impl(InputT x0, InputT x1,
using input_type = typename dpf_type::input_type;
constexpr auto depth = dpf_type::depth;
utils::flip_msb_if_signed_integral(x0);
proto.encode_walk_shares(x0, x1, arith);
const node root0 = dpf::unset_lo_bit(static_cast<node>(root_sampler()));
const node root1 = dpf::set_lo_bit(static_cast<node>(root_sampler()));