Checkpoint the party/runtime stack before share-program and malicious-mode work.

Ship the TLS mesh, composer, Beaver/Yao/leaf MPC, prep/online paths, apps, and docs so the tree is pushable before elevating share_expr, security_mode, and prep resume.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Ryan Henry 2026-09-28 05:59:19 -06:00
parent 695f8e84f7
commit 0d22946a0e
1835 changed files with 170291 additions and 2849 deletions

View file

@ -1,12 +1,16 @@
/// @file dpf/doerner_shelat.hpp
/// @brief Doerner–Shelat generation of a dealer DPF key.
/// @details Two shares of the point are walked level by level — XOR shares by
/// default, or additive shares when tagged with `arith_input`.
/// default. `arith_input` holds additive shares of the point. A beaver
/// ripple-carry converts them to XOR shares of the sum bits, and that
/// sharing is what the walk consumes. The sum is not opened.
/// Correction words, advice bits, seeds, and leaves are the ones
/// `make_dpf` would emit for the reconstructed point, the same roots,
/// and the same beaver coins. Beaver pads used to hide the path bit
/// cancel and are not part of the key. Pad randomness must not come
/// from `uniform_fill` if the beaver tape is being matched.
/// `make_dpf` would emit for that point, the same roots, and the same
/// beaver coins. Beaver pads used to hide a path bit cancel and are
/// not part of the key. Pad randomness must not come from
/// `uniform_fill` if the beaver tape is being matched.
/// @note Following Jack Doerner and abhi shelat, CCS 2017 (ePrint 2017/827): one correction word opened per level from shares of the point.
/// @note Guo, Yang, Wang, Zhang, Xie, Zhang, and Liu (ePrint 2022/1431, §5.2) generate a DPF in the COT/OLE hybrid in n+3 rounds, with no beaver-pad dealer. This header uses that dealer tape and one opening round per level.
/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors)
/// @license Released under a GNU General Public v2.0 (GPLv2) license;
/// see [LICENSE.md](@ref license) for details.
@ -16,16 +20,23 @@
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <type_traits>
#include <utility>
#include <vector>
#include "hedley/hedley.h"
#include "simde/simde/x86/avx2.h"
#include "dpf/dpf_key.hpp"
#include "dpf/experiment_note.hpp"
#include "dpf/random.hpp"
#include "dpf/dcf.hpp"
#include "dpf/constrained_cmp.hpp"
#include "dpf/beaver.hpp"
#include "dpf/xor_wrapper.hpp"
#include "dpf/leaf_node.hpp"
#include "dpf/leaf_arithmetic.hpp"
namespace dpf
{
@ -99,6 +110,40 @@ struct ds_randomness
PadRng pad;
};
/// @brief Pad stream whose `block()` and `bit()` come from one PRG seed.
/// @details Drop-in for `detail::urandom_pad_rng` on Doerner–Shelat dealers.
/// `pseudorandom_root_sampler<PRG>` is the matching root source.
template <typename PRG = dpf::prg::aes128>
struct prg_pad_rng
{
using block_type = typename PRG::block_type;
explicit prg_pad_rng(block_type seed = dpf::uniform_sample<block_type>())
: seed_(seed)
{
note_experiment_seed("prg_pad_rng", seed_);
}
block_type block()
{
return PRG::eval(seed_, n_++);
}
uint8_t bit()
{
const block_type drawn = block();
unsigned char low = 0;
std::memcpy(&low, &drawn, 1);
return static_cast<uint8_t>(low & 1u);
}
const block_type & seed() const noexcept { return seed_; }
private:
block_type seed_{};
std::uint32_t n_ = 0;
};
namespace detail
{
@ -164,31 +209,128 @@ simde__m128i ds_gate(uint8_t bit, simde__m128i block) noexcept
template <typename PadRng>
ds_cw_pads ds_sample_cw(PadRng & pad)
{
// Ideal (semi-honest): each party holds (rand, bit, gamma) with
// gamma0 ⊕ gamma1 = (bit1 · rand0) ⊕ (bit0 · rand1).
// The full product (bit1 · rand0) is NOT given to party 0: that would
// leak bit1, and with the opened blind bit = path1 ⊕ bit1 it would
// open the peer path bit (and thus α under an oblivious walk). Shares
// of the XOR of both products hide both pad bits from each party.
// The product share is a beaver session over the XOR ring; the clear
// bit and block stay with the party that owns them.
using Ring = dpf::xor_wrapper<simde_uint128>;
using traits = dpf::beavers::ring_traits<Ring>;
auto to_ring = [](simde__m128i block) {
simde_uint128 raw{};
std::memcpy(&raw, &block, sizeof(block));
return Ring{raw};
};
auto to_block = [](const Ring & ring) {
simde__m128i block{};
const auto raw = static_cast<typename Ring::value_type>(ring);
std::memcpy(&block, &raw, sizeof(block));
return block;
};
const Ring rand0 = to_ring(pad.block());
const Ring rand1 = to_ring(pad.block());
const bool bit0 = (pad.bit() & 1u) != 0;
const bool bit1 = (pad.bit() & 1u) != 0;
auto sampler = [&pad, &to_ring]() -> Ring {
return to_ring(pad.block());
};
dpf::beavers::session<Ring> s;
auto b0 = s.bit();
auto b1 = s.bit();
auto r0 = s.input();
auto r1 = s.input();
auto prod = s(b1 * r0 + b0 * r1);
s.pin(prod);
s.sample(sampler);
s.bind(b0, bit0 ? traits::one() : traits::zero(), sampler);
s.bind(b1, bit1 ? traits::one() : traits::zero(), sampler);
s.bind(r0, rand0, sampler);
s.bind(r1, rand1, sampler);
s.evaluate();
const auto gamma = s.value(prod);
ds_cw_pads p{};
const simde__m128i zero = simde_mm_setzero_si128();
p.p0.rand = pad.block();
p.p1.rand = pad.block();
p.p0.bit = static_cast<uint8_t>(pad.bit() & 1u);
p.p1.bit = static_cast<uint8_t>(pad.bit() & 1u);
p.p0.gamma = p.p1.bit ? p.p0.rand : zero;
p.p1.gamma = p.p0.bit ? p.p1.rand : zero;
p.p0.rand = to_block(rand0);
p.p1.rand = to_block(rand1);
p.p0.bit = static_cast<uint8_t>(bit0);
p.p1.bit = static_cast<uint8_t>(bit1);
p.p0.gamma = to_block(gamma.p0);
p.p1.gamma = to_block(gamma.p1);
return p;
}
/// @brief Pack a XOR-ring `bit_mul` into the classical DS AND pad shape.
/// @tparam Ring XOR ring whose unit is the all-ones word
/// @param bm the sampled bit-mul material
/// @param a0_share party 0's XOR share of the opened bit
/// @return pads ready for `ds_and_open`
template <typename Ring>
ds_and_pads ds_and_from_bit_mul(const dpf::beavers::bit_mul_beaver<Ring> & bm,
uint8_t a0_share)
{
using traits = dpf::beavers::ring_traits<Ring>;
const Ring opened = bm.bit.open();
const uint8_t a = (opened == traits::one()) ? uint8_t{1} : uint8_t{0};
ds_and_pads p{};
p.a0 = static_cast<uint8_t>(a0_share & 1u);
p.a1 = static_cast<uint8_t>(a ^ p.a0);
auto to_block = [](const Ring & r) {
simde__m128i b{};
const auto v = static_cast<typename Ring::value_type>(r);
static_assert(sizeof(v) == sizeof(simde__m128i),
"ds AND packs a 128-bit XOR ring into an AES block");
std::memcpy(&b, &v, sizeof(b));
return b;
};
p.b0_share = to_block(bm.scalar.p0);
p.b1_share = to_block(bm.scalar.p1);
p.c0_share = to_block(bm.product.p0);
p.c1_share = to_block(bm.product.p1);
return p;
}
/// @brief Sample one DS AND from `sample_bit_mul`, driven by `pad` or `rng`.
/// @tparam PadRng pad stream with `block()` / `bit()`
/// @tparam Sample callable returning a 128-bit XOR ring element
/// @param pad the Doerner–Shelat pad stream (bit for the clear Beaver bit)
/// @param rng ring sampler for `sample_bit_mul` (defaults to `pad.block()`)
/// @return classical AND pads for `ds_and_open`
template <typename PadRng, typename Sample>
ds_and_pads ds_sample_and(PadRng & pad, Sample && rng)
{
using Ring = dpf::xor_wrapper<simde_uint128>;
using traits = dpf::beavers::ring_traits<Ring>;
auto & ring_rng = rng;
std::size_t phase = 0;
auto sampler = [&]() -> Ring {
if (phase == 0)
{
++phase;
return (pad.bit() & 1u) ? traits::one() : traits::zero();
}
++phase;
return ring_rng();
};
auto bm = dpf::beavers::sample_bit_mul<Ring>(sampler);
const uint8_t a0 = static_cast<uint8_t>(pad.bit() & 1u);
return ds_and_from_bit_mul(bm, a0);
}
template <typename PadRng>
ds_and_pads ds_sample_and(PadRng & pad)
{
ds_and_pads p{};
const uint8_t a = static_cast<uint8_t>(pad.bit() & 1u);
const simde__m128i B = pad.block();
const simde__m128i C = ds_gate(a, B);
p.a0 = static_cast<uint8_t>(pad.bit() & 1u);
p.a1 = static_cast<uint8_t>(a ^ p.a0);
p.b0_share = pad.block();
p.b1_share = ds_xor(B, p.b0_share);
p.c0_share = pad.block();
p.c1_share = ds_xor(C, p.c0_share);
return p;
using Ring = dpf::xor_wrapper<simde_uint128>;
auto from_pad = [&pad]() -> Ring {
const simde__m128i b = pad.block();
simde_uint128 v{};
std::memcpy(&v, &b, sizeof(b));
return Ring{v};
};
return ds_sample_and(pad, from_pad);
}
HEDLEY_NO_THROW
@ -279,6 +421,123 @@ inline simde__m128i ds_deliver(uint8_t b_exp, simde__m128i base, simde__m128i M,
return ds_xor(ds_xor(local, z.z0), z.z1);
}
/// @brief XOR shares of a bit-Beaver triple `(α, β, α∧β)`.
struct ds_bit_triple
{
uint8_t a0;
uint8_t a1;
uint8_t b0;
uint8_t b1;
uint8_t c0;
uint8_t c1;
};
/// @brief Sample one bit-AND triple from `pad`, via `sample_beaver2` on the XOR ring.
/// @tparam PadRng pad stream with `bit()`
/// @param pad the pad stream
/// @return shares of `(α, β, α∧β)`
template <typename PadRng>
ds_bit_triple ds_sample_bit_and(PadRng & pad)
{
using Bit = dpf::xor_wrapper<std::uint8_t>;
auto sampler = [&pad]() -> Bit {
std::uint8_t packed = 0;
for (int i = 0; i < 8; ++i)
{
packed = static_cast<std::uint8_t>(
packed | (static_cast<std::uint8_t>(pad.bit() & 1u) << i));
}
return Bit{packed};
};
const auto triple = dpf::beavers::sample_beaver2<Bit>(sampler);
auto low = [](Bit x) {
return static_cast<uint8_t>(static_cast<std::uint8_t>(x) & 1u);
};
return ds_bit_triple{
low(triple.a.p0), low(triple.a.p1),
low(triple.b.p0), low(triple.b.p1),
low(triple.ab.p0), low(triple.ab.p1)};
}
/// @brief One party's share of `x ∧ y` after `d = x⊕α` and `e = y⊕β` are open.
/// @param d the opened mask of `x`
/// @param e the opened mask of `y`
/// @param a this party's share of `α`
/// @param b this party's share of `β`
/// @param c this party's share of `α∧β`
/// @param hold_de party 0 adds the public `d∧e` term
/// @return this party's XOR share of the product
HEDLEY_NO_THROW
inline uint8_t ds_bit_and_party(uint8_t d, uint8_t e, uint8_t a, uint8_t b,
uint8_t c, bool hold_de) noexcept
{
uint8_t z = static_cast<uint8_t>((d & b) ^ (e & a) ^ c);
if (hold_de)
z = static_cast<uint8_t>(z ^ (d & e));
return static_cast<uint8_t>(z & 1u);
}
/// @brief Joint evaluation of one bit-AND. `d` and `e` are the opened masks.
/// @param t the triple
/// @param x0 party 0's share of `x`
/// @param x1 party 1's share of `x`
/// @param y0 party 0's share of `y`
/// @param y1 party 1's share of `y`
/// @return XOR shares of `x ∧ y`
HEDLEY_NO_THROW
inline std::pair<uint8_t, uint8_t> ds_bit_and_shares(const ds_bit_triple & t,
uint8_t x0, uint8_t x1, uint8_t y0, uint8_t y1) noexcept
{
const uint8_t d = static_cast<uint8_t>(x0 ^ x1 ^ t.a0 ^ t.a1);
const uint8_t e = static_cast<uint8_t>(y0 ^ y1 ^ t.b0 ^ t.b1);
return {ds_bit_and_party(d, e, t.a0, t.b0, t.c0, true),
ds_bit_and_party(d, e, t.a1, t.b1, t.c1, false)};
}
/// @brief Replace additive shares with XOR shares of their sum.
/// @details One beaver bit-AND per bit except the last. Party 0's sum-bit
/// share is `a ⊕ c0`; party 1's is `b ⊕ c1`. The carry share is
/// `((a⊕c) ∧ (b⊕c)) ⊕ c`, which is the majority. Neither share is
/// the sum, and the sum is not written down.
/// @tparam PadRng pad stream with `bit()`
/// @tparam InputT input domain type
/// @param pads the pad stream
/// @param x0 party 0's additive share, replaced by its XOR share of the sum
/// @param x1 party 1's additive share, replaced by its XOR share of the sum
template <typename PadRng, typename InputT>
void split_additive_to_xor(PadRng & pads, InputT & x0, InputT & x1)
{
constexpr auto to_int = utils::to_integral_type<InputT>{};
using FromI = typename utils::make_from_integral_value<InputT>::integral_type;
using U = std::make_unsigned_t<FromI>;
const U u0 = static_cast<U>(to_int(x0));
const U u1 = static_cast<U>(to_int(x1));
U s0 = 0;
U s1 = 0;
uint8_t c0 = 0;
uint8_t c1 = 0;
constexpr std::size_t nbits = utils::bitlength_of_v<InputT>;
for (std::size_t i = 0; i < nbits; ++i)
{
const uint8_t a = static_cast<uint8_t>((u0 >> i) & U{1});
const uint8_t b = static_cast<uint8_t>((u1 >> i) & U{1});
const uint8_t sum0 = static_cast<uint8_t>(a ^ c0);
const uint8_t sum1 = static_cast<uint8_t>(b ^ c1);
s0 = static_cast<U>(s0 | (static_cast<U>(sum0) << i));
s1 = static_cast<U>(s1 | (static_cast<U>(sum1) << i));
if (i + 1 == nbits)
break;
const ds_bit_triple triple = ds_sample_bit_and(pads);
const auto prod = ds_bit_and_shares(triple,
static_cast<uint8_t>(a ^ c0), c1,
c0, static_cast<uint8_t>(b ^ c1));
c0 = static_cast<uint8_t>(prod.first ^ c0);
c1 = static_cast<uint8_t>(prod.second ^ c1);
}
x0 = utils::make_from_integral_value<InputT>{}(static_cast<FromI>(s0));
x1 = utils::make_from_integral_value<InputT>{}(static_cast<FromI>(s1));
}
/// @brief Per-level messages prepared before the CW protocol runs (blinds + pads).
struct ds_level_blinds
{
@ -326,6 +585,35 @@ struct ds_cmp_gen_state
const void * paint_ctx = nullptr;
};
/// @brief Local joint-sim mux of a packed naked leaf from XOR bit shares.
/// @details Mirrors `party/oblivious_select.hpp` `mux_leaf_share`: each level
/// selects with `bit0 ^ bit1` so the call site never forms a clear
/// point for `make_leaves`. The joint simulator holds both shares.
/// @tparam Leaf packed leaf type
/// @tparam Make candidate builder `Leaf(unsigned lane)`
/// @tparam Bit0 party-0 bit accessor
/// @tparam Bit1 party-1 bit accessor
template <typename Leaf, typename Make, typename Bit0, typename Bit1>
Leaf mux_naked_leaf_local(std::size_t lg, Make && make, Bit0 && bit0_at,
Bit1 && bit1_at)
{
if (lg == 0)
return make(0u);
std::vector<Leaf> cand(std::size_t{1} << lg);
for (std::size_t i = 0; i < cand.size(); ++i)
cand[i] = make(static_cast<unsigned>(i));
for (std::size_t b = 0; b < lg; ++b)
{
const uint8_t bit = static_cast<uint8_t>(
(bit0_at(b) ^ bit1_at(b)) & 1u);
std::vector<Leaf> next(cand.size() / 2);
for (std::size_t k = 0; k < next.size(); ++k)
next[k] = bit ? cand[2 * k + 1] : cand[2 * k];
cand.swap(next);
}
return cand[0];
}
/// @brief Local joint simulation: today's `ds_cw_outs` / `ds_open_advice` / `ds_and_open`.
/// @details An MPC backend would send `blinds` and return the same `ds_level_open` shape.
/// @tparam PadRng pad stream for the Doerner–Shelat protocol
@ -451,83 +739,23 @@ struct local_cw_protocol
std::forward<BlockSampler>(sample));
}
/// @brief Majority of three bits (next carry of a full adder).
/// @param a the `a`
/// @param b the `b`
/// @param c the `c`
/// @return Majority of three bits (next carry of a full adder)
HEDLEY_NO_THROW
static constexpr uint8_t majority(uint8_t a, uint8_t b, uint8_t c) noexcept
{
return static_cast<uint8_t>((a & b) | (a & c) | (b & c));
}
/// @brief One additive digit: sum bit `a XOR b XOR cin`, carry out = majority.
/// @param a the `a`
/// @param b the `b`
/// @param cin the `cin`
/// @param cout the `cout`
/// @return One additive digit: sum bit `a XOR b XOR cin`, carry out = majority
HEDLEY_NO_THROW
static constexpr uint8_t open_sum_bit(uint8_t a, uint8_t b, uint8_t cin,
uint8_t & cout) noexcept
{
cout = majority(a, b, cin);
return static_cast<uint8_t>(a ^ b ^ cin);
}
/// @brief Reconstruct `a0 + a1` via a LSB→MSB carry chain, then flip the MSB
/// when the domain is signed — matching `make_dpf` on the sum. The call
/// site never forms the sum; an MPC backend would open the same bits.
/// @brief Encode shares for the XOR-style CW walk.
/// @details XOR inputs flip party 0's MSB, which is linear over XOR.
/// Additive inputs are converted first: `split_additive_to_xor`
/// draws one bit-Beaver per carry and leaves XOR shares of the
/// sum. The MSB flip is then the same one XOR inputs take, so
/// the walk matches `make_dpf` on the sum. The sum is not opened
/// and party 1's share is not cleared.
/// @tparam InputT input domain type
/// @param a0 the `a0`
/// @param a1 the `a1`
/// @return Reconstruct `a0 + a1` via a LSB→MSB carry chain, then flip the MSB when the domain
/// is signed — matching `make_dpf` on the sum
/// @param x0 party 0's share
/// @param x1 party 1's share
/// @param arith `true` when `x0`, `x1` are additive
template <typename InputT>
InputT open_arith_point(InputT a0, InputT a1) const
{
constexpr auto to_int = utils::to_integral_type<InputT>{};
using FromI = typename utils::make_from_integral_value<InputT>::integral_type;
using U = std::make_unsigned_t<FromI>;
const U u0 = static_cast<U>(to_int(a0));
const U u1 = static_cast<U>(to_int(a1));
U sum = 0;
uint8_t carry = 0;
constexpr std::size_t nbits = utils::bitlength_of_v<InputT>;
for (std::size_t i = 0; i < nbits; ++i)
{
const uint8_t b0 = static_cast<uint8_t>((u0 >> i) & U{1});
const uint8_t b1 = static_cast<uint8_t>((u1 >> i) & U{1});
const uint8_t s = open_sum_bit(b0, b1, carry, carry);
sum = static_cast<U>(sum | (static_cast<U>(s) << i));
}
InputT out = utils::make_from_integral_value<InputT>{}(
static_cast<FromI>(sum));
utils::flip_msb_if_signed_integral(out);
return out;
}
/// @brief Encode shares for the XOR-style CW walk. XOR mode flips party 0's MSB
/// (linear over XOR). Arithmetic mode opens the sum (carry + signed MSB)
/// and returns `(alpha, 0)` so the walk matches `make_dpf(alpha)`.
/// @tparam InputT input domain type
/// @param x0 the `x0`
/// @param x1 the `x1`
/// @param arith the `arith`
template <typename InputT>
void encode_walk_shares(InputT & x0, InputT & x1, bool arith) const
void encode_walk_shares(InputT & x0, InputT & x1, bool arith)
{
if (arith)
{
const InputT alpha = open_arith_point(x0, x1);
x0 = alpha;
x1 = InputT{};
}
else
{
utils::flip_msb_if_signed_integral(x0);
}
split_additive_to_xor(pads, x0, x1);
utils::flip_msb_if_signed_integral(x0);
}
/// @brief Constrained comparison Π_CCMP: open `1{x0 < x1}` when `|x0−x1|=1`.
@ -542,14 +770,18 @@ struct local_cw_protocol
}
/// @brief Open a public leaf CW for a shared payload.
/// @details Ring: `β = y0 + y1`; `g = CCMP(t0,t1)` selects `β − M` vs `M − β`
/// (matches `make_leaf` with `sign = t0`). Characteristic 2: `β = y0 ⊕ y1`
/// and CW = `β ⊕ M` (sign mux is a no-op under XOR).
/// @details Splits the payload into leaf words (`naked(y0) ± naked(y1)`) so
/// scalar `β = y0 + y1` (or `y0 ⊕ y1`) is never formed. The packing
/// lane is muxed from XOR bit shares of the point, matching
/// `mux_leaf_share` on the socket. Ring: `g = CCMP(t0,t1)` selects
/// `N − M` vs `M − N` (matches `make_leaf` with `sign = t0`).
/// Characteristic 2: CW = `N ⊕ M` (sign mux is a no-op under XOR).
/// @tparam ExteriorPRG PRG that expands the root. Defaults to `InteriorPRG`
/// @tparam I output index
/// @tparam OutputsTuple outputs tuple
/// @tparam InteriorBlock interior block
/// @tparam OutputT output type
/// @tparam InputT input domain type
/// @param seed0 the `seed0`
/// @param seed1 the `seed1`
/// @param t0 the `t0`
@ -557,13 +789,14 @@ struct local_cw_protocol
/// @param y0 the `y0`
/// @param y1 the `y1`
/// @param pos_base the `pos_base`
/// @param lane_x lane of the shared payload
/// @return the opened leaf correction word
/// @param x0 party 0's XOR share of the (lane) point
/// @param x1 party 1's XOR share of the (lane) point
/// @return the opened leaf correction word
template <typename ExteriorPRG, std::size_t I = 0, typename OutputsTuple,
typename InteriorBlock, typename OutputT>
typename InteriorBlock, typename OutputT, typename InputT>
auto open_arith_leaf(const InteriorBlock & seed0, const InteriorBlock & seed1,
uint8_t t0, uint8_t t1, OutputT y0, OutputT y1, std::size_t pos_base,
std::size_t lane_x) -> dpf::leaf_node_t<typename ExteriorPRG::block_type,
InputT x0, InputT x1) -> dpf::leaf_node_t<typename ExteriorPRG::block_type,
OutputT>
{
using output_type = OutputT;
@ -573,36 +806,99 @@ struct local_cw_protocol
using leaf_type = dpf::leaf_node_t<node_type, output_type>;
HEDLEY_PRAGMA(GCC diagnostic pop)
constexpr std::size_t lg =
dpf::lg_outputs_per_leaf_v<output_type, node_type>;
constexpr auto to_int = utils::to_integral_type<InputT>{};
auto bit0_at = [&](std::size_t b) {
return static_cast<uint8_t>((to_int(x0) >> b) & 1u);
};
auto bit1_at = [&](std::size_t b) {
return static_cast<uint8_t>((to_int(x1) >> b) & 1u);
};
auto naked_of = [&](output_type y) {
return mux_naked_leaf_local<leaf_type>(lg,
[&](unsigned i) {
return dpf::make_naked_leaf<node_type>(
static_cast<InputT>(i), y);
},
bit0_at, bit1_at);
};
// Split payload across leaf words; never form scalar β.
const leaf_type naked =
dpf::add_leaf<output_type>(naked_of(y0), naked_of(y1));
const auto M = dpf::make_leaf_mask<ExteriorPRG, I, OutputsTuple, InteriorBlock>(
seed0, seed1, pos_base);
output_type beta{};
if constexpr (utils::has_characteristic_two_v<output_type>)
{
(void)t0;
(void)t1;
beta = static_cast<output_type>(y0 ^ y1);
const leaf_type naked = dpf::make_naked_leaf<node_type>(lane_x, beta);
return dpf::subtract_leaf<output_type>(naked, M);
}
else
{
const uint8_t g = open_ccmp(t0, t1);
beta = static_cast<output_type>(y0 + y1);
const leaf_type naked = dpf::make_naked_leaf<node_type>(lane_x, beta);
// CW = (−1)^{t1}(β − M): g=0 → β−M; g=1 → M−β. Matches make_leaf(sign=t0).
// CW = (−1)^{t1}(N − M): g=0 → N−M; g=1 → M−N. Matches make_leaf(sign=t0).
if (g & 1u)
return dpf::subtract_leaf<output_type>(M, naked);
return dpf::subtract_leaf<output_type>(naked, M);
}
}
/// @brief Open a group of leaf correction words for one prefix group. In this
/// local joint simulation both XOR shares of the point are present, so the
/// point is reconstructed *inside* the protocol and handed to `leaf_fn`
/// (which runs `make_leaves` for the group). The Doerner–Shelat gen never
/// forms `x = x0 ^ x1` at its own call site; an MPC backend would instead
/// run a per-group leaf CW exchange that never reveals `x`. After
/// `encode_walk_shares`, arithmetic inputs are already `(alpha, 0)`.
/// @brief Open the comparison threshold lane from XOR shares of the point.
/// @details Reconstructs only inside this protocol hook for paint units,
/// domain-edge triviality, and blocked suffixes. Per-level value
/// words use share bits (`bit0 ^ bit1`) instead of this value.
/// @tparam InputT input domain type
/// @param x0 party 0's XOR share
/// @param x1 party 1's XOR share
/// @param nbits width of the comparison lane
/// @return the comparison threshold as an integer lane
template <typename InputT>
unsigned __int128 open_cmp_threshold(InputT x0, InputT x1,
std::size_t nbits) noexcept
{
constexpr auto to_int = utils::to_integral_type<InputT>{};
constexpr std::size_t bl = utils::bitlength_of_v<InputT>;
const InputT x = utils::xor_input_shares(x0, x1);
if (nbits >= bl)
return static_cast<unsigned __int128>(to_int(x));
return static_cast<unsigned __int128>(to_int(x) >> (bl - nbits));
}
/// @brief Public correction seed from XOR shares of the path prefix.
/// @details Reconstructs the prefix only inside this protocol hook and
/// returns `make_cs` (same digest as `oblivious_cs` when both
/// seeds are in-process). The clear prefix is not returned.
/// @tparam InputT input domain type
/// @param level fold level (or blocked tag | level)
/// @param x0 party 0's XOR share of the encoded point
/// @param x1 party 1's XOR share of the encoded point
/// @param bits number of high path bits in the prefix
/// @param s0 party 0's on-path seed
/// @param s1 party 1's on-path seed
/// @return the public correction seed
template <typename InputT>
cs_block open_correction_seed(std::size_t level, InputT x0, InputT x1,
std::size_t bits, simde__m128i s0, simde__m128i s1) noexcept
{
constexpr auto to_int = utils::to_integral_type<InputT>{};
constexpr std::size_t bl = utils::bitlength_of_v<InputT>;
const InputT x = utils::xor_input_shares(x0, x1);
// Prefer `uint64_t` over `psnip_uint64_t{...}`: that macro expands to
// `long unsigned int`, which is not a valid braced/cast type-id alone.
const uint64_t prefix = (bits == 0 || bits > bl)
? uint64_t{0}
: static_cast<uint64_t>(to_int(x) >> (bl - bits));
return detail::vdpf::make_cs(level, prefix, s0, s1);
}
/// @brief Open a group of leaf correction words for one prefix group.
/// @details Hands both XOR shares to `leaf_fn`. The builder may reconstruct
/// the point only to emit public CWs (never return α to the DS
/// call site). An MPC backend would run a per-group leaf CW
/// exchange that never reveals `x`. Additive inputs have already
/// been replaced by XOR shares of the sum.
/// @tparam InputT input domain type
/// @tparam LeafFn leaf fn
/// @param x0 the `x0`
@ -611,7 +907,7 @@ struct local_cw_protocol
template <typename InputT, typename LeafFn>
void open_leaf_group(InputT x0, InputT x1, LeafFn && leaf_fn)
{
std::forward<LeafFn>(leaf_fn)(utils::xor_input_shares(x0, x1));
std::forward<LeafFn>(leaf_fn)(x0, x1);
}
};
@ -705,8 +1001,8 @@ void ds_advance_level(ds_gen_state<NodeT> & st, InputT x0, InputT x1,
vblinds.R0 = v0[1];
vblinds.L1 = v1[0];
vblinds.R1 = v1[1];
const int ai = static_cast<int>(
(cmp->thresh >> (cmp->nbits - 1 - level)) & 1);
// Path bit from share bits (threshold bit equals the walk bit).
const int ai = static_cast<int>((bit0 ^ bit1) & 1u);
if (cmp->paint)
{
const uint64_t unit = dcf_impl::paint_unit(cmp->kind, level,
@ -890,30 +1186,31 @@ HEDLEY_PRAGMA(GCC diagnostic pop)
const uint8_t t0 = static_cast<uint8_t>(sign0);
const uint8_t t1 = static_cast<uint8_t>(dpf::get_lo_bit(parent1));
input_type x = utils::xor_input_shares(x0, x1);
leaf_tuple leaves0{};
leaf_tuple leaves1{};
beaver_tuple beavers0{};
beaver_tuple beavers1{};
if (arith_out)
{
constexpr auto to_int = utils::to_integral_type<input_type>{};
const std::size_t lane = static_cast<std::size_t>(to_int(x));
auto cw = proto.template open_arith_leaf<ExteriorPRG, 0, outputs_tuple>(
dpf::unset_lo_2bits(parent0), dpf::unset_lo_2bits(parent1), t0, t1,
y0, y1, std::size_t{0}, lane);
y0, y1, std::size_t{0}, x0, x1);
std::get<0>(leaves0) = cw;
std::get<0>(leaves1) = cw;
}
else
{
auto built = dpf::make_leaves<ExteriorPRG>(x,
dpf::unset_lo_2bits(parent0), dpf::unset_lo_2bits(parent1), sign0,
std::size_t{0}, y0);
leaves0 = std::move(built.first.first);
beavers0 = std::move(built.first.second);
leaves1 = std::move(built.second.first);
beavers1 = std::move(built.second.second);
// Reconstruct the point only inside the leaf protocol hook.
proto.open_leaf_group(x0, x1, [&](input_type sx0, input_type sx1) {
const input_type x = utils::xor_input_shares(sx0, sx1);
auto built = dpf::make_leaves<ExteriorPRG>(x,
dpf::unset_lo_2bits(parent0), dpf::unset_lo_2bits(parent1),
sign0, std::size_t{0}, y0);
leaves0 = std::move(built.first.first);
beavers0 = std::move(built.first.second);
leaves1 = std::move(built.second.first);
beavers1 = std::move(built.second.second);
});
(void)y1;
}
@ -1000,18 +1297,29 @@ HEDLEY_PRAGMA(GCC diagnostic pop)
const node parent1 = st.seed1();
const bool sign0 = dpf::get_lo_bit(parent0);
input_type x = utils::xor_input_shares(x0, x1);
auto built = dpf::make_leaves<ExteriorPRG>(x,
dpf::unset_lo_2bits(parent0), dpf::unset_lo_2bits(parent1), sign0,
std::size_t{0}, std::forward<OutputT>(y), std::forward<OutputTs>(ys)...);
typename dpf_type::leaf_tuple leaves0{};
typename dpf_type::beaver_tuple beavers0{};
typename dpf_type::leaf_tuple leaves1{};
typename dpf_type::beaver_tuple beavers1{};
proto.open_leaf_group(x0, x1, [&](input_type sx0, input_type sx1) {
const input_type x = utils::xor_input_shares(sx0, sx1);
auto built = dpf::make_leaves<ExteriorPRG>(x,
dpf::unset_lo_2bits(parent0), dpf::unset_lo_2bits(parent1), sign0,
std::size_t{0}, std::forward<OutputT>(y),
std::forward<OutputTs>(ys)...);
leaves0 = std::move(built.first.first);
beavers0 = std::move(built.first.second);
leaves1 = std::move(built.second.first);
beavers1 = std::move(built.second.second);
});
input_type off0{};
input_type off1{};
return dpf::make_party_key_pair(
dpf_type{root0, correction_words, correction_advice,
built.first.first, built.first.second, off0},
leaves0, beavers0, off0},
dpf_type{root1, correction_words, correction_advice,
built.second.first, built.second.second, off1});
leaves1, beavers1, off1});
}
/// @brief Single-output plaintext β (disambiguates from arith_out overload).