Checkpoint the party/runtime stack before share-program and malicious-mode work.

Ship the TLS mesh, composer, Beaver/Yao/leaf MPC, prep/online paths, apps, and docs so the tree is pushable before elevating share_expr, security_mode, and prep resume.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Ryan Henry 2026-09-28 05:59:19 -06:00
parent 695f8e84f7
commit 0d22946a0e
1835 changed files with 170291 additions and 2849 deletions

View file

@ -58,6 +58,16 @@ enum class reduced : unsigned
expm1,
log1p,
};
/// \complexity The `switch` does a constant amount of range reduction and a constant number of `eval_principal` cubics (`Θ(log P)` each).
/// On the small interval, `expm1_series` loops `n = 1 .. 24` and `log1p_series` loops `n = 1 .. 80`, and both stop when the running power is 0.
/// Extra space `Θ(1)`.
/// @see grotto::eval_principal
/// @see grotto::eval_window
/// @see grotto::eval_closed
/// @param which the reduced map
/// @param fractional_bits one of 8, 12, ..., 32
/// @param raw fixed-point argument, value `raw / 2^{fractional_bits}`
/// @return fixed-point result at the same scale
HEDLEY_WARN_UNUSED_RESULT
inline std::int64_t eval_reduced(reduced which, unsigned fractional_bits, std::int64_t raw);
@ -222,12 +232,10 @@ inline std::int64_t ln2_raw(unsigned fractional_bits)
return scale_unit(ln2_64, fractional_bits);
}
inline std::int64_t eval_ln_positive(unsigned fractional_bits, std::int64_t raw)
{
const dyadic part = split_positive(raw, fractional_bits);
const std::int64_t ln_m = eval_principal(principal::ln, fractional_bits, part.mantissa_raw);
return ln_m + static_cast<std::int64_t>(part.power) * ln2_raw(fractional_bits);
}
inline std::int64_t eval_ln_positive(unsigned fractional_bits, std::int64_t raw);
inline std::int64_t eval_log10_positive(unsigned fractional_bits, std::int64_t raw);
inline u128 exp_scale64(unsigned fractional_bits, std::int64_t raw, std::int64_t & n_bin);
inline std::int64_t finish_wide(u128 wide, int right_shift);
inline std::int64_t eval_exp_at_scale(unsigned fractional_bits, std::int64_t raw)
{
@ -240,39 +248,10 @@ inline std::int64_t eval_exp_at_scale(unsigned fractional_bits, std::int64_t raw
const std::int64_t lifted = eval_exp_at_scale(16, static_cast<std::int64_t>(lifted_arg));
return round_i128(lifted, static_cast<unsigned>(lift));
}
const std::int64_t ln2 = ln2_raw(fractional_bits);
if (ln2 <= 0)
throw std::logic_error("range lut: ln 2 constant");
std::int64_t n_bin = raw / ln2;
std::int64_t remainder = raw - n_bin * ln2;
if (remainder < 0)
{
remainder += ln2;
--n_bin;
}
while (remainder >= ln2)
{
remainder -= ln2;
++n_bin;
}
const std::int64_t step = std::int64_t{1} << (fractional_bits - 13);
const std::int64_t chunks = remainder / step;
const std::int64_t tiny = remainder - chunks * step;
std::int64_t table_raw = tiny << 13;
const std::int64_t one = one_raw(fractional_bits);
if (table_raw > one)
table_raw = one;
std::int64_t exp_s = eval_principal(principal::exp, fractional_bits, table_raw);
for (unsigned bit = 0; bit < 13; ++bit)
{
if ((static_cast<unsigned long long>(chunks) & (1ull << bit)) == 0)
continue;
// `chunk` is `exp(2^{i-13}) * 2^64`, so the product's high limb is the raw product.
const u128 prod = static_cast<u128>(exp_s) * exp_chunk_64[bit];
exp_s = round_mag(prod, 64, false);
}
return shift_pow2(exp_s, static_cast<int>(n_bin));
std::int64_t n_bin = 0;
const u128 wide = exp_scale64(fractional_bits, raw, n_bin);
const int shift = static_cast<int>(64u - fractional_bits) - static_cast<int>(n_bin);
return finish_wide(wide, shift);
}
HEDLEY_NO_THROW
@ -318,7 +297,7 @@ struct angle
/// only as far as the low bits the quadrant logic reads.
/// @param fractional_bits the number of fractional bits
/// @param raw the underlying integer
/// @param multiplier_64 the `multiplier_64`
/// @param multiplier_64 multiplier already scaled by `2^64`
/// @return `{ |x| * multiplier }` at this precision, with the integer part reduced only as far as
/// the low bits the quadrant logic reads
inline angle reduce_positive(unsigned fractional_bits, std::int64_t raw, u128 multiplier_64)
@ -498,7 +477,7 @@ inline hyp sinh_cosh(unsigned fractional_bits, std::int64_t raw)
/// @brief `ln(2^{k+1} ± 1) / 2`, the saturation threshold used by `tanh` and `coth`.
/// @param fractional_bits the number of fractional bits
/// @param plus the `plus`
/// @param plus true for the plus saturation threshold, false for the minus threshold
/// @return `ln(2^{k+1} ± 1) / 2`, the saturation threshold used by `tanh` and `coth`
inline std::int64_t beta_raw(unsigned fractional_bits, bool plus)
{
@ -518,6 +497,249 @@ constexpr int half_pow_of(int power) noexcept
return (power & 1) != 0 ? (power - 1) / 2 : power / 2;
}
struct u256
{
u128 lo;
u128 hi;
};
inline u256 mul_u128(u128 a, u128 b)
{
const auto a0 = static_cast<std::uint64_t>(a);
const auto a1 = static_cast<std::uint64_t>(a >> 64);
const auto b0 = static_cast<std::uint64_t>(b);
const auto b1 = static_cast<std::uint64_t>(b >> 64);
const u128 p00 = u128{a0} * b0;
const u128 p01 = u128{a0} * b1;
const u128 p10 = u128{a1} * b0;
const u128 p11 = u128{a1} * b1;
const u128 col = (p00 >> 64) + static_cast<std::uint64_t>(p01) + static_cast<std::uint64_t>(p10);
u256 out;
out.lo = static_cast<std::uint64_t>(p00) | (col << 64);
out.hi = p11 + (p01 >> 64) + (p10 >> 64) + (col >> 64);
return out;
}
inline u128 round_u256(u256 value, unsigned shift)
{
if (shift == 0)
return value.lo;
if (shift >= 256)
return 0;
u256 bumped = value;
const unsigned bit = shift - 1;
if (bit < 128)
{
const u128 before = bumped.lo;
bumped.lo += u128{1} << bit;
if (bumped.lo < before)
++bumped.hi;
}
else
bumped.hi += u128{1} << (bit - 128);
if (shift < 128)
{
if (shift == 0)
return bumped.lo;
return (bumped.lo >> shift) | (bumped.hi << (128 - shift));
}
return bumped.hi >> (shift - 128);
}
/// @brief `ln(m) * 2^64` for `m` in `[1/2, 1]`, via `2 artanh((m-1)/(m+1))`.
/// @details `|z| <= 1/3`, so forty odd powers sit well below `2^{-64}`.
inline u128 ln_mantissa_scale64(unsigned fractional_bits, std::int64_t mantissa_raw)
{
const u128 one = u128{1} << 64;
const u128 m = static_cast<u128>(mantissa_raw) << (64u - fractional_bits);
if (m >= one)
return 0;
const u128 num = one - m;
const u128 den = one + m;
const u128 z = ((num << 64) + den / 2) / den;
const u128 z2 = round_u256(mul_u128(z, z), 64);
u128 acc = z;
u128 power = z;
for (int n = 1; n <= 40; ++n)
{
power = round_u256(mul_u128(power, z2), 64);
const unsigned denom = static_cast<unsigned>(2 * n + 1);
const u128 term = (power + denom / 2) / denom;
if (term == 0)
break;
acc += term;
}
return acc << 1;
}
inline std::int64_t round_scale64_to_k(u128 mag, bool neg, unsigned fractional_bits)
{
return round_mag(mag, 64u - fractional_bits, neg);
}
inline void ln_magnitude_scale64(unsigned fractional_bits, std::int64_t raw, u128 & mag, bool & neg)
{
const dyadic part = split_positive(raw, fractional_bits);
// `ln(m) <= 0` on `[1/2, 1]`, so `ln(m * 2^e) = e·ln 2 - |ln m|`.
const u128 ln_m = ln_mantissa_scale64(fractional_bits, part.mantissa_raw);
if (part.power >= 0)
{
const u128 lift = ln2_64 * static_cast<u128>(part.power);
if (lift >= ln_m)
{
mag = lift - ln_m;
neg = false;
}
else
{
mag = ln_m - lift;
neg = true;
}
}
else
{
mag = ln2_64 * static_cast<u128>(-part.power) + ln_m;
neg = true;
}
}
inline std::int64_t eval_ln_positive(unsigned fractional_bits, std::int64_t raw)
{
u128 mag = 0;
bool neg = false;
ln_magnitude_scale64(fractional_bits, raw, mag, neg);
return round_scale64_to_k(mag, neg, fractional_bits);
}
/// @brief `(rem << 64) / den`, rounded. `rem < den` and `den < 2^96`.
inline u128 div_rem_lshift64(u128 rem, u128 den)
{
const u128 hi = (rem << 32) / den;
const u128 mid = (rem << 32) % den;
const u128 lo = (mid << 32) / den;
const u128 leftover = (mid << 32) % den;
u128 out = (hi << 32) + lo;
if (leftover >= den / 2)
++out;
return out;
}
inline std::int64_t eval_log10_positive(unsigned fractional_bits, std::int64_t raw)
{
u128 ln_mag = 0;
bool neg = false;
ln_magnitude_scale64(fractional_bits, raw, ln_mag, neg);
const u128 quot = ln_mag / ln10_64;
const u128 rem = ln_mag % ln10_64;
const u128 log_mag = (quot << 64) + div_rem_lshift64(rem, ln10_64);
return round_scale64_to_k(log_mag, neg, fractional_bits);
}
/// @brief `exp(x) * 2^64`. The `ln 2` split and the `2^{-13}` chunks stay at
/// scale 64 and are rounded once into the caller's precision.
inline u128 exp_scale64(unsigned fractional_bits, std::int64_t raw, std::int64_t & n_bin)
{
const u128 x64 = static_cast<u128>(raw < 0 ? -static_cast<__int128>(raw) : raw)
<< (64u - fractional_bits);
const bool neg = raw < 0;
u128 mag = x64;
n_bin = 0;
if (mag >= ln2_64)
{
n_bin = static_cast<std::int64_t>(mag / ln2_64);
mag -= ln2_64 * static_cast<u128>(n_bin);
}
if (neg)
{
if (mag == 0)
n_bin = -n_bin;
else
{
n_bin = -n_bin - 1;
mag = ln2_64 - mag;
}
}
const u128 step = u128{1} << 51;
const u128 chunks = mag / step;
u128 tiny = mag % step;
u128 acc = u128{1} << 64;
u128 power = tiny;
for (int n = 1; n <= 16; ++n)
{
const u128 term = (power + static_cast<u128>(n) / 2) / static_cast<u128>(n);
if (term == 0)
break;
acc += term;
power = round_u256(mul_u128(term, tiny), 64);
}
for (unsigned bit = 0; bit < 13; ++bit)
{
if (((chunks >> bit) & 1u) == 0)
continue;
acc = round_u256(mul_u128(acc, exp_chunk_64[bit]), 64);
}
return acc;
}
/// @brief Two Newton steps at scale `2k`, then the exact power-of-two lift.
/// @details The principal cubic is half an ulp at scale `k`. Shifting that
/// rounded word left multiplies the error. Refining before the shift
/// leaves an absolute error below one output ulp across the domain.
inline u128 newton_inv(unsigned fractional_bits, std::int64_t mantissa_raw, std::int64_t seed_raw)
{
const unsigned K = fractional_bits * 2u;
u128 m = static_cast<u128>(mantissa_raw) << fractional_bits;
u128 y = static_cast<u128>(seed_raw) << fractional_bits;
const u128 two = u128{2} << K;
for (int step = 0; step < 2; ++step)
{
const u128 my = round_u256(mul_u128(m, y), K);
if (my >= two)
break;
y = round_u256(mul_u128(y, two - my), K);
}
return y;
}
inline u128 newton_rsqrt(unsigned fractional_bits, std::int64_t mantissa_raw, std::int64_t seed_raw)
{
const unsigned K = fractional_bits * 2u;
u128 m = static_cast<u128>(mantissa_raw) << fractional_bits;
u128 y = static_cast<u128>(seed_raw) << fractional_bits;
const u128 three = u128{3} << K;
for (int step = 0; step < 2; ++step)
{
const u128 yy = round_u256(mul_u128(y, y), K);
const u128 myy = round_u256(mul_u128(m, yy), K);
if (myy >= three)
break;
const u128 corr = round_u256(mul_u128(y, three - myy), K);
y = (corr + 1) >> 1;
}
return y;
}
inline std::int64_t finish_wide(u128 wide, int right_shift)
{
if (right_shift >= 256)
return 0;
if (right_shift >= 0)
{
u256 value{wide, 0};
const u128 rounded = round_u256(value, static_cast<unsigned>(right_shift));
if (rounded > static_cast<u128>(INT64_MAX))
throw std::overflow_error("range lut: reciprocal does not fit int64");
return static_cast<std::int64_t>(rounded);
}
const int left = -right_shift;
if (left >= 127)
throw std::overflow_error("range lut: reciprocal does not fit int64");
const u128 shifted = wide << static_cast<unsigned>(left);
if (shifted > static_cast<u128>(INT64_MAX))
throw std::overflow_error("range lut: reciprocal does not fit int64");
return static_cast<std::int64_t>(shifted);
}
inline __int128 div_round_i128(__int128 num, int den)
{
const bool neg = num < 0;
@ -576,31 +798,21 @@ inline std::int64_t eval_expm1(unsigned fractional_bits, std::int64_t raw)
if (raw > -ln2 && raw < ln2)
return expm1_series(fractional_bits, raw);
std::int64_t n_bin = raw / ln2;
std::int64_t remainder = raw - n_bin * ln2;
if (remainder < 0)
{
remainder += ln2;
--n_bin;
}
while (remainder >= ln2)
{
remainder -= ln2;
++n_bin;
}
const std::int64_t exp_r = eval_exp_at_scale(fractional_bits, remainder);
const std::int64_t one = one_raw(fractional_bits);
std::int64_t n_bin = 0;
const u128 wide = exp_scale64(fractional_bits, raw, n_bin);
if (n_bin >= 63)
throw std::overflow_error("range lut: exponent overflow");
u128 exp64 = wide;
if (n_bin >= 0)
{
const __int128 wide = static_cast<__int128>(shift_pow2(exp_r, static_cast<int>(n_bin))) - one;
if (wide > INT64_MAX || wide < INT64_MIN)
throw std::overflow_error("range lut: exponent overflow");
return static_cast<std::int64_t>(wide);
}
const int places = static_cast<int>(-n_bin);
if (places > static_cast<int>(fractional_bits) + 1)
return -one;
return shift_pow2(exp_r, -places) - one;
exp64 <<= static_cast<unsigned>(n_bin);
else if (-n_bin >= 128)
exp64 = 0;
else
exp64 >>= static_cast<unsigned>(-n_bin);
const u128 unit = u128{1} << 64;
const bool below = exp64 < unit;
const u128 diff = below ? unit - exp64 : exp64 - unit;
return round_mag(diff, 64u - fractional_bits, below);
}
/// @brief `log1p` on `|x| <= 1/2`. Every term of a negative argument is negative.
@ -647,6 +859,16 @@ inline std::int64_t eval_log1p(unsigned fractional_bits, std::int64_t raw)
}
} // namespace range_detail
/// \complexity The `switch` does a constant amount of range reduction and a constant number of `eval_principal` cubics (`Θ(log P)` each).
/// On the small interval, `expm1_series` loops `n = 1 .. 24` and `log1p_series` loops `n = 1 .. 80`, and both stop when the running power is 0.
/// Extra space `Θ(1)`.
/// @see grotto::eval_principal
/// @see grotto::eval_window
/// @see grotto::eval_closed
/// @param which the reduced map
/// @param fractional_bits one of 8, 12, ..., 32
/// @param raw fixed-point argument, value `raw / 2^{fractional_bits}`
/// @return fixed-point result at the same scale
HEDLEY_WARN_UNUSED_RESULT
inline std::int64_t eval_reduced(reduced which, unsigned fractional_bits, std::int64_t raw)
@ -668,15 +890,7 @@ inline std::int64_t eval_reduced(reduced which, unsigned fractional_bits, std::i
return lg_m + (static_cast<std::int64_t>(part.power) << fractional_bits);
}
case reduced::log10:
{
const dyadic part = split_positive(raw, fractional_bits);
const std::int64_t ln_m = eval_principal(principal::ln, fractional_bits, part.mantissa_raw);
const std::int64_t mantissa = mul_raw(
ln_m, scale_unit(inv_ln10_64, fractional_bits), fractional_bits);
const std::int64_t lift = static_cast<std::int64_t>(part.power)
* scale_unit(log10_2_64, fractional_bits);
return mantissa + lift;
}
return eval_log10_positive(fractional_bits, raw);
case reduced::exp:
return eval_exp_at_scale(fractional_bits, raw);
case reduced::exp2:
@ -797,24 +1011,32 @@ inline std::int64_t eval_reduced(reduced which, unsigned fractional_bits, std::i
case reduced::inv:
{
const dyadic part = split_positive(raw, fractional_bits);
const std::int64_t reciprocal = eval_principal(
const std::int64_t seed = eval_principal(
principal::inv, fractional_bits, part.mantissa_raw);
return shift_pow2(reciprocal, -part.power);
const u128 wide = newton_inv(fractional_bits, part.mantissa_raw, seed);
return finish_wide(wide, static_cast<int>(fractional_bits) + part.power);
}
case reduced::rsqrt:
{
const dyadic part = split_positive(raw, fractional_bits);
std::int64_t root = eval_principal(principal::rsqrt, fractional_bits, part.mantissa_raw);
const std::int64_t seed = eval_principal(
principal::rsqrt, fractional_bits, part.mantissa_raw);
u128 wide = newton_rsqrt(fractional_bits, part.mantissa_raw, seed);
if ((part.power & 1) != 0)
root = mul_raw(root, scale_unit(rsqrt2_64, fractional_bits), fractional_bits);
return shift_pow2(root, -half_pow_of(part.power));
wide = round_u256(mul_u128(wide, rsqrt2_64), 64);
return finish_wide(wide,
static_cast<int>(fractional_bits) + half_pow_of(part.power));
}
case reduced::invsq:
{
const dyadic part = split_positive(raw, fractional_bits);
const std::int64_t square = eval_principal(
principal::invsq, fractional_bits, part.mantissa_raw);
return shift_pow2(square, -2 * part.power);
const std::int64_t seed = eval_principal(
principal::inv, fractional_bits, part.mantissa_raw);
const u128 inv = newton_inv(fractional_bits, part.mantissa_raw, seed);
const unsigned K = fractional_bits * 2u;
const u128 wide = round_u256(mul_u128(inv, inv), K);
return finish_wide(wide, static_cast<int>(K) - static_cast<int>(fractional_bits)
+ 2 * part.power);
}
case reduced::expm1:
return eval_expm1(fractional_bits, raw);