Checkpoint the party/runtime stack before share-program and malicious-mode work.

Ship the TLS mesh, composer, Beaver/Yao/leaf MPC, prep/online paths, apps, and docs so the tree is pushable before elevating share_expr, security_mode, and prep resume.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Ryan Henry 2026-09-28 05:59:19 -06:00
parent 695f8e84f7
commit 0d22946a0e
1835 changed files with 170291 additions and 2849 deletions

View file

@ -1029,6 +1029,11 @@ INPUT = include/dpf.hpp \
doc/pages/multipoint.md \
doc/pages/dealer_free.md \
doc/pages/beaver.md \
doc/pages/yao.md \
doc/pages/arith.md \
doc/pages/compose.md \
doc/pages/network_and_mpc.md \
doc/pages/experiment.md \
doc/pages/api.md \
doc/pages/bibliography.md \
doc/pages/jet_and_ring.md \
@ -1489,7 +1494,8 @@ HTML_EXTRA_FILES = doc/doc-extras.js \
doc/papers/storrier-vadapalli-lyons-henry-grotto-eprint-2023-108.pdf \
doc/papers/boyle-gilboa-ishai-kolobov-it-dpf-eprint-2023-028.pdf \
doc/papers/chou-orlandi-simplest-ot-eprint-2015-267.pdf \
doc/papers/boyar-peralta-aes-sbox-eprint-2011-332.pdf
doc/papers/boyar-peralta-aes-sbox-eprint-2011-332.pdf \
doc/papers/reis-ugurbil-wagh-henry-de-vega-wave-hello-eprint-2025-013.pdf
# The HTML_COLORSTYLE tag can be used to specify if the generated HTML output
# should be rendered with a dark or light theme.

View file

@ -22,12 +22,19 @@
<tab type="user" url="@ref multipoint_keys" visible="yes" title="Multipoint keys"/>
<tab type="user" url="@ref multiparty" visible="yes" title="Multiparty &amp; 3-server"/>
<tab type="user" url="@ref dealer_free" visible="yes" title="Dealer-free keygen"/>
<tab type="user" url="@ref beaver_triples" visible="yes" title="Beaver triples"/>
<tab type="user" url="@ref jet_and_ring" visible="yes" title="Grotto"/>
<tab type="user" url="@ref repr_and_twist" visible="yes" title="Representation shift"/>
<tab type="user" url="@ref ppvc_manual" visible="yes" title="Programmable vectors"/>
<tab type="user" url="@ref applications" visible="yes" title="Application sketches"/>
</tab>
<tab type="usergroup" visible="yes" title="Network &amp; MPC" url="@ref network_and_mpc">
<tab type="user" url="@ref network_and_mpc" visible="yes" title="Overview"/>
<tab type="user" url="@ref protocol_compose" visible="yes" title="Protocol composition"/>
<tab type="user" url="@ref beaver_triples" visible="yes" title="Beaver triples"/>
<tab type="user" url="@ref arith_runtime" visible="yes" title="Arithmetic share runtime"/>
<tab type="user" url="@ref yao_leaf" visible="yes" title="Yao on a leaf"/>
<tab type="user" url="@ref experiment_costs" visible="yes" title="Logging &amp; statistics"/>
</tab>
<tab type="usergroup" visible="yes" title="Manual" url="@ref getting_started">
<tab type="user" url="@ref input_types" visible="yes" title="Domains"/>
<tab type="user" url="@ref output_types" visible="yes" title="Payloads"/>

View file

@ -116,10 +116,14 @@
function mountPageNav() {
var contents = document.querySelector("#doc-content .contents") || document.querySelector(".contents");
if (!contents || contents.querySelector(".page-nav")) return;
// Doxygen's theme makes `.contents` a row so the outline can sit beside
// the prose. A nav inserted as a sibling becomes another column and
// lands on top of the first paragraph. Keep it inside the text column.
var column = contents.querySelector(".textblock") || contents;
var top = pageNav("top");
if (!top) return;
contents.insertBefore(top, contents.firstChild);
contents.appendChild(pageNav("bottom"));
column.insertBefore(top, column.firstChild);
column.appendChild(pageNav("bottom"));
}
if (document.readyState === "loading") {

View file

@ -105,6 +105,9 @@
/// @example grotto/repr_and_twist.cpp repr_and_twist.cpp
/// @brief Fibonacci / geometric representation shift and twisted monomials
/// @example grotto/dwt_lut.cpp dwt_lut.cpp
/// @brief Haar and bior(5,3) compressed lookup tables
/// @}
@ -178,6 +181,9 @@
/// @example applications/sabre.cpp sabre.cpp
/// @brief Sabre mailbox write with a verifiable-DPF audit
/// @example protocol/compose_schedule.cpp compose_schedule.cpp
/// @brief Composer schedules: fused audit, early-stop, prefixes, RSS, ABY scale
/// @example applications/ledger23.cpp ledger23.cpp
/// @brief (2,3) ledger append with a verified point key
@ -205,4 +211,7 @@
/// @example mwe/ppvc.cpp ppvc.cpp
/// @brief a point-programmable vector commitment
/// @example mwe/shamir.cpp shamir.cpp
/// @brief (K,N) Shamir, with (2,3) as the slope specialization
/// @}

View file

@ -14,7 +14,7 @@ $generatedby&#160;<a href="https://www.doxygen.org/index.html"><img class="foote
</small></address>
</div><!-- doc-content -->
<!--END !GENERATE_TREEVIEW-->
<link href="$relpath^stylesheet.css?v=20260927" rel="stylesheet" type="text/css"/>
<script type="text/javascript" src="$relpath^doc-extras.js"></script>
<link href="$relpath^stylesheet.css?v=20260928" rel="stylesheet" type="text/css"/>
<script type="text/javascript" src="$relpath^doc-extras.js?v=20260928"></script>
</body>
</html>

View file

@ -5,13 +5,20 @@
/// outputs, and what is revealed. A protocol may open a masked value or a
/// public offset; that appears in the figure only when the parties learn it.
///
/// \htmlonly
/// <div class="eli5"><b>ELI5.</b> An ideal functionality is the specification the protocol is measured against: who holds what, what goes in, what comes out, and which values become public.</div>
/// \endhtmlonly
///
///
/// ## Sharing
///
///
/// - \ref secret_share.hpp "F_Open"
/// - \ref shamir3.hpp "F_Shamir"
///
/// ## Dealer keys
///
///
/// - \ref dpf_key.hpp "F_DPF"
/// - \ref incremental.hpp "F_IDPF"
/// - \ref grow.hpp "F_Grow"
@ -23,15 +30,20 @@
///
/// ## Two-party generation and evaluation
///
///
/// - \ref iknp.hpp "F_IKNP"
/// - \ref doerner_shelat.hpp "F_DS"
/// - \ref grow_ds.hpp "F_GrowDS"
/// - \ref geneval.hpp "F_GenEval"
/// - \ref beaver.hpp "F_Beaver and F_BeaverAuth"
/// - \ref yao.hpp "F_Yao"
/// - \ref yao_share.hpp "F_YaoShare"
/// - \ref constrained_cmp.hpp "F_CCMP"
/// - \ref verifiable.hpp "F_VDPF, F_Sketch, and F_OblivHash"
///
/// ## Three evaluators
///
///
/// - \ref dpf3.hpp "F_DPF3"
/// - \ref dpf3_ds.hpp "F_DPF3DS"
/// - \ref dpf3_cmp.hpp "F_DPF3CMP"
@ -39,6 +51,7 @@
///
/// ## Offset corrections
///
///
/// - \ref offset_horner.hpp "F_Horner"
/// - \ref offset_poly.hpp "F_Poly"
/// - \ref offset_jet.hpp "F_Jet"
@ -266,6 +279,25 @@
/// }
/// \enddot
/// @file dpf/iknp.hpp
///
/// @par Ideal functionality
/// \dot "Functionality F_IKNP"
/// digraph F_IKNP {
/// graph [bgcolor="transparent"];
/// node [shape=plaintext, fontname="Helvetica", fontsize=11];
/// F [label=<
/// <TABLE BORDER="1" CELLBORDER="0" CELLSPACING="0" CELLPADDING="7" COLOR="#1e293b" BGCOLOR="#f8fafc">
/// <TR><TD ALIGN="LEFT" BGCOLOR="#1e293b"><FONT COLOR="white" POINT-SIZE="12"><B>F_IKNP</B></FONT></TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Parties.</B> P0 and P1. Semi-honest. Base OT is Chou-Orlandi.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Input.</B> sample: both parties pass the same lengths<BR/>(bit x block, bit x bit, B2A, correction-word pads).<BR/>transfer_labels: the sender holds two 128-bit strings per row;<BR/>the receiver holds a choice bit per row.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Output.</B> sample gives each party its share of the pads:<BR/>bit x block, bit AND, a daBit, and a correction-word gamma<BR/>that hides the peer pad bit.<BR/>transfer_labels gives the receiver exactly the chosen string.<BR/>The sender's output buffer is cleared. An empty transfer sends nothing.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Leakage.</B> Lengths are public. Choice bits, the unchosen string,<BR/>and the peer pad bit stay hidden.<BR/>One role_state is one direction; the first call runs the base OT.</TD></TR>
/// </TABLE>
/// >];
/// }
/// \enddot
/// @file dpf/doerner_shelat.hpp
///
/// @par Ideal functionality
@ -617,3 +649,45 @@
/// >];
/// }
/// \enddot
/// @file dpf/yao.hpp
///
/// @par Ideal functionality
/// \dot "Functionality F_Yao"
/// digraph F_Yao {
/// graph [bgcolor="transparent"];
/// node [shape=plaintext, fontname="Helvetica", fontsize=11];
/// F [label=<
/// <TABLE BORDER="1" CELLBORDER="0" CELLSPACING="0" CELLPADDING="7" COLOR="#1e293b" BGCOLOR="#f8fafc">
/// <TR><TD ALIGN="LEFT" BGCOLOR="#1e293b"><FONT COLOR="white" POINT-SIZE="12"><B>F_Yao</B></FONT></TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Parties.</B> P0 garbles. P1 evaluates. Semi-honest.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Input.</B> A public straight-line bit netlist.<BR/>Each shared input is an XOR share of that bit.<BR/>A private input is known to one party.<BR/>The usual source of those bits is a DPF leaf, via F_YaoShare.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Output.</B> XOR shares of each output bit.<BR/>P0's share is the permute bit of the zero label.<BR/>P1's share is the color of the label it holds.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Leakage.</B> None beyond the output shares.<BR/>Tables are one-time. P1 does not learn Delta.<BR/>P0 does not learn P1's private bits or P1's shares.</TD></TR>
/// </TABLE>
/// >];
/// }
/// \enddot
/// @file dpf/yao_share.hpp
///
/// @par Ideal functionality
/// \dot "Functionality F_YaoShare"
/// digraph F_YaoShare {
/// graph [bgcolor="transparent"];
/// node [shape=plaintext, fontname="Helvetica", fontsize=11];
/// F [label=<
/// <TABLE BORDER="1" CELLBORDER="0" CELLSPACING="0" CELLPADDING="7" COLOR="#1e293b" BGCOLOR="#f8fafc">
/// <TR><TD ALIGN="LEFT" BGCOLOR="#1e293b"><FONT COLOR="white" POINT-SIZE="12"><B>F_YaoShare</B></FONT></TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Parties.</B> P0 and P1. For a replicated leaf, P2 is idle.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Input.</B> One share each of an integer the DPF already produced:<BR/>subtractive (point leaf), additive (comparison leaf),<BR/>an fss_share, or the party-0 and party-1 replicated views.<BR/>The reverse calls take XOR shares of the low width bits.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Output.</B> XOR shares of those bits, least-significant bit first,<BR/>or ring shares of the integer the bits encode, in the leaf's scheme.<BR/>y2rss deals a fresh replicated triple of that integer.</TD></TR>
/// <TR><TD ALIGN="LEFT"><B>Leakage.</B> A2B opens a masked x - r. B2A opens a masked bit.<BR/>Both masks are uniform. The integer stays shared.<BR/>This is not the local (3,3) cast dpf::rss2y / dpf::y2rss.</TD></TR>
/// </TABLE>
/// >];
/// }
/// \enddot
/// \htmlonly
/// <div class="tldr"><b>TL;DR.</b> Each figure states the parties, the inputs, the outputs, and what is revealed. A masked value or a public offset appears only when the parties learn it.</div>
/// \endhtmlonly

View file

@ -3,22 +3,14 @@
%LIBDPF_INCLUDE_GETTING_STARTED_INTRODUCTION%
<!-- PAGE SEPARATOR -->
\page basics First program
%LIBDPF_INCLUDE_GETTING_DPF_BASICS%
<!-- PAGE SEPARATOR -->
\page getting_started Manual
The domain, the payload, and the walk.
@ -31,112 +23,68 @@ Shorter domains make shorter keys.
Samples for each of those are under [Code examples](@ref listings).
<!-- PAGE SEPARATOR -->
\page input_types Domains
%LIBDPF_INCLUDE_GETTING_STARTED_INPUT_TYPES%
<!-- PAGE SEPARATOR -->
\page output_types Payloads
%LIBDPF_INCLUDE_GETTING_STARTED_OUTPUT_TYPES%
<!-- PAGE SEPARATOR -->
\page evaluation Evaluation
%LIBDPF_INCLUDE_GETTING_STARTED_EVALUATION%
<!-- PAGE SEPARATOR -->
\page iterables Iterables
%LIBDPF_INCLUDE_GETTING_STARTED_ITERABLES%
<!-- PAGE SEPARATOR -->
\page bugs Bugs
%LIBDPF_INCLUDE_MISC_BUGS%
<!-- PAGE SEPARATOR -->
\page changes Changelog
%LIBDPF_INCLUDE_MISC_CHANGES%
<!-- PAGE SEPARATOR -->
\page todo TODO
%LIBDPF_INCLUDE_MISC_TODO%
<!-- PAGE SEPARATOR -->
\page submodules Submodules
%LIBDPF_INCLUDE_MISC_SUBMODULES%
<!-- PAGE SEPARATOR -->
\page authors Authors
%LIBDPF_INCLUDE_MISC_AUTHORS%
<!-- PAGE SEPARATOR -->
\page license License
%LIBDPF_INCLUDE_MISC_LICENSE%
<!-- PAGE SEPARATOR -->
\page listings Code examples
%LIBDPF_INCLUDE_CODE_LISTINGS%

View file

@ -30,7 +30,10 @@ you want.
| [dpf::eval_interval](@ref dpf/eval_interval.hpp) | An inclusive range, one share per input. |
| [dpf::eval_sequence](@ref dpf/eval_sequence.hpp) | A sorted list of inputs. |
| [dpf::eval_full](@ref dpf/eval_full.hpp) | Every input in the domain. |
| [dpf::reconstruct](@ref dpf/secret_share.hpp) | Both shares. Leaf shares subtract. Comparison shares add. |
| [dpf::reconstruct](@ref dpf/secret_share.hpp) | Both shares. Leaf shares subtract. Comparison shares add. Shamir shares use Lagrange. |
| [dpf::shamir::deal](@ref dpf/shamir.hpp) / `make_shamir_shares` | (K,N) Shamir shares over `fp61` or `gf2n`. For `gf2n`, `N < 2^k`. `(2,3)` is `make_shamir_shares(secret, slope)`. |
| [dpf::shamir::share_secret](@ref dpf/random.hpp) | The same split with uniform higher coefficients. |
| [dpf::shamir::reconstruct](@ref dpf/shamir.hpp) | Any K shares. Further shares are a consistency check, not a correction. |
| [dpf::eval_point](@ref dpf/interval.hpp) `(dpf::ic, ...)` | One public input on an interval key. |
## Comparisons and tags
@ -50,19 +53,55 @@ you want.
The catalogs, with the types that are inputs and the types that are outputs:
- [Input types](@ref input_types): integers, `modint`, `xint`, `bitstring`, `keyword`, `keyword2`, fixed-point.
- [Output types](@ref output_types): `bit`, `twobit`, `nyble`, fields, curve points, shares, `vec`.
- [Output types](@ref output_types): `bit`, `twobit`, `nyble`, `gf2` through `gf264`, prime fields, curve points, shares, `vec`.
`bit`, `twobit`, and `nyble` are packed output lanes. `keyword2` is a domain, not a leaf.
`bit`, `twobit`, `nyble`, and `gf2` / `gf22` / `gf24` are packed output lanes. `keyword2` is a domain, not a leaf.
## After the offset is public
[Grotto](@ref guided_tour) evaluates a function of x once the public offset is open.
Several piecewise LUTs share one comparison via [make_lut_union_plan](@ref grotto/lut_union.hpp).
The pages are [offset Horner, jets, and ring switch](@ref jet_and_ring) and
[representation shift and twisted jets](@ref repr_and_twist).
Haar and bior(5,3) tables are [make_haar_dwt_lut](@ref grotto/dwt_lut.hpp)
and [make_bior53_dwt_lut](@ref grotto/dwt_lut.hpp).
## Multiplication and sessions
| Call | What it does |
| --- | --- |
| [dpf::beavers::session](@ref dpf/beaver.hpp) | ABY2.0 blinds, products, dots, and polynomial schedules. |
| [dpf::yao::b2y](@ref dpf/yao_share.hpp) / `a2y` / `fss2y` / `rss2y` | A leaf share to LSB-first XOR bits. Point leaves use `b2y`. Comparisons use `a2y`. |
| [dpf::yao::y2b](@ref dpf/yao_share.hpp) / `y2a` / `y2fss` / `y2rss` | Those bits back to the leaf's share type. |
| [dpf::yao::netlist](@ref dpf/yao.hpp) / `session::eval` | The boolean circuit on those bits. Party 0 garbles. |
| [dpf::arith_garble::circuit](@ref dpf/arith_garble.hpp) | Free add, public scale, projection. Ball–Malkin–Rosulek. |
| [dpf::flute::eval_pair](@ref dpf/flute.hpp) / `eval_trio` | Public LUT on masked bits. Two or three online bits per output. A DPF point stays a key. |
| [dpf::yao::eval_if](@ref dpf/yao_stack.hpp) / `eval_one_hot` | Stacked branch and k-way switch. Rows follow the heaviest branch. |
| [dpf::yao::aes128](@ref dpf/yao_aes.hpp) / `aes_mmo` | Packaged AES-128 and the zero-key MMO block on that session. |
| [dpf::beavers::schedule_objective](@ref dpf/beaver.hpp) | `prep` peels for Appendix E; `rounds` keeps one online round. |
| [dpf::protocol::composer](@ref dpf/compose.hpp) | Domain-tagged FSS / ABY / RSS strands on one RoundSink plan. |
| [dpf::shuffle::shuffle_hidden_pass](@ref dpf/shuffle.hpp) | Hidden reorder of an RSS column. A secret index stays a DPF. |
| [dpf::protocol::composer::shuffle_hidden](@ref dpf/compose.hpp) | Three `shuffle_send` waves for that reorder. |
| [dpf::protocol::composer::client_servers](@ref dpf/compose.hpp) | PIR: one upload round, one answer round, no server-server open. |
| [dpf::protocol::plan_to_schedule](@ref dpf/compose.hpp) / `drive_via_schedule` | Lower a plan onto `schedule_session` (edge, receive rule, branch/next). |
| [dpf::protocol::schedule_session](@ref dpf/protocol.hpp) | Ready instance runs on this thread; flush sends the largest prefix per edge. |
| [dpf::net::edge_mesh](@ref dpf/net/edge_mesh.hpp) / `make_memory_star` | N duplex RoundSinks (star / clique / dealer). |
| [dpf::protocol::session_host](@ref dpf/session_host.hpp) | Queue micro-plans on a durable mesh. |
| [dpf::protocol::iknp_setup_graph](@ref dpf/iknp_graphs.hpp) / `du_atallah_mul_graph` | IKNP / Du-Atallah / star upload-answer as schedule rounds. |
| [dpf::protocol::pirsona_bitmore_fetch](@ref dpf/mesh_apps.hpp) / `hushmap_add_schedule` | PIRsona BitMore fetch and hushmap ADD skeletons. |
| [dpf::protocol::drive_star](@ref dpf/app_plans.hpp) / named `*_plan` helpers | Star drive + application micro-plans (PIR, mailbox, SUBLEQ, Pika, …). |
| [dpf::app::run](@ref dpf/app_flow.hpp) / `run_plan` | Drive both parties and print `rounds` and `bytes`. |
| [dpf::log](@ref dpf/log.hpp) / [app::start_logging](@ref dpf/run_log.hpp) | Leveled run log and provenance banner. |
| [dpf::experiment](@ref dpf/experiment.hpp) / [Logging & statistics](@ref experiment_costs) | Replayable master seed; CSV cost breakdown. |
| [dpf::app::measure_plan](@ref dpf/app_flow.hpp) / `run_measured` | Drive a plan under an experiment; optional `DPF_EXPERIMENT_DIR` CSV dump. |
| [dpf::app::run_fleet](@ref dpf/app_flow.hpp) | Many instances. Parked receives yield to the side that is behind. |
## Where to read next
- [Which DPF?](@ref which_dpf) if you are still choosing the object.
- [Evaluating DPFs](@ref evaluation) for point, interval, sequence, and full-domain cost.
- [Network, parties, and MPC](@ref network_and_mpc) for the runtime around the keys.
- [Logging, statistics, and experiments](@ref experiment_costs) for the run log and cost CSVs.
- [Protocol composition](@ref protocol_compose) for fused walks, early-stop, and RSS refresh.
- [Bibliography](@ref bibliography) for the papers behind the keys.
- [Application mockups](@ref applications) for the DPF step inside a larger protocol.

View file

@ -5,9 +5,18 @@ Each one is a single process.
A dealer stands in where the paper generates keys from shares.
They compile from the repository root:
c++ -std=c++17 -march=native -I include -I thirdparty examples/applications/duoram3.cpp
c++ -std=c++17 -march=native -pthread -I include -I thirdparty examples/applications/duoram3.cpp
The online compose sketch at the end of each file uses
[`dpf::app::run_measured`](@ref dpf/app_flow.hpp): it prints rounds, live
bytes, wall/CPU, PRG evals, random bytes, and the master seed hex. Set
`DPF_EXPERIMENT_DIR=/tmp/run` to also write the CSV tables described in
[Experiments](@ref experiment_costs).
The same compile line with the other filenames builds the rest. Timing for
the party protocols, including word garbling, stacked branches, FLUTE, and
the hidden column reorder, is `party_bench` / `profile_party` (see
[The battery](@ref experiment_costs)).
The same line with the other filenames builds the rest.
Optional Python bindings configure with `-DLIBDPF_PYTHON=ON` in the test
build directory, then `make pydpf` (and `make pydpf_pytest`).
`pydpf` exposes point / interval / full / sequence / recipe eval on
@ -16,7 +25,6 @@ and `it_dpf3`. Not in this module yet: geneval, Doerner–Shelat, VDPF
`prove`/`sketch`, DCF, or grotto.
What those programs had to do by hand is [the library surface underneath](@ref application_gaps).
| Sketch | What the DPF step is |
| --- | --- |
| [3-party Duoram](@ref app_duoram) | Unit key, rotate, inner product |
@ -38,17 +46,25 @@ What those programs had to do by hand is [the library surface underneath](@ref a
| [Range count](@ref app_range_count) | Interval payload |
| [Floram](@ref app_floram) | ORAM read |
| [Three-server PIR](@ref app_pir3) | Information-theoretic DPF |
| [Protocol composition](@ref protocol_compose) | Fused walks, early-stop, RSS refresh, ABY scale |
| PIRsona fetch | BitMore star upload+answer (`pirsona_fetch.cpp`) |
| Hushmap ADD | Dealer tape + two opens (`hushmap_add.cpp`) |
| [What the walk folds in](@ref application_gaps) | Library surface those programs used to do by hand |
## 3-party Duoram {#app_duoram}
\htmlonly
<div class="eli5"><b>ELI5.</b> Preprocessing plants a unit DPF at a random index r. Online the parties open i* − r, rotate the expanded unit vector by that public shift, and dot with the memory. The update rotates a payload vector the same way and adds it in.</div>
\endhtmlonly
Vadapalli, Henry, and Goldberg ([USENIX Security 2023](@ref bib_duoram)) keep a memory in
shares and read or add at a secret index.
Preprocessing builds unit DPFs at a random index `r`.
Online, the parties open `i* - r` and cyclic-shift the expanded vector.
The read is the dot product of that vector with the memory.
The update adds a payload vector, shifted the same way.
Opening the whole memory, rather than one index, is a hidden shuffle of
that column ([an array of shares](@ref share_shuffle)), not another DPF.
The program uses one dealer unit key for the read and one payload key
for the update.
@ -71,6 +87,10 @@ three evaluators.
## MPC SUBLEQ {#app_subleq}
\htmlonly
<div class="eli5"><b>ELI5.</b> The address is not known when the keys are built, so the unit vectors are expanded early into a deferred buffer. Online, opening the address rotates that buffer. The read is a dot with memory; the write adds the scaled unit vector back.</div>
\endhtmlonly
Jiang and Henry ([MSc thesis, University of Calgary](@ref bib_subleq)) emulate the
subtract-and-branch-if-less-than-or-equal-to-zero (SUBLEQ) OISC for
private function evaluation. One instruction is
@ -109,6 +129,10 @@ and out-of-bounds prefix-parity checks from the thesis.
## BitMore, `2^L` servers {#app_bitmore}
\htmlonly
<div class="eli5"><b>ELI5.</b> The label is L bits. Each bit is its own 1-bit DPF, expanded over the whole domain. Stacking the L bit-vectors and reading a column produces the 2^L-server answer for that label.</div>
\endhtmlonly
Hafiz and Henry ([PoPETs 2019](@ref bib_bitmore), §5.2) query `ell = 2^L` servers with `L`
independent 1-bit DPFs, all at the same row.
Server `j` receives key number `j_e` from DPF `e`.
@ -131,6 +155,10 @@ The two answers XOR to the record.
## Keyword PIR {#app_keyword}
\htmlonly
<div class="eli5"><b>ELI5.</b> The keyword is hashed into cuckoo buckets. Each bucket is a point key, and the record is the inner product of the probed buckets with the dictionary. The S&amp;P 2025 seed-packing of those buckets is not what this program does.</div>
\endhtmlonly
Gilboa and Ishai ([EUROCRYPT 2014](@ref bib_dpf2014)) retrieve one record by a keyword.
[dpf::keyword](@ref dpf/keyword.hpp) is the domain, so the DPF point is
the keyword itself.
@ -158,6 +186,10 @@ stack.
## Prio and the heavy-hitter prefix walk {#app_prio}
\htmlonly
<div class="eli5"><b>ELI5.</b> A one-hot vote is a unit DPF in the Prio field. Each server adds the expanded vector into a running histogram. The heavy-hitter pass is the same key read at successive prefixes: the servers add the opened prefix shares and keep the heavy nodes.</div>
\endhtmlonly
Corrigan-Gibbs and Boneh ([NSDI 2017](@ref bib_prio)) aggregate client encodings.
A frequency count is a one-hot vector.
A unit DPF is that vector, compressed.
@ -182,6 +214,10 @@ and add the opened values.
## I-DPF max and k-th {#app_idpf_agg}
\htmlonly
<div class="eli5"><b>ELI5.</b> Each secret integer is one incremental DPF with a unit payload on every prefix. At each bit the servers open the two children. Max keeps the child that holds mass. The k-th keeps the 1-child when its count covers k, and otherwise descends the 0-child with k reduced.</div>
\endhtmlonly
Cheng, Mitrokotsa, Zhang, and Hartmann ([ePrint 2024/1190](@ref bib_idpfagg)) aggregate
secret values with an incremental DPF.
Communication tracks the bit length of the domain, not how many secret
@ -200,6 +236,10 @@ share that walk with the gtest.
## LLAMA {#app_llama}
\htmlonly
<div class="eli5"><b>ELI5.</b> The comparison is the gate. eval_point on a gt or lt key opens to the payload when the public query is on the true side of the secret, including across the sign bit, because signed inputs flip the high bit before the walk.</div>
\endhtmlonly
Gupta, Kumaraswamy, Chandran, and Gupta ([ePrint 2022/793](@ref bib_llama)) evaluate a
nonlinear gate from a dealer key and one opened masked input
`x_hat = x + r`.
@ -214,6 +254,10 @@ to 1.
## Pika {#app_pika}
\htmlonly
<div class="eli5"><b>ELI5.</b> The dealer keys a unit DPF at a fresh r and the parties open x = r − a. Rotating the public table by x and dotting with the DPF reads the entry at a. The sign of an early-stop bit leaf is recorded at keygen, so the evaluators never open r.</div>
\endhtmlonly
Wagh ([PoPETs 2022](@ref bib_pika), Fig. 1) looks up `Func(a)` in a table of a bounded
domain.
The dealer keys a unit DPF at a fresh index `r` and the parties open
@ -224,11 +268,17 @@ A word payload of `1` opens to `+1`.
The paper's early-stop bit leaf opens to `+1` or `-1`; the dealer records
that sign at keygen with `dpf::unit_sign` (the final control bit `Gen`
sees), so the evaluators never open `r`.
A networked early-stop *walk* (BGI Remark 3.4) drops ν CW rounds with
`fss_point_early_stop` on a [composer](@ref protocol_compose).
\include{cpp} applications/pika.cpp
## Express {#app_express}
\htmlonly
<div class="eli5"><b>ELI5.</b> A mailbox write is a full-domain add of one DPF into the shared array. Every box is touched by the expand; only the programmed box survives when the shares are opened.</div>
\endhtmlonly
Eskandarian, Corrigan-Gibbs, Zaharia, and Boneh ([USENIX Security 2021](@ref bib_express),
§3.1) write one mailbox.
Two servers hold subtractive shares of the mailboxes.
@ -246,10 +296,19 @@ The extractable full-domain leaf now matches point evaluation on every
lane, so this one call replaces the earlier `eval_point`-per-address
loop and the separate `sketch_fold` pass.
On a networked walk the audit share rides in the last correction-word
flush — `fss_point_fused` / `level_walk_fused` on a
[composer](@ref protocol_compose) — so the sketch does not add a round.
The same shape is Sabre's proof token.
\include{cpp} applications/express.cpp
## PRAC {#app_prac}
\htmlonly
<div class="eli5"><b>ELI5.</b> Binary search needs a unit vector on a stride that grows by one bit per comparison. One incremental key holds all of those prefixes, and a prefix inner product dots a stride without a separate point key per slot. A heap update is a three-lane vector at the parent and its two children.</div>
\endhtmlonly
Sasy, Vadapalli, and Goldberg ([ePrint 2023/1897](@ref bib_prac)) run dynamic data
structures on a 3-party Duoram.
The new DPF shapes are an incremental key and a wide leaf.
@ -280,6 +339,10 @@ The protocol appends each comparison bit after the key exists.
## Splinter {#app_splinter}
\htmlonly
<div class="eli5"><b>ELI5.</b> The secret WHERE value is a unit DPF. The server dots it with a column that was already summed by attribute, which is the grouped SUM, and with an all-ones column, which is the COUNT. The queried attribute is not revealed.</div>
\endhtmlonly
Wang, Yun, Goldwasser, Vaikuntanathan, and Zaharia ([NSDI 2017](@ref bib_splinter)) answer
private queries on public data with two-server FSS.
The client's private `WHERE` value is a unit DPF at that attribute.
@ -296,6 +359,10 @@ one selector DPF; Splinter composes several FSS instances for those.
## Mastic {#app_mastic}
\htmlonly
<div class="eli5"><b>ELI5.</b> This is the Poplar prefix walk with a weight instead of 1 on every prefix. Servers sum the prefix shares across clients and drop prefixes under the threshold. A path sketch can check that each client programmed a single path.</div>
\endhtmlonly
Mastic (private weighted heavy-hitters and attribute-based metrics) is
Poplar's prefix walk with a weight payload.
Each client keys an [idpf](@ref dpf/placement.hpp) whose β on every
@ -311,6 +378,10 @@ heavy with total weight 8.
## Waldo {#app_waldo}
\htmlonly
<div class="eli5"><b>ELI5.</b> Each append is a fresh unit DPF added into the value shares; old events are not rewritten. A threshold query is a comparison inner product of those shares with a public magnitude column, so the sum past a secret threshold does not reveal the threshold or the matches.</div>
\endhtmlonly
Dauterman, Rathee, Popa, and Stoica ([S&P 2022](@ref bib_waldo)) build a private time-series
database from FSS.
The store is append-only: each event is a fresh unit DPF folded into the
@ -330,6 +401,10 @@ comparison key.
## Sabre {#app_sabre}
\htmlonly
<div class="eli5"><b>ELI5.</b> The write is the same full-domain add as Express. The audit folds a constant-size proof token along that key. verify accepts one honest point and rejects a key that was hot in more than one place.</div>
\endhtmlonly
Vadapalli, Storrier, and Henry ([S&P 2022](@ref bib_sabre)) send anonymous messages with a
fast audit.
The write is Express's full-domain add (`eval_full_add_into`).
@ -337,6 +412,8 @@ The audit is a *verifiable* DPF proof rather than Express's `fp61`
sketch: `prove_full(key, dpf::prove(π))` folds a constant-size token per
party, and `dpf::verify(π0, π1)` accepts an honest single-point write and
rejects the mismatched fold a multi-point key produces.
Networked audits pack the proof share into the last CW with
`fss_point_fused` ([protocol composition](@ref protocol_compose)).
Still by hand: Sabre's blame / accountability phase that identifies a
cheating client is protocol logic above the DPF proof.
@ -345,6 +422,10 @@ cheating client is protocol logic above the DPF proof.
## A (2,3) ledger {#app_ledger23}
\htmlonly
<div class="eli5"><b>ELI5.</b> An append is one verifiable three-party point at the slot. The three proof tokens are checked before the point is added into the slot shares. Any two servers reconstruct a balance.</div>
\endhtmlonly
A replicated ledger held as (2-of-3) shares by three servers, on this
group's `dpf3` VDPF+ construction.
Each append is one `make_dpf3(slot, amount, dpf::verifiable{})`; a
@ -362,6 +443,10 @@ Still by hand: the transaction / consensus layer around the append
## Private set intersection {#app_psi}
\htmlonly
<div class="eli5"><b>ELI5.</b> Each element on one side is a unit DPF. Dotting it with the other side's table is the membership test. The cuckoo layout and the OPRF that would sit around that test are not in the DPF call.</div>
\endhtmlonly
Kolesnikov, Kumaresan, Rosulek, and Trieu ([CCS 2016](@ref bib_kkrt)) test membership
with an oblivious PRF.
On this domain the PRF is a table both servers hold.
@ -380,6 +465,10 @@ DMPF seed packing needs a PCG this library does not provide.
## Range count {#app_range_count}
\htmlonly
<div class="eli5"><b>ELI5.</b> A value falls in [lo, hi) when the greater-than bit at hi and the greater-than bit at lo differ. Each secret value is one comparison key. The count is the sum of those opened bits.</div>
\endhtmlonly
Each secret value is one comparison.
The interval `[lo, hi)` is public.
`eval_point(dpf::cmp, key, q)` opens to 1 when `q` is strictly above the
@ -396,6 +485,10 @@ This program is the other direction, secret values and a public range.
## Floram {#app_floram}
\htmlonly
<div class="eli5"><b>ELI5.</b> The address is shared, not known to a dealer. The Doerner–Shelat opening builds the unit key level by level, and the read or write is the inner product of that key with the array.</div>
\endhtmlonly
Doerner and shelat ([CCS 2017](@ref bib_ds)) read and write an array at a secret
address.
Both parties see the memory.
@ -413,6 +506,10 @@ is the FSS access.
## Three-server PIR {#app_pir3}
\htmlonly
<div class="eli5"><b>ELI5.</b> All three servers hold the database. A Shamir DPF3 key makes each server return an inner product; any two of those field elements open the record. The information-theoretic key instead adds all three inner products.</div>
\endhtmlonly
The database is public and replicated on three servers.
The computational path is one (2,3) point key from
@ -432,8 +529,21 @@ three dots sum to the record. Distinct from `make_dpf3`.
## What the walk now folds in {#application_gaps}
\htmlonly
<div class="eli5"><b>ELI5.</b> Rotate-then-dot, the sign of a bit leaf, bit columns, prefix dots, and path sketches used to be loops around eval. They are parameters of one walk now.</div>
\endhtmlonly
The calls the eight programs used to build by hand are now the library
surface. See [dpf/eval_walk.hpp](@ref dpf/eval_walk.hpp).
Multi-protocol *schedules* (FSS + ABY + RSS on one RoundSink) are
[protocol composition](@ref protocol_compose). Each listing under
`examples/applications/` records that paper's online flow on a
`composer` and calls `dpf::app::run`, which drives both parties on an
in-process sink and prints `name rounds= bytes=`. That line is the
experiment: compare it with the round and bandwidth column of the
paper. PIR listings are a client and two or three servers (one upload
round, one answer round). The servers do not open shares with each
other.
## Shift, then add {#gap_shift}
@ -562,3 +672,7 @@ full-domain `H` publishes nothing. The walk is O(|H| · n) and never
materializes the domain. An audit opening of a replica-seed pool is that
copath with `program_hidden = false`, so the live seeds stay out. The
one-point layout stays for PSI; `{α}` with programming agrees with it.
\htmlonly
<div class="tldr"><b>TL;DR.</b> Each program is only the DPF step, in one process, with a dealer standing in for shared keygen. Memory reads are a unit vector, a public shift or a prefix, and a dot. PIR and grouped sums are inner products. Heavy hitters are prefix walks. Mailbox writes and the ledger are full-domain adds, plus a proof when the write must be a single point.</div>
\endhtmlonly

151
doc/pages/arith.md Normal file
View file

@ -0,0 +1,151 @@
# Arithmetic share runtime {#arith_runtime}
\htmlonly
<div class="eli5"><b>ELI5.</b> Beside FSS keys you get ordinary secret shares: add them locally, multiply with Beaver or RSS, flip between bits and numbers with edaBits, and truncate fixed-point products. The composer schedules those opens next to FSS walks.</div>
\endhtmlonly
Semi-honest 2PC and honest-majority 3PC. Openings may carry Shark/SPDZ IT-MACs
from the beaver session. A leaf that must enter a boolean netlist uses
[b2y / a2y](@ref yao_leaf) and comes back with `y2b` / `y2a`. There is no
Yao domain on the composer.
## Domains
| Domain | Meaning |
| --- | --- |
| `a` / `b` / `fss` | Existing additive, subtractive, FSS leaf |
| `rss` / `y` | Existing replicated / product factor |
| `bin` | (2,2) XOR bit shares (packed) |
| `bin_rss` | (2,3) replicated bits |
`composer::as` never casts to or from `bin` / `bin_rss`. Use `bin_a2b` /
`bin_b2a` / `bin_inject` (see [compose.hpp](@ref dpf/compose.hpp) opcodes
400–405).
## Building blocks
| Header | Role |
| --- | --- |
| [rss_seed.hpp](@ref dpf/rss_seed.hpp) | Pairwise PRG seeds, zero-sharing, RSS mul local |
| [ot_pack.hpp](@ref dpf/ot_pack.hpp) | Consumable B2A / bit pads or dealer dabits |
| [edabit.hpp](@ref dpf/edabit.hpp) | daBits and edaBits; A2B |
| [bit_inject.hpp](@ref dpf/bit_inject.hpp) | Bit × arithmetic (2PC / RSS) |
| [trunc.hpp](@ref dpf/trunc.hpp) | Probabilistic and exact truncate; mul_trunc |
| [share_cmp.hpp](@ref dpf/share_cmp.hpp) | Share–share compare, ReLU, max, range, div |
| [share_vec.hpp](@ref dpf/share_vec.hpp) | Lane-block share vectors |
| [fixed_share.hpp](@ref dpf/fixed_share.hpp) | `fixed<Int,Frac>` |
| [share_expr.hpp](@ref dpf/share_expr.hpp) | Imperative recorder |
| [gilboa.hpp](@ref dpf/gilboa.hpp) | One-off product / tape fill |
| [matmul.hpp](@ref dpf/matmul.hpp) | Matrix triples and matmul |
| [shuffle.hpp](@ref dpf/shuffle.hpp) | Hidden reorder of an RSS column. A secret index stays a DPF |
| [arith_garble.hpp](@ref dpf/arith_garble.hpp) | Free add and a unary projection, after a word has left the key |
| [flute.hpp](@ref dpf/flute.hpp) | Public table on short masked bits. A DPF point stays a key |
| [cost_pass.hpp](@ref dpf/cost_pass.hpp) | DCF vs edaBit vs A2B strategy |
| [yao.hpp](@ref dpf/yao.hpp) | Half-gates netlist on a leaf's XOR bits. Party 0 garbles |
| [yao_share.hpp](@ref dpf/yao_share.hpp) | Leaf share to those bits and back (`b2y`, `a2y`, `fss2y`, `rss2y`) |
| [yao_aes.hpp](@ref dpf/yao_aes.hpp) | The PRG's zero-key MMO block, and AES-128 under a shared key |
## When to use which comparison
- **One input public or keyed:** DCF / `geneval_*` / interval keys.
- **Both inputs arithmetic shares:** `share_cmp` (mask, open, MSB / DCF at the public difference).
- **The leaf must enter a deep bit circuit:** [Yao](@ref yao_leaf). Not a comparison, a mux, or a public-offset LUT.
## A small word after the key {#arith_garble_word}
Comparisons, intervals, and public-offset polynomials stay on the key.
[Grotto](@ref jet_and_ring) covers a public offset. This section is the
word you already hold as shares, when the next step is addition, a public
scale, or a unary map, and you do not want an opening between the gates.
[arith_garble.hpp](@ref dpf/arith_garble.hpp) is that circuit
([Ball, Malkin, and Rosulek, CCS 2016](@ref bib_garble_gadgets)). Addition
is free. Scaling by a public constant coprime to the modulus is free. A
unary map of a mod-`m` wire sends `m − 1` ciphertexts. A threshold of `b`
bits that already live in `Z_{b+1}` is one such map, so the row count is
`b`. A product in a small prime field is the discrete-log reduction: project
to the exponent, add, project back, and drop the zero cases.
The beaver [session](@ref beaver_triples) is the other tool. It opens one
masked wire and multiplies interactively. It does not garble these gadgets.
```cpp
dpf::arith_garble::circuit c;
auto x = c.input(7);
auto y = c.input(7);
c.out(c.mul(c.add(x, y), x));
auto opened = dpf::arith_garble::eval_pair(c, in);
```
`opened.mask` is the garbler's share and `opened.color` is the evaluator's.
`open_shares` subtracts them mod the output modulus.
## A public table on short shares {#flute_lut}
A secret point in a public table is a DPF dotted with that table. FLUTE is
the case where the index is already a handful of masked bits sitting in an
ABY2.0 or three-party XOR sharing, and the result has to stay in that
sharing.
[flute.hpp](@ref dpf/flute.hpp) follows Brüggemann, Hundt, Schneider, Suresh,
and Yalame, [IEEE S&P 2023](@ref bib_flute). The table is an inner product of
those bits. The online exchange is two bits per output bit for two parties,
and three bits per output bit for `eval_trio`, independent of how wide the
index is. The index is at most 8 bits. Bit 0 of the index is the least
significant bit of the row.
```cpp
auto pair = dpf::flute::eval_pair(delta, n_out, columns, bits);
auto trio = dpf::flute::eval_trio(delta, n_out, columns, bits);
```
`columns[w * 2^δ + row]` is output bit `w` on that row. `opened` is the
clear bit. `masked` XOR the party's mask shares is the same bit.
## An array of shares, not a secret index {#share_shuffle}
A secret index is a DPF. One read of a shared memory is a unit key, a public
rotate, and a dot product, as in [3-party Duoram](@ref app_duoram). The
shuffle is the other job: the parties already hold a share of every row, and
the next step opens values. The opened order must not be the stored order.
`shuffle_party` derives one permutation from `k01` and applies it to every
component. A caller who passes the whole seed bundle can recompute that
order. Use it when the order is allowed to be known.
`shuffle_hidden_pass` is the order no single party should learn. Three
passes use the pairwise seeds `k01`, `k12`, and `k20`, leaving out the party
who does not hold that seed (party 2, then 0, then 1). Each party is given
`rss::party_seeds` only. On a pass, the two parties who share the seed
permute a two-party split of the column. The left-out party sends nothing
and receives one fresh component. Both messages go to the sender's RSS
neighbor. `permute` means `out[i] = in[pi[i]]`. Replaying the clear column
is `π01`, then `π12`, then `π20`.
```cpp
auto opened = dpf::shuffle::shuffle_hidden_triple(column, bundle, /*index=*/0);
```
`composer::shuffle_hidden(n, value_bytes)` records those three exchanges as
`shuffle_send`. `aux` on each node is the left-out party.
This is for a column you already share: a Duoram memory before a bulk open,
a histogram, a PSI payload. It does not replace a key. One hidden cell is
still a DPF. A comparison or a sort by a secret key is a DCF or
`share_cmp`. A gather to data-dependent indices is a DPF per access. The
seed permutation is chosen before the data.
## Truncation
- `trunc_prob` — local right shift, error in `{0,1}`, no round.
- `trunc_exact` — edaBit / carry correction.
- `mul_trunc` — product then shift (fixed-point multiply).
## Networking
`schedule_round` carries `phase::{setup,online}`, `round_dir::{duplex,send_next,recv_prev}`,
and `receive_rule::ring_next`. `drive_options` adds `pipeline_credit`, `cleartext`,
and `check_open`. Pairwise seeds replace `dealer_zero` via `rss_zero_mask`.
**Go deeper:** [Beaver](@ref beaver_triples), [compose](@ref protocol_compose),
[Grotto carry](@ref jet_and_ring).

View file

@ -62,6 +62,10 @@ Width literals (`100_u12`, `7_x12`, `1.5_fixed16`, `1_bit`, `2_twobit`,
## DPF Trees {#dpf_trees}
\htmlonly
<div class="eli5"><b>ELI5.</b> A key stores a root seed and one correction word per level. Expanding the seed walks the tree. On the secret path the correction forces the leaf to the programmed value; off that path the two parties' corrections cancel, so the opened value is zero.</div>
\endhtmlonly
Keys store a seed and a list of *correction words*.
Evaluation walks a binary tree from the root toward `x`.
At each level a correction word mixes the two children so only the secret
@ -79,3 +83,7 @@ full-domain evaluation versus `2N`. See [tree_traits.hpp](@ref dpf/tree_traits.h
For a slow, friendly walk through every feature, start at the
[guided tour](@ref guided_tour).
\htmlonly
<div class="tldr"><b>TL;DR.</b> A (2,2) DPF gives each party a short key for one secret point. make_dpf builds it. Point leaves open by subtraction and comparisons by addition. The key is a seed plus one correction word per level.</div>
\endhtmlonly

View file

@ -1,16 +1,39 @@
# Beaver triples {#beaver_triples}
\htmlonly
<div class="eli5"><b>ELI5.</b> A product opens as d = x − a and e = y − b, with a and b the preprocessing blinds. The product share is de plus the blinded cross terms, all local once d and e are public. The session keeps blinds that were already opened and only samples monomials it has not seen.</div>
\endhtmlonly
ABY2.0-style sessions open masked wires once (Patra, Schneider, Suresh,
and Yalame, USENIX Security 2021 / [ePrint 2020/1225](@ref bib_aby2)).
A fresh triple follows Beaver, [CRYPTO 1991](@ref bib_beaver): both masked factors are
reconstructed, and the product share is a local correction.
Optional MAC tags are the Shark/SPDZ check.
Constant-round word arithmetic is a separate gadget. [Ball, Malkin, and
Rosulek, CCS 2016](@ref bib_garble_gadgets) give free addition, free scaling
by a public constant, and a unary projection of `m − 1` ciphertexts.
[arith_garble.hpp](@ref dpf/arith_garble.hpp) is that circuit. A session does
not become one: it still opens δ once per wire. A public table on masked
bits is [FLUTE](@ref bib_flute) in [flute.hpp](@ref dpf/flute.hpp): the table
is a multi-fan-in inner product, and the online exchange is two bits per
output bit. `eval_trio` is the same product on three XOR shares of each mask.
One call that samples a list of formulae opens the new wires in one
round. Communication is one masked value per newly opened wire, plus a
tag share of the same width when MACs are on. Preprocessing is one blind
per wire and one product share per monomial.
**Go deeper:** [beaver.hpp](@ref dpf/beaver.hpp),
`schedule_objective::prep` (default) peels shared factors for Appendix-E
prep savings and may add interactive rounds.
`schedule_objective::rounds` emits the polynomial in one online round
(Pika / online Grotto). Composer-owned sessions use `rounds`.
Compose FSS walks, ABY products, and RSS refreshes on one sink with
[protocol composition](@ref protocol_compose).
**Go deeper:** [a small word](@ref arith_garble_word),
[a public table](@ref flute_lut), [beaver.hpp](@ref dpf/beaver.hpp),
[compose.hpp](@ref dpf/compose.hpp),
[F_Beaver](@ref beaver.hpp), [F_BeaverAuth](@ref beaver.hpp),
and the cost notes in the [guided tour](@ref tour_beaver).

View file

@ -169,7 +169,121 @@ USENIX Security 2021. Full version:
One public reconstruction per newly opened wire.
**Used in** [Beaver triples](@ref beaver_triples) · [Guided tour](@ref tour_beaver)
**Used in** [Beaver triples](@ref beaver_triples) · [Arithmetic share runtime](@ref arith_runtime) · [Guided tour](@ref tour_beaver)
### ABY {#bib_aby}
Daniel Demmler, Thomas Schneider, and Michael Zohner.
*ABY — A Framework for Efficient Mixed-Protocol Secure Two-Party Computation.*
NDSS 2015.
[ePrint 2014/386](https://eprint.iacr.org/2014/386)
Arithmetic / boolean / Yao sharing and conversions.
**Used in** [Arithmetic share runtime](@ref arith_runtime) · [A boolean function of a leaf](@ref yao_leaf)
### Half-gates {#bib_halfgates}
Samee Zahur, Mike Rosulek, and David Evans.
*Two Halves Make a Whole: Reducing Data Transfer in Garbled Circuits using Half Gates.*
EUROCRYPT 2015.
[ePrint 2014/756](https://eprint.iacr.org/2014/756)
Free-XOR AND rows. `yao::session` sends two blocks per AND.
**Used in** [A boolean function of a leaf](@ref yao_leaf) · [Arithmetic share runtime](@ref arith_runtime)
### Garbling gadgets {#bib_garble_gadgets}
Marshall Ball, Tal Malkin, and Mike Rosulek.
*Garbling Gadgets for Boolean and Arithmetic Circuits.*
CCS 2016.
[ePrint 2016/969](https://eprint.iacr.org/2016/969)
Free addition and public scaling in `(Z_m)^k`, and a unary projection of
`m − 1` ciphertexts. A fan-in-`b` symmetric gate is a projection of the sum.
`arith_garble.hpp` is this gadget. The session in `beaver.hpp` stays the
interactive one-open product.
**Used in** [Beaver triples](@ref beaver_triples) · [Arithmetic share runtime](@ref arith_runtime)
### Stacked garbling {#bib_stacked}
David Heath and Vladimir Kolesnikov.
*Stacked Garbling: Garbled Circuit Proportional to Longest Execution Path.*
CRYPTO 2020.
[ePrint 2020/973](https://eprint.iacr.org/2020/973)
One XOR-stack of the branch materials. Inactive branches are rebuilt from seeds.
**Used in** [A boolean function of a leaf](@ref yao_leaf)
### One-hot garbling {#bib_onehot}
David Heath and Vladimir Kolesnikov.
*One Hot Garbling.*
CCS 2021.
The same stack over `k` branches. The demux carries the inactive seeds.
**Used in** [A boolean function of a leaf](@ref yao_leaf)
### FLUTE {#bib_flute}
Andreas Brüggemann, Robin Hundt, Thomas Schneider, Ajith Suresh, and Hossein Yalame.
*FLUTE: Fast and Secure Lookup Table Evaluations.*
IEEE S&P 2023.
[ePrint 2023/499](https://eprint.iacr.org/2023/499)
A public LUT is a multi-fan-in inner product on ABY2.0 masked bits.
Online cost is two bits per output bit.
**Used in** [Beaver triples](@ref beaver_triples) · [Arithmetic share runtime](@ref arith_runtime)
### ABY3 {#bib_aby3}
Payman Mohassel and Peter Rindal.
*ABY3: A Mixed Protocol Framework for Machine Learning.*
CCS 2018.
[ePrint 2018/403](https://eprint.iacr.org/2018/403)
Honest-majority 3PC, RSS, probabilistic truncation, matrix triples.
**Used in** [Arithmetic share runtime](@ref arith_runtime)
### edaBits / CrypTFlow2 {#bib_edabits}
Daniel Escudero, Satrajit Ghosh, Marcel Keller, Rahul Rachuri, and Peter Scholl.
*Improved Primitives for MPC over Mixed Arithmetic-Binary Circuits.*
CRYPTO 2020.
[ePrint 2020/338](https://eprint.iacr.org/2020/338)
**Used in** [Arithmetic share runtime](@ref arith_runtime) · [A boolean function of a leaf](@ref yao_leaf)
### EzPC {#bib_ezpc}
Nishanth Chandran, Divya Gupta, Aseem Rastogi, Rahul Sharma, and Shardul Tripathi.
*EzPC: Programmable and Efficient Secure Two-Party Computation for Machine Learning.*
EuroS&P 2019.
[ePrint 2017/1109](https://eprint.iacr.org/2017/1109)
**Used in** [Arithmetic share runtime](@ref arith_runtime)
### Gilboa multiplication {#bib_gilboa}
Niv Gilboa.
*Two Party RSA Key Generation.*
CRYPTO 1999, LNCS 1666, pp. 116–129.
**Used in** [Arithmetic share runtime](@ref arith_runtime)
### Sharemind {#bib_sharemind}
Dan Bogdanov, Sven Laur, and Jan Willemson.
*Sharemind: A Framework for Fast Privacy-Preserving Computations.*
ESORICS 2008, LNCS 5283, pp. 192–206.
**Used in** [Arithmetic share runtime](@ref arith_runtime)
### Beaver triples {#bib_beaver}
@ -221,6 +335,22 @@ Offset Horner is not that piecewise-polynomial construction.
**Used in** [Grotto](@ref jet_and_ring) · [Guided tour](@ref tour_grotto)
### Wave Hello {#bib_wave}
José Reis, Mehmet Ugurbil, Sameer Wagh, Ryan Henry, and Miguel de Vega.
*Wave Hello to Privacy: Efficient Mixed-Mode MPC using Wavelet Transforms.*
PoPETs 2025(2), pp. 697–718.
<a href="https://doi.org/10.56553/popets-2025-0083">Publisher</a> ·
[ePrint 2025/013](https://eprint.iacr.org/2025/013) ·
<a href="reis-ugurbil-wagh-henry-de-vega-wave-hello-eprint-2025-013.pdf">PDF</a>
Equations (7) and (8) are the cleartext Haar and bior(5,3) lookup.
`make_haar_dwt_lut` and `make_bior53_dwt_lut` evaluate those tables.
The online phase pairs Haar with a deterministic Pika truncation and
bior(5,3) with segment parity.
**Used in** [Wavelet lookup tables](@ref dwt_luts) · [Guided tour](@ref tour_grotto)
## Protocols {#bib_sec_protocols}
### Duoram {#bib_duoram}
@ -379,3 +509,7 @@ The DPF work is a prepaid wildcard unit vector, rotated once the
address is opened.
**Used in** [MPC SUBLEQ](@ref app_subleq)
\htmlonly
<div class="tldr"><b>TL;DR.</b> The point key is ePrint 2018/707, not the longer EUROCRYPT 2015 key. Proofs and cuckoo multipoint are 2021/580. Comparisons are 2020/1392. The dealer-free opening is 2017/827, and the pads are IKNP plus the 2015/267 base OT. Three-party spines are 2024/1658. The information-theoretic table is 2023/028. Grotto is 2023/108.</div>
\endhtmlonly

View file

@ -3,13 +3,25 @@
Everything past a plain two-party point key.
The home page is the short version. Open a page for the snippet and the header.
## Keys and evaluation
- [Verifiability & authenticity](@ref verifiability)
- [Programmability](@ref programmability)
- [Comparisons & ranges](@ref comparisons)
- [Multipoint keys](@ref multipoint_keys)
- [Multiparty & 3-server](@ref multiparty)
- [Dealer-free keygen](@ref dealer_free)
- [Beaver triples](@ref beaver_triples)
- [Grotto](@ref jet_and_ring)
- [Point-programmable vector commitments](@ref ppvc_manual)
- [Application sketches](@ref applications)
## Network and MPC around the keys
These are first-class in the tree — see the overview, then dig in:
- [Network, parties, and MPC](@ref network_and_mpc) — map of the runtime stack
- [Protocol composition](@ref protocol_compose)
- [Beaver triples](@ref beaver_triples)
- [Arithmetic share runtime](@ref arith_runtime)
- [A boolean function of a leaf](@ref yao_leaf)
- [Logging, statistics, and experiments](@ref experiment_costs) — run log, CSVs, replayable seeds

View file

@ -1,5 +1,9 @@
# Comparisons & ranges {#comparisons}
\htmlonly
<div class="eli5"><b>ELI5.</b> The predicate is part of the key, not a test you run after expanding a point key. Greater-than, less-than, and equality each return one payload on the true side and another on the false side. Interval containment is one key for both endpoints. Shares of a comparison add.</div>
\endhtmlonly
A distributed comparison function returns a payload when a predicate holds
on the secret point. Shares are additive: `reconstruct` adds them.

204
doc/pages/compose.md Normal file
View file

@ -0,0 +1,204 @@
# Protocol composition {#protocol_compose}
\htmlonly
<div class="eli5"><b>ELI5.</b> Record every expand, multiply, and open as a node with a share domain. Identical work is interned. Independent opens share one RoundSink round; a dependency chain becomes successive waves. The schedule is what a hand-tuned Express or Duoram party would have written by hand.</div>
\endhtmlonly
`dpf::protocol::composer` ([compose.hpp](@ref dpf/compose.hpp)) records
multi-protocol strands on one sink. Values are tagged with a share
domain matching [secret_share.hpp](@ref dpf/secret_share.hpp):
| Domain | Meaning |
| --- | --- |
| `fss` | FSS / DPF leaf share |
| `a` | (2,2) additive |
| `b` | (2,2) subtractive |
| `rss` | (2,3) replicated |
| `y` | (3,3) additive (RSS product factor) |
Party-count changes are never implied: use `reshare` (or `rss_from_y`
for `y`→`rss`). Local casts use `as`.
## Building a schedule
```cpp
dpf::protocol::composer c(/*party=*/0);
auto seed = c.input(dpf::protocol::domain::fss, 16);
auto leaf = c.fss_point(seed, /*depth=*/8, /*slot_bytes=*/16);
auto p = c.default_plan(); // == schedule(); RoundSink uses this
// p.rounds() == p.exchange_waves() == 8 (no empty compute-only sink rounds)
```
Drive with [drive](@ref dpf::protocol::drive) or party
`util::drive_composed` / `util::drive_composed_trio`, optionally with
`drive_options` (`from_exchange_wave`, `compact_sink`, `beavers`).
Express/Sabre audits use `util::schedule_fused_audit` (or `fss_point_fused`).
The same plan lowers onto the RoundSink batch loop with
`plan_to_schedule` → [`schedule_session`](@ref dpf::protocol::schedule_session)
→ `finish_schedule`, or the one-shot `drive_via_schedule`. Each exchange wave
becomes a [`schedule_round`](@ref dpf::protocol::schedule_round) whose
`produce` runs on this thread when that instance's peer slot is ready (the
`schedule_session::drive` scan). Rounds carry an
[`edge_id`](@ref dpf/net/edge_mesh.hpp) on an [`edge_mesh`](@ref dpf/net/edge_mesh.hpp)
(star / clique / dealer), a [`receive_rule`](@ref dpf::protocol::receive_rule)
(`domain_open`, `copy_peer`, `field_sum`, `any_two`, `verify_*`, `eq_check`),
optional `branch` (skip send) and `next` (jump after an open).
[`session_host`](@ref dpf/session_host.hpp) queues micro-plans on a durable mesh.
Pad graphs live in [pad_graphs.hpp](@ref dpf/pad_graphs.hpp)
(`du_atallah_mul_graph`, `star_upload_answer_graph`); IKNP setup in
[iknp_graphs.hpp](@ref dpf/iknp_graphs.hpp); PIRsona / hushmap builders in
[mesh_apps.hpp](@ref dpf/mesh_apps.hpp). Named application plans and
`drive_star` / `submit_and_drive` live in [app_plans.hpp](@ref dpf/app_plans.hpp).
Replayable seeds and paper cost CSVs: see [Experiments](@ref experiment_costs).
In short, `dpf::experiment` / `app::measure_plan` / `app::run_measured` attach a
per-thread master seed (default random, or `replay`) and emit CSV breakdowns.
`plan::rounds()` equals `exchange_waves()` and `slot_bytes_all().size()` —
the sink and the scheduler share one count. Trailing compute-only DAG waves
still appear in `waves()` / `wave(i)` but do not allocate RoundSink rounds.
Composer-owned ABY sessions default to
`beavers::schedule_objective::rounds` so online sign×polynomial stays
one round. Dealer benches that want Appendix-E peels keep `prep`.
Independent circuits use `aby_lane(i)`. Live δ openings use
`drive_options::beavers` (`util::u64_beaver_host` wraps
`party_batch_stepper`).
## Hand-schedule shapes the API covers
| Shape | Call | What it saves |
| --- | --- | --- |
| L‖R PRG stretch | `expand_pair` / default `level_walk` | Half the AES vs separate child expands |
| Shared-seed fan (Grotto LUT) | `fan` of `fss_cmp` on one seed | One expand per level, not N×depth |
| Several piecewise LUTs | `schedule_lut_union` | One `fss_cmp`; the union's prefix walk is local |
| Express / Sabre audit | `fss_point_fused` / `schedule_fused_audit` | Sketch in last CW — no +1 exchange wave |
| BGI Remark 3.4 early-stop | `fss_point_early_stop` | Drop ν interactive CW rounds |
| Poplar / idpf prefix checkpoints | `level_walk_prefixes` | Prefix share after each CW, no extra rounds |
| Adaptive idpf_agg | `step` → drive tail → `retain` → `step`… | One packed L‖R open per depth; child bit never on the wire |
| DCF `block_width` | `level_walk_sized` | Per-level CW slot bytes |
| Doerner–Shelat keygen | `level_walk_ds` / `level_walk_ds_sized` | blind‖CW‖advice‖AND + OH AND-layers |
| Keyword PIR / PSI buckets | `multipoint_fan` / `exchange_pack` | Bucket CWs pack; answers one open |
| RSS mul + neighbor refresh | `rss_product_replicated` / `rss_from_y` | One y-exchange, not a reconstructing open |
| FSS leaf → ABY scale | `aby_product` after `fss_point` | Leaf stays on the beaver barrier critical path |
| Multi-lane ABY | `aby_lane(i)` / `aby_product(..., lane)` | Independent barriers, shared wave |
| Round-aware Beaver | `composer::aby<Ring>()` | `schedule_objective::rounds` |
| Prepaid expand / rotate | `defer_expand` / `rotate_share` | Zero online FSS rounds (SUBLEQ offline) |
| Duoram leaf_later | `leaf_later_walk` / `apply_leaf_correction` | Path CWs only; leaf apply is local |
| RSS column, hidden order | `shuffle_hidden` | Three `shuffle_send` waves. `aux` is the left-out party: 2, then 0, then 1 |
Adaptive prefixes: `step_adaptive_prefix` schedules only the next packed
open; drive with `from_exchange_wave = exchanges_flushed` and a sink sized
by `plan::slot_bytes_from` (`compact_sink = true`); then
`retain_adaptive_prefix` adds one local tip. Never unroll a full-depth
adaptive walk into one static schedule.
3PC RSS: `rss_from_y` schedules a `domain::y` exchange (neighbor receive).
`util::drive_composed_trio` splits peer maps per exchange — `y` on the
neighbor ring, `dealer_deliver` from p2, everything else on p0↔p1.
Cross-party `reshare` refuses a reconstructing open; use
`reshare_with_mask`. Opens use domain algebra (`a`/`fss` sum, `b`
subtractive, `y` copy, optional `field_open::fp61`). Authenticated Beaver
sessions size barriers as `auth_opening` and drive through
`u64_auth_beaver_host`. `exchange_fuse` packs extra payloads into one open.
`schedule_cuckoo_probes` turns occupied bucket ids into `multipoint_fan`.
## Client and servers {#compose_client}
Keyword PIR and both three-server PIRs are a star, not a correction-word
walk. `client_servers(servers, query_bytes, answer_bytes)` records two
waves and nothing between the servers:
1. Upload. The slot is `servers * query_bytes` (every key in one round).
2. Answer. The slot is `servers * answer_bytes`. It depends on the upload,
so it cannot share that wave.
`drive` copies the peer's message into the exchange node. It does not add
the shares. A two-party sink stands in for one client and one server; the
slot width is still the full parallel payload, which is what
`dpf::app::exercise` reports as `bytes`.
```cpp
auto q = c.client_servers(/*servers=*/2, /*query_bytes=*/256, /*answer_bytes=*/4);
auto p = c.default_plan();
// p.rounds() == 2
// p.slot_bytes(0) == 512, p.slot_bytes(1) == 8
```
## Parking and the fleet {#compose_fleet}
`drive` used to spin until the peer flushed, then throw. A step that
simply takes a long time looks like that failure, and a pool that always
resumes the side already waiting on a receive never runs the peer who
could unblock it.
`drive_options::park_if_waiting` returns instead. `drive_cursor` remembers
the wave, the exchange index, and whether the submit already happened.
The next `drive` with that cursor receives if the peer has caught up, or
parks again. `one_exchange` stops after a single completed round so a
scheduler can interleave other instances.
`dpf::app::run_fleet(composer, instances, chaos_seed)` is that scheduler.
Each instance is a pair of parties on an in-process sink. Workers prefer
a side that still has a submit to do over a side parked on a receive, and
among those they prefer the one further behind. Delays are per side and
per step. They do not line up, which is the case that makes round-robin
and "always resume the waiter" stall.
`dpf::app::run(name, composer, expect_rounds)` is the one-shot experiment
used by `examples/applications/`. It drives both parties and prints
```
name rounds=R bytes=B
```
`bytes` is the sum of one lane's exchange slots.
Opcodes from `beaver_delta` (300) through `beaver_delta + 1023` are δ
barriers. `user_base` (1000) sits inside that window, so a hand-rolled
kernel opcode in that range is skipped as a beaver. Use 5000 and up for
kernels `drive` must run or reject.
## What eval_full reuses
`eval_full(key, buffer)` keeps a thread-local workspace for that key type.
The same key evaluated again does not rebuild the interior. A different
key still does. Leaves of one block are stretched eight at a time
(`eval_x8`). Pass your own memoizer when two evaluations on one thread
must not share that cache.
DS OH: `level_walk_ds(..., oh=true)` emits
`net::ds_oh_exchanges_per_level` (80) AND-layer opens per level. Size sinks
with `net::compose_ds_slot_bytes` (schedule-exact) or the conservative
`net::ds_walk_slot_bytes` for hand `dist_ds` paths.
\include{cpp} protocol/compose_schedule.cpp
## Gap analysis {#compose_gaps}
| Item | Status |
| --- | --- |
| Point-key stretch | Builtin `fss_expand_pair` runs `prg::aes128`; `fss_step+1` XORs the opened correction onto the children |
| Setup plus walk | `iknp_setup_graph` / `plan_with_pad_setup` — pad frames are `schedule_round`s |
| Edge mesh | `edge_mesh` + `make_memory_star` / `make_memory_clique`; `drive_via_schedule(plan, mesh, …)` |
| Receive rules | `field_sum`, `any_two`, `verify_sketch` / `verify_proof`, `eq_check` in `apply_peer_slot` |
| Jump / host | `schedule_round::next` + `session_host` for SUBLEQ / idpf_agg / hushmap ops |
| PIRsona / hushmap | `pirsona_bitmore_fetch`, `pirsona_gd_update_graph`, `hushmap_add_schedule` |
| 2PC → RSS | `reshare` casts to a `y` share (p2 holds 0) and `rss_from_y`. `reshare_fresh` adds a dealer-sampled zero mask |
| Auth openings | `auth_beaver_host<Ring>` / `u64_auth_beaver_host`. `party_auth_batch_stepper::apply_peer` rejects a bad tag |
| One DS sink | `ds_walk_slot_bytes` is the envelope of the hand walk and `compose_ds_slot_bytes` (5 peer opens per level + OH + mux) |
| Wide opens | Every multiple of 8 bytes is a lane add/sub (`field_open::fp61` for the Mersenne field) |
| Fuse offset | `exchange_fuse` stores the segment offset in `aux` (`uint32_t`) |
Still outside this scheduler: the cuckoo PRP, OPRF evaluation, the full IKNP wire body inside each pad `produce` (frames and count are in the schedule; `iknp::sample` still owns the crypto), and key-specific control-bit / verifiable / Doerner–Shelat advice logic. Those last ones override the builtin kernels. Trio multi-edge `drive_via_schedule` still needs `edge_sinks` wired from the party helper.
### Closed earlier
Incremental drive, domain-correct `a`/`b`/`y` opens, split y/2PC peer maps, OH AND-layers, `level_walk_ds_sized`, `default_plan`, staged adaptive retain, multipoint fan, multi-lane ABY, `defer_expand` / `leaf_later_walk`, Express trailer fuse.
**Go deeper:** [compose.hpp](@ref dpf/compose.hpp),
[beaver.hpp](@ref dpf/beaver.hpp),
[sink_exchange.hpp](@ref dpf/net/sink_exchange.hpp),
[Application sketches](@ref applications),
[guided tour](@ref tour_party).

View file

@ -1,5 +1,9 @@
# Dealer-free keygen {#dealer_free}
\htmlonly
<div class="eli5"><b>ELI5.</b> The index is already split. Doerner–Shelat turns those shares into a reusable key by opening one masked correction per level. geneval stops after a single public query and returns answer shares, not a key. IKNP is how the pads are sampled when no dealer supplies them.</div>
\endhtmlonly
The two parties already hold shares of the secret index.
Nobody sends `alpha` to a dealer.

View file

@ -75,6 +75,10 @@ mockup [I-DPF max and k-th](@ref app_idpf_agg).
## Assigning a wildcard leaf {#wildcard_assign}
\htmlonly
<div class="eli5"><b>ELI5.</b> The blank leaf is a Beaver slot from keygen. Filling it in rewrites the correction word: the parties exchange one blinded share of the new payload and both apply the same patch. assign_cmp rewrites the n comparison words locally and sends nothing. An updatable leaf is the same patch later, O(λ) and independent of the depth.</div>
\endhtmlonly
An output wildcard is a placeholder for a payload filled after keygen.
The type and the `dpf::wildcards` names are on
[Output types](@ref output_types). Evaluation of an unassigned slot throws
@ -149,6 +153,10 @@ either party's type accepts both parties. Name that type with
## Memoizers {#memoizers}
\htmlonly
<div class="eli5"><b>ELI5.</b> The first walk stores interior nodes. The next query starts from the deepest stored node that still lies on its path, instead of from the root. An interval memoizer stores the nodes that cover a range; a sequence memoizer stores the nodes along a sorted list.</div>
\endhtmlonly
## Path memoizers {#path_memoizers}
`eval_point` walks one root-to-leaf path. `make_basic_path_memoizer<Key>()`
@ -331,6 +339,10 @@ size both for that domain.
## Deferred input evaluation {#defer_eval}
\htmlonly
<div class="eli5"><b>ELI5.</b> The PRG expand runs once, into a full-domain buffer, before the index is known. When the offset opens, get() rotates that buffer. The tree is not expanded again.</div>
\endhtmlonly
Additive input blinding evaluates in tree coordinates `x ↦ x + δ`, where
`δ` is reconstructed by `assign_wildcard_input` into `offset_x`. Eager
`eval_interval` folds that map into the traversed range and throws if the
@ -364,6 +376,10 @@ Buffers must outlive both.
## dpf::eval_inner_product {#eval_inner_product}
\htmlonly
<div class="eli5"><b>ELI5.</b> The walk is the same as an interval or full-domain eval, but each leaf is multiplied by a public weight and added into one accumulator. The expanded vector is not stored. A sequence inner product does that only at the listed points.</div>
\endhtmlonly
`eval_inner_product` multiply-accumulates DPF shares against another vector
during the walk. It does not write the output vector.
@ -469,6 +485,10 @@ on that list: at most `O(n m)` expands and `m` output slots.
## Buffered PRG {#buffered_prg}
\htmlonly
<div class="eli5"><b>ELI5.</b> One AES expand produces more blocks than a single tree node needs. The buffered PRG keeps the leftover blocks and serves the next nodes from them, so a wide walk makes fewer expands. The keys do not change.</div>
\endhtmlonly
`dpf::randomness::buffered_prg<PRG, Ts...>` (alias
`dpf::randomness::aes_buffered_prg<Ts...>`) is a forward cursor with one
PRG stream per value type. `get<I>()` and `fill<I>(out, n)` consume the
@ -496,6 +516,10 @@ element. `fill_values` / `fill_masks` of `q` elements are `Θ(q)`.
## Three-party (2,3) DPF {#dpf3}
\htmlonly
<div class="eli5"><b>ELI5.</b> Each evaluator key is two VDPF+ spines. eval_point walks both, Θ(n) expands, then scales into fp61. Opening two or three as_share values is a constant amount of field arithmetic. An updatable rewrite patches four leaves and refreshes an offset, independent of n.</div>
\endhtmlonly
`make_dpf3(α, β)` builds three evaluator keys after Guy Zyskind, Avishay Yanai, and Alex "Sandy" Pentland, [ePrint 2024/1658](@ref bib_dpf3), Figure 3:
two VDPF+ spines plus Shamir embedding in `fp61`. `eval_point` returns a
field share; open with `dpf::reconstruct` on any two (or all three)
@ -525,6 +549,10 @@ two-party comparison, interval, or cuckoo packing, on top of those spines.
## Information-theoretic 3-server DPF {#it_dpf3}
\htmlonly
<div class="eli5"><b>ELI5.</b> There is no PRG tree. For this domain each key is an additive share of a 256-word table. The three shares sum to beta at alpha and to zero elsewhere. A PIR answer is three inner products with the database; those three dots sum to the record.</div>
\endhtmlonly
`make_it_dpf3(α, β)` ([ePrint 2023/028](@ref bib_itdpf)) is a different object from
`make_dpf3`. Each of three parties holds an additive share of the
characteristic vector on `{0..255}`; the **sum** of all three
@ -573,7 +601,13 @@ claimed speedup over dealer keygen or over Half-Tree §5.2.
<div class="tabbed">
- <b class="tab-title">eval_dpf3_point.cpp</b> \include{cpp} evaluation/eval_dpf3_point.cpp
- <b class="tab-title">eval_dpf3_doerner_shelat.cpp</b> \include{cpp} evaluation/eval_dpf3_doerner_shelat.cpp
- <b class="tab-title">eval_dpf3_cmp_ic.cpp</b> \include{cpp} evaluation/eval_dpf3_cmp_ic.cpp
</div>
\htmlonly
<div class="tldr"><b>TL;DR.</b> eval_point is one path. An interval costs the path plus the length of the range. A full domain costs the size of the domain; an inner product does that walk and keeps only the accumulator. Wildcard assign and an updatable rewrite patch the leaf and do not depend on the depth. Three-party eval is two walks and a short field open. The information-theoretic key is a 256-word share, not a walk.</div>
\endhtmlonly

192
doc/pages/experiment.md Normal file
View file

@ -0,0 +1,192 @@
# Logging, statistics, and experiments {#experiment_costs}
Paper and production runs need three things the bare key API does not:
**leveled run logs**, **granular cost statistics**, and **replayable coins**.
All three live in the same measurement stack —
[`dpf/log.hpp`](@ref dpf/log.hpp), [`dpf/run_log.hpp`](@ref dpf/run_log.hpp),
[`dpf::experiment`](@ref dpf/experiment.hpp), and the
[`prg::count`](@ref dpf/prg_count.hpp) / [`thread_work`](@ref dpf/thread_work.hpp)
counters.
Uninstrumented code stays quiet and keeps reading `/dev/urandom`. Logging and
experiments are opt-in. `run_measured`, the battery, and `party_node` call
[`app::start_logging`](@ref dpf/run_log.hpp) and install an experiment covering
the thread that constructs it. `run_parties`, `measure_plan`, and the battery
give party `i` its own stream, `ex.derive_party(i)`, whose master is SHA-256 of
the experiment's master and `i`, and they use it for every trial, warmup
included. A kernel handed to a compute pool draws from its party's stream.
Replaying the master replays every party, as long as each party makes the same
draws in the same order. The party masters are noted as `p0/master`,
`p1/master`, and so on.
## Run log {#run_log}
[`dpf/log.hpp`](@ref dpf/log.hpp) writes one `key=value` line per event to
stderr, syslog, or a file. Nothing is written until `log::configure` runs
(normally through `app::start_logging`), so library code can log freely and a
test that never configures the log stays quiet. Every line starts with
`ts`, `lvl`, `inv` (invocation id), `pid`, `tid`, then `role=<party>` when the
thread has one, then `ev=<event>` and the event's fields. Seed bytes go through
`record::seed` as hex, as a SHA-256 fingerprint, or not at all.
[`app::start_logging`](@ref dpf/run_log.hpp) turns it on and writes the
provenance banner first, at `info`:
| Event | Contents |
| --- | --- |
| `ev=start` | program, cwd, UTC/local start, invocation id shared with `runs.csv` |
| `ev=build` | git rev (`LIBDPF_GIT_REV`), compiler, `NDEBUG`, ISA, sanitizers |
| `ev=host` | hostname, kernel, CPU model, affinity, governor, turbo, memory, load; `ev=if` per interface |
| `ev=argv` | shell-quoted command line |
| `ev=env` | `DPF_*` / `LIBDPF_*` and perf-relevant vars (`LD_PRELOAD`, …) |
| `ev=config` | every `run_config` setting |
After that come seeds with their source, listeners and links (endpoints, peer
authentication, encryption, applied socket options, kernel RTT), each party's
plan and costs, trial statistics, and failures. `ev=end` at exit gives elapsed
time. The settings are `run_config` keys:
```
DPF_LOG_LEVEL=debug DPF_LOG=file:/tmp/run.log,stderr DPF_LOG_SEEDS=hash
--log_level=debug --log=syslog --log_seeds=off
```
Levels are `silent`, `error`, `warning`, `info` (the default), `debug`, and
`trace`. With `log_seeds=full` the log holds every master, so anyone who has
the file can regenerate that party's randomness. `hash` prints a SHA-256
fingerprint instead.
## Replayable seeds
```cpp
dpf::experiment ex("keyword_pir"); // fresh 32-byte master
auto master = ex.seed(); // record for the paper
auto x = dpf::uniform_sample<std::uint64_t>();
auto again = dpf::experiment::replay("keyword_pir", master);
assert(dpf::uniform_sample<std::uint64_t>() == x);
```
While the experiment is installed, every `uniform_fill` / `uniform_sample`
draw on that thread (DPF roots, Beaver default sampler, pads, Shamir, IKNP,
field rejection) comes from AES-CTR of the master. The master itself is always
drawn from OS entropy, even when nested under another experiment.
Constructors that already own a seed note it automatically:
| Type | Seed name |
| --- | --- |
| (master) | `master` |
| `beavers::oracle` | `beavers::oracle` (+ `lane_table`) |
| `buffered_prg` | `buffered_prg` |
| `prg_pad_rng` | `prg_pad_rng` |
| `pseudorandom_root_sampler` | `pseudorandom_root_sampler` |
Call `ex.note_seed("label", value)` for anything else. Replay is
**order-sensitive**: the same party must make the same `uniform_sample` calls.
Indexed `beavers::oracle(seed)` stays the seekable Beaver path; its seed still
appears in the report when constructed under the experiment.
## Measuring a compose plan {#statistics}
```cpp
auto plan = dpf::protocol::fss_point_plan(0);
auto ex = dpf::app::measure_plan("fss_point", plan);
ex.write_csv("/tmp/run1");
```
Or from an application sketch:
```cpp
// Prints rounds, live bytes, wall/CPU, PRG evals, random bytes, seed hex.
// Set DPF_EXPERIMENT_DIR=/tmp/run1 to also emit CSVs.
dpf::app::run_measured("keyword_pir",
dpf::protocol::keyword_pir_compose_plan(0, depth), 2);
```
What is recorded:
| Metric | Source |
| --- | --- |
| Interactive rounds / DAG depth | `plan::exchange_waves()` / `plan::waves()` |
| Critical path | back-walk of max-`wave_of` inputs |
| Schedule bytes per edge | `slot_bytes` + `wave_channel` |
| Bytes out/in per round | the round's send and receive slots, on its own channel |
| Wall / CPU | `steady_clock` / party 0's thread CPU plus its pool kernels' CPU |
| Symmetric-key blocks | `prg::count(purpose, primitive)` per thread |
| Random bytes | `uniform_fill` TLS counter |
| Seeds | master + every `note_experiment_seed` |
Symmetric-key blocks are counted by purpose and primitive. The purposes are
`expand` (tree expansion, leaf conversion, label and column PRGs), `hash`
(IKNP row hashes, garbled-gate hashes), and `harness` (the experiment's own
seed stream). The primitives are AES-128, AES-256, ChaCha, and LowMC.
`prg_evals` is `expand` plus `hash` over every primitive. Harness blocks are
never part of it. The counters are per thread; a kernel a party hands to a
compute pool (`compute_threads > 0`) is charged to that party when it returns
([`dpf/thread_work.hpp`](@ref dpf/thread_work.hpp)).
## CSV layout
`write_csv(dir)` appends. Every row ends with the id of the process that
wrote it (`invocation`), so rows from different runs into one directory stay
apart. A file whose header differs from this layout is renamed to
`<name>.before-<UTC>.csv` before new rows go in.
- `summary.csv` — one row per run, including `master_seed` hex. `wall_ns`,
`cpu_ns`, and the counters come from the last (instrumented) trial;
`median_ns` and `slowest_median_ns` are medians over the timed trials for
party 0 and for the slowest party.
- `rounds.csv` — per-round deltas and edge name
- `edges.csv` — protocol totals per peer / rss_next / dealer
- `seeds.csv` — every noted seed as hex
- `critical_path.csv` — node id, wave, opcode, effect
- `config.csv` — every `run_config` setting
- `trials.csv` — every party's wall time for each timed trial
- `wire.csv` — party 0's link counters, headers included
- `sym.csv` — symmetric-key blocks by purpose and primitive
- `runs.csv` — the invocation id that ties rows to the run log, and whether
the master was fresh, provided, or derived
Before any party starts its clock, all parties wait at a start gate until
every one has finished setup.
## The battery
The harness is `party_bench` and `profile_party`. Both spawn `p0`, `p1`, and
`p2` and dial [`net::trio`](@ref dpf/net/trio.hpp), the library's localhost
mesh (`p0-p1`, `p0-p2`, `p1-p2`). A flow's bytes and rounds are the frames on
that mesh. The repeat barrier is not included.
```
party_bench --tag bench --repeat 3 --warmup 1
party_bench --case arith_mul_p11 --repeat 5
profile_party --suite gadget --repeat 3 --warmup 1
```
`--tag bench` is the default list. The gadget rows are tagged `bench` as well,
so they sit on that list. `profile_party --suite gadget` runs only those rows.
`profile_party --suite all` appends them after the core and extreme suites.
A secret index stays a DPF. The gadget rows time what you do after a key, or
on shares the parties already hold:
- `arith_proj_*`, `arith_mul_*`, `arith_thresh_*`, `arith_chain_mul4` —
Ball–Malkin–Rosulek word garbling. Party 0 sends the evaluator view on the
p0–p1 link. Party 1 evaluates that view and opens.
- `yao_if_*`, `yao_onehot_*` — stacked and one-hot garbling on the garbler.
The active branch's tables and the evaluator's labels go across the p0–p1
channel, and party 1 evaluates those bytes. Base OT stays on the `iknp` tag.
- `flute_d*` — FLUTE. Party 2 deals the mask shares. Parties 0 and 1 exchange
the online bits and open.
- `shuffle_n*` — three hidden-shuffle passes. Each pass's array is sent on
the ring and consumed as the next party's inbound. Party 0 opens the sum.
Sizes are the modulus, the branch width, the table width, and the column
length. The manual pages are [a small word](@ref arith_garble_word),
[a public table](@ref flute_lut), [a secret branch](@ref yao_stack), and
[an array of shares](@ref share_shuffle).
See also [Network, parties, and MPC](@ref network_and_mpc),
[Protocol composition](@ref protocol_compose), and the
[application mockups](@ref applications).

View file

@ -19,7 +19,12 @@ A *distributed point function* (DPF) is a way to share it with short keys.
`libdpf++` builds those keys and evaluates them quickly in C++17.
People use DPFs for private lookup (PIR), multi-party computation (MPC),
and other privacy tools. See also the [ideal functionalities](@ref ideal_functionalities)
and other privacy tools. This tree also ships the **network and MPC stack**
those protocols need: TLS party sessions, RoundSink rounds, Beaver / Yao /
arithmetic shares on leaf values, composition, leveled run logs, and
paper-cost statistics. That map is [Network, parties, and MPC](@ref network_and_mpc);
logging and CSVs are [Logging, statistics, and experiments](@ref experiment_costs).
See also the [ideal functionalities](@ref ideal_functionalities)
for what each protocol is allowed to learn.
## Prior work {#tour_prior}
@ -87,8 +92,8 @@ auto [k0, k1] = dpf::make_dpf(X{1000}, std::uint64_t{1});
Width literals sit next to those types: `100_u12` (`modint`), `7_x12`
(`xint`), `dpf::literals::operator""_bitstring`, and `1.5_fixed16` through
`_fixed64`. `dpf::bit`, `dpf::twobit`, and `dpf::nyble` are outputs, not
domains. See [Input types](@ref input_types).
`_fixed64`. `dpf::bit`, `dpf::twobit`, `dpf::nyble`, and `dpf::gf2` through
`dpf::gf264` are outputs, not domains. See [Input types](@ref input_types).
## Outputs: what sits at that point {#tour_outputs}
@ -102,6 +107,7 @@ Many outputs can share one leaf when they fit.
| `dpf::bit` | One XOR bit, packed | [bit.hpp](@ref dpf/bit.hpp) |
| `dpf::twobit` | Z/4Z, packed 2-bit lanes | [twobit.hpp](@ref dpf/twobit.hpp) |
| `dpf::nyble` | Z/16Z, packed nibbles | [nyble.hpp](@ref dpf/nyble.hpp) |
| `gf2` … `gf264` | GF(2^k), XOR add, field multiply | [gf2.hpp](@ref dpf/gf2.hpp) |
| `dpf::bitstring` | XOR string | [bitstring.hpp](@ref dpf/bitstring.hpp) |
| `grotto::fixedpoint` | Fixed-point raw word | [fixedpoint.hpp](@ref grotto/fixedpoint.hpp) |
| `dpf::vec<T, N>` | `N` lanes, no carry between them | [vec.hpp](@ref dpf/vec.hpp) |
@ -109,7 +115,7 @@ Many outputs can share one leaf when they fit.
| `field64` / `field128` | Prime fields | [field64.hpp](@ref dpf/field64.hpp) |
| `fp61` | Field for 3-party DPFs | [fp61.hpp](@ref dpf/fp61.hpp) |
| `p256` / `p256_scalar` | Curve and order | [p256.hpp](@ref dpf/p256.hpp) |
| Typed shares | (2,2) additive and subtractive, (3,3) additive, (2,3) replicated | [secret_share.hpp](@ref dpf/secret_share.hpp) |
| Typed shares | (2,2) additive and subtractive, (3,3) additive, (2,3) replicated, (K,N) Shamir | [secret_share.hpp](@ref dpf/secret_share.hpp), [shamir.hpp](@ref dpf/shamir.hpp) |
```cpp
auto [k0, k1] = dpf::make_dpf(
@ -118,6 +124,12 @@ auto [k0, k1] = dpf::make_dpf(
// assign the payload later; eval before assign throws
```
Shamir shares are `shamir::share<T, Party, K, N>`. Any `K` of `N` open the
constant term. `(2,3)` is `shamir_share`. The fields are `fp61` and `gf2n`.
For `gf2n`, `N` must be less than `2^k`. `make_dpf3` uses the `(2,3)` case
on `fp61`. `examples/mwe/shamir.cpp` deals a `(3,5)` secret in `fp61` and a
`(2,3)` secret in `gf28`.
Packed output literals are `1_bit`, `2_twobit`, and `10_nyble`.
`dpf::vec<T, N>` is `N` lanes of an ordinary output with no carry between
lanes; the construction is on [Output types](@ref output_types).
@ -216,6 +228,10 @@ or zip two parties' buffers.
## Comparisons and ranges {#tour_dcf}
\htmlonly
<div class="eli5"><b>ELI5.</b> A comparison key is not a single spike. It returns the true payload on one side of the secret point and the false payload on the other, and the shares add instead of subtract. An interval key packs the two endpoint comparisons into one key. idcf repeats a correction at every depth; cmp_prefix stops after L bits.</div>
\endhtmlonly
A *distributed comparison function* (DCF) returns a payload when a predicate
holds on the secret point. The four predicates are `dpf::lt`, `dpf::leq`,
`dpf::gt`, and `dpf::geq`. Each takes the true payload and an optional false
@ -284,6 +300,10 @@ ideal figures [F_DCF](@ref dcf.hpp), [F_BDCF](@ref blocked_dcf.hpp), [F_IC](@ref
## Verifiable and extractable keys {#tour_vdpf}
\htmlonly
<div class="eli5"><b>ELI5.</b> verifiable carries an extra seed on each correction word and folds it into one proof token. Equal tokens across parties mean the seeds were the honest ones. extractable is a separate weight-1 sketch in fp61: any second hot point fails it. output_mac authenticates the opened leaf, not the path.</div>
\endhtmlonly
Pass `dpf::verifiable{}` or `dpf::extractable{}` as an extra `make_dpf`
argument. `verifiable` follows de Castro and Polychroniadou, EUROCRYPT
2022 ([ePrint 2021/580](@ref bib_vdpf)): one extra correction seed per level (their hash
@ -308,6 +328,10 @@ auto [e0, e1] = dpf::make_dpf(std::uint8_t{42}, std::uint64_t{7},
## Many points at once {#tour_multipoint}
\htmlonly
<div class="eli5"><b>ELI5.</b> The m secret points are placed in cuckoo buckets with three hashes, one ordinary point key per bucket, so the key grows with m and not with the domain. Evaluation probes the three buckets that could hold the query. One verifiable tag is a single proof for the whole set.</div>
\endhtmlonly
`make_multipoint(alphas, betas)` packs many points into cuckoo buckets,
following de Castro and Polychroniadou, EUROCRYPT 2022, §4 (ePrint
2021/580): `κ = 3` hashes, one point key per bucket.
@ -326,6 +350,10 @@ auto [k0, k1] = dpf::make_multipoint(alphas, betas);
## A vector with one programmable coordinate {#tour_ppvc}
\htmlonly
<div class="eli5"><b>ELI5.</b> Commit binds both roots of each aligned 1-bit DPF pair under a Naor string, before anyone chooses the coordinate. Open reveals one side of each pair, which writes that coordinate or the sum of the vector onto a public index. verify checks the opened roots against the committed strings.</div>
\endhtmlonly
`dpf::ppvc` commits to a vector in `(Z/2^s Z)^n` before the hidden
coordinate is chosen. The commitment binds both roots of `s` aligned
1-bit DPF pairs. Opening one side of each pair writes that coordinate,
@ -343,6 +371,10 @@ const auto op = scheme::open(st, 0, 0x5a, std::uint8_t{40});
## Tree shapes: classic and Half-Tree {#tour_trees}
\htmlonly
<div class="eli5"><b>ELI5.</b> The default key is the CCS 2016 layout: one correction word per level, with the last few levels packed into the leaf when the output group is small. Half-Tree keeps that key shape and changes the expand to H(s) and H(s) XOR s, which is about half as many permutation calls. Section 5.2 of that paper is a different thing: two-party keygen in the COT/OLE hybrid, which this generator does not use.</div>
\endhtmlonly
The default interior PRG walks a Boyle–Gilboa–Ishai tree (CCS 2016,
full version [ePrint 2018/707](@ref bib_fss2018)).
Select `prg::aes128_ccr` as the *interior* PRG to use Half-Tree expands
@ -358,6 +390,10 @@ about `4n`, and `1.5N` calls for a full-domain evaluation versus `2N`.
## Two-party keygen without a dealer: Doerner–Shelat {#tour_ds}
\htmlonly
<div class="eli5"><b>ELI5.</b> Each party holds a share of the index and neither sends the index. Level by level they open a masked correction word. The key that comes out is the same object a dealer would have built with make_dpf. geneval uses the same opening on one public query and does not return a reusable key.</div>
\endhtmlonly
Two parties hold XOR (or additive) shares of `alpha`.
A third party deals pads and learns nothing.
The result matches what an honest `make_dpf` would emit.
@ -410,6 +446,10 @@ a round of its own beyond that split.
### Two-party socket walk without p2 (IKNP) {#tour_iknp}
\htmlonly
<div class="eli5"><b>ELI5.</b> IKNP turns a few slow base oblivious transfers into a long tape of correlated pads. Those pads stand in for the dealer in the Doerner–Shelat walk. The base OTs are Chou–Orlandi on P-256. When a level must stay hidden, the hash is the Boyar–Peralta AES S-box: 32 ANDs per block, which is why an oblivious tape is much longer than a reveal tape.</div>
\endhtmlonly
When there is no pad dealer, p0 and p1 sample the same Doerner–Shelat
pad correlations with semi-honest OT extension after Yuval Ishai, Joe
Kilian, Kobbi Nissim, and Erez Petrank, CRYPTO 2003 (`dpf::iknp::sample`),
@ -482,6 +522,10 @@ party mesh [trio.hpp](@ref dpf/net/trio.hpp).
## Multiplication and circuits: Beaver {#tour_beaver}
\htmlonly
<div class="eli5"><b>ELI5.</b> Preprocessing gives every wire a blind. To multiply, the parties open the two inputs masked by those blinds, then fix the product with a local correction. A later gate reuses blinds it already holds and only samples new monomials. A MAC is a second share of the same width, checked in batch.</div>
\endhtmlonly
ABY2.0-style sessions open masked wires once, following Patra, Schneider,
Suresh, and Yalame, USENIX Security 2021 (full version [ePrint 2020/1225](@ref bib_aby2)).
A fresh triple follows Donald Beaver, [CRYPTO 1991](@ref bib_beaver), which reconstructs
@ -497,12 +541,75 @@ A later round reuses blinds it already holds and samples only the new
monomials. The dealer keeps each full blind; the parties receive the
additive splits.
A word that has already left the key can be added and projected in one shot
([a small word](@ref arith_garble_word)). A public table on a short masked
index is [FLUTE](@ref flute_lut). A secret point in a public table stays a
DPF. The session above is still the interactive product.
**Go deeper:** [beaver.hpp](@ref dpf/beaver.hpp),
[F_Beaver](@ref beaver.hpp), [F_BeaverAuth](@ref beaver.hpp),
[constrained_cmp.hpp](@ref dpf/constrained_cmp.hpp) for `F_CCMP`.
## A column of shares {#tour_shuffle}
The secret position in this library is a key. One cell of a shared array is
a unit DPF, a public rotate, and a dot product
([Duoram](@ref app_duoram)).
A hidden shuffle is what you do when you already hold every row and you are
about to open the column. Three passes, from the pairwise seeds `k01`,
`k12`, and `k20`, leave one party out of each permutation. The opened order
is not the stored order, and no single party can recompute it.
`shuffle_hidden_pass` is one party's step. `shuffle_party` is the other
helper: one permutation from `k01`, which every holder of that seed can
recompute.
The shuffle does not look up a cell, and it does not sort. Those stay a DPF
and a DCF.
**Go deeper:** [An array of shares](@ref share_shuffle),
[shuffle.hpp](@ref dpf/shuffle.hpp).
## When the leaf is a circuit {#tour_yao}
\htmlonly
<div class="eli5"><b>ELI5.</b> Eval already gave you a share of the leaf. If the next step is a bit circuit the key does not contain, split that share into XOR bits, garble the circuit, and share the answer back as a leaf.</div>
\endhtmlonly
A point leaf is subtractive, so the split is `b2y` and the return is `y2b`.
A comparison leaf is additive, so the split is `a2y`. An `fss_share` opens
like a point leaf. Parties 0 and 1 garbling a replicated leaf use `rss2y`;
party 2 does not send. The bits are least-significant first. Both shares
are arguments to the conversion. The masked `x - r` that A2B opens is uniform.
The netlist is XOR, AND, XNOR, and NOT. Party 0 garbles with half-gates
([ePrint 2014/756](@ref bib_halfgates)): 32 bytes per AND, one message, XOR
shares out. The block this is for is `yao::aes_mmo`, the same zero-key
Matyas–Meyer–Oseas block `prg::aes128::eval` uses on a tree expand, 5120
ANDs. AES-128 under a shared key is the other packaged netlist, 6400 ANDs.
The Doerner–Shelat oblivious hash still evaluates that S-box as GMW layers
in [the dealer-free section](@ref tour_ds). `aes_mmo` is one of those blocks
in constant rounds, for a leaf or a seed you already hold as bits.
A comparison, an interval, a public-offset polynomial, and a product of
two leaves do not come here. Those are a key, Grotto, or one Beaver open.
`cost_pass` does not grow a Yao strategy for them.
A secret if/else or a menu of blocks on those bits is stacked garbling
([a secret branch](@ref yao_stack)). The transmitted rows follow the heavier
block. The leaf is still split in and shared back out. A comparison stays
on the key.
**Go deeper:** [A boolean function of a leaf](@ref yao_leaf),
[yao.hpp](@ref dpf/yao.hpp), [yao_share.hpp](@ref dpf/yao_share.hpp),
[F_Yao](@ref yao.hpp), [F_YaoShare](@ref yao_share.hpp).
## Three evaluators {#tour_dpf3}
\htmlonly
<div class="eli5"><b>ELI5.</b> make_dpf3 gives each of three parties a pair of two-party keys, and the payload is a Shamir share in fp61, so any two evaluation shares open the value and one share is independent of it. make_it_dpf3 is not that object: each party holds an additive share of a length-256 table, and the three shares sum to the point function.</div>
\endhtmlonly
`(2,3)` point keys follow Zyskind, Yanai, and Pentland, [ePrint 2024/1658](@ref bib_dpf3),
Figure 3: each evaluator key is a pair of `(2,2)`-VDPF+ keys.
Each key is a Shamir share in `fp61`.
@ -554,6 +661,10 @@ three `eval_it_dpf3` values is the point function. See
## Grotto: math after a public offset {#tour_grotto}
\htmlonly
<div class="eli5"><b>ELI5.</b> The parties open the public distance eta = x − r. A polynomial, a binomial jet, a carry, or a table lookup is then a correction of shares they already have. That correction does not walk another DPF.</div>
\endhtmlonly
Open `eta = x - r`. Then cheap public corrections give rich functions of `x`
without another tree walk.
@ -567,7 +678,8 @@ without another tree walk.
| Twisted jets | `c^m \lambda^c`, including dyadic `1/2` |
| Carry | Truncate, arithmetic shift, extend on shared limbs |
| Prefix parity | XOR or signed prefix sums along a key |
| LUTs | Constant, easy, dyadic, range, window, principal |
| LUT union | Several piecewise tables on one comparison and one prefix walk |
| LUTs | Constant, easy, dyadic, range, window, principal, Haar and bior(5,3) |
| Closed form / exact steps | Compositions and digit or bit counts |
| `fixed_mul` | Fixed-point product into a chosen width |
@ -599,8 +711,15 @@ local fixed-width arithmetic (`fixed_mul` uses at most 8 limbs). Carry
keys are one comparison per live recipe flag, on the limb width, plus
one Beaver bit-opening when the recipe multiplies share MSBs. Prefix
parity on `m` endpoints is one resumed path walk, `O(m n)` expands in
the worst case; that walk follows [ePrint 2023/108](@ref bib_grotto). The degree-0 exact
LUTs follow Appendix D of the same paper. Other LUT calls are a knot
the worst case; that walk follows [ePrint 2023/108](@ref bib_grotto).
Several piecewise LUTs share one such comparison:
[make_lut_union_plan](@ref grotto/lut_union.hpp) unions their breakpoints,
and [schedule_lut_union](@ref grotto/lut_union.hpp) is one `fss_cmp` of
`n` rounds. The prefix walk over the union is local.
The degree-0 exact
LUTs follow Appendix D of the same paper. Haar and bior(5,3) tables
compress a uniform grid and evaluate in \f$\Theta(1)\f$ arithmetic
([ePrint 2025/013](@ref bib_wave)). Other LUT calls are a knot
search plus a constant-size Horner; exact steps loop over the word.
Detail is on the Grotto pages.
@ -618,17 +737,55 @@ Ship keys over ASIO peers with `dpf::asio::make_dpf`.
## Running protocols {#tour_party}
The overview of links, composition, leaf MPC, and measurement is
[Network, parties, and MPC](@ref network_and_mpc).
The `party/` programs run a three-role mesh (`p0`, `p1`, dealer `p2`).
Flows cover Beaver, geneval, DCF, DPF3, Grotto, and adversarial checks.
Use `--list` and `--tag` to filter.
Composed protocols record strands on a `dpf::protocol::composer`
([compose.hpp](@ref dpf/compose.hpp)). Values are tagged with a share
domain (`fss`, `a`, `b`, `rss`, `y`). An FSS leaf consumed by an ABY2.0
product gets a local `b2a` (or `fss2a`) inserted by `as`, and the leaf stays
on the beaver barrier's critical path so Duoram / SUBLEQ scale cannot float
before the walk. Like blinds and expansions are interned across sub-strands,
and independent opens share one RoundSink round. Party-count changes use an
explicit `reshare` (`rss_from_y` for y→rss). Composer-owned Beaver sessions use
`beavers::schedule_objective::rounds` so sign×polynomial stays one online
round; dealer benches that want Appendix-E peels keep the default `prep`
objective. Express/Sabre-style audits use `fss_point_fused` /
`level_walk_fused` so the sketch rides in the last CW flush.
BGI early-stop is `fss_point_early_stop`; Poplar checkpoints are
`level_walk_prefixes`; DCF `block_width` sizes are `level_walk_sized`.
Doerner–Shelat is `level_walk_ds` / `level_walk_ds_sized` (OH AND-layers
match `ds_oh_exchanges_per_level`); adaptive idpf_agg is staged
`step_adaptive_prefix` → drive tail (`from_exchange_wave`) →
`retain_adaptive_prefix`; keyword PIR buckets are `multipoint_fan`;
multi-lane ABY is `aby_lane`; prepaid SUBLEQ expands are `defer_expand`.
Party drivers use `util::drive_composed` / `util::drive_composed_trio` on
`composer::default_plan()` with `u64_beaver_host` or
`u64_auth_beaver_host`. A client/server query is `client_servers`
(one upload, one answer). Many instances with uneven stalls are
`dpf::app::run_fleet`: a worker parks instead of spinning and runs
whichever side can still submit. Cross-party reshare is `reshare_with_mask`;
pads are `dealer_deliver`; extra payloads share a round via
`exchange_fuse`; occupied cuckoo buckets are `schedule_cuckoo_probes`.
The full API and what stays outside compose are
[Protocol composition](@ref protocol_compose).
**Go deeper:** [trio.hpp](@ref dpf/net/trio.hpp),
[compose.hpp](@ref dpf/compose.hpp),
`party/registry.hpp` (in-tree).
## Suggested reading order {#tour_order}
1. [First program](@ref basics), then this tour if you want the map in prose.
2. [Capabilities](@ref capabilities): verifiability, programmability, comparisons, multipoint, three servers, dealer-free keygen, Beaver, Grotto.
2. [Capabilities](@ref capabilities) for key features; [Network, parties, and MPC](@ref network_and_mpc) for the runtime; [Logging, statistics, and experiments](@ref experiment_costs) for the run log and CSV costs.
3. [Domains](@ref input_types) and [Payloads](@ref output_types).
4. [Evaluation](@ref evaluation) and the [code examples](@ref listings).
5. [Application sketches](@ref applications).
\htmlonly
<div class="tldr"><b>TL;DR.</b> Point keys are make_dpf: subtract to open, one correction word per level, early-stop packing when the output is small. Comparisons and intervals add instead of subtract. verifiable, extractable, and cuckoo multipoint share one paper (ePrint 2021/580). Dealer-free keygen is the Doerner–Shelat opening, with IKNP pads when nobody deals them. Three parties are either Shamir spines (any two open) or an information-theoretic table (all three add). After a public offset, Grotto corrects polynomials, carries, and tables without another walk. Live runs use the party mesh, composer, Beaver/Yao/arith on leaves, start_logging, and experiment CSVs — see Network &amp; MPC and Logging &amp; statistics.</div>
\endhtmlonly

View file

@ -1,3 +1,5 @@
An *input type* is the domain of the secret index.
Shorter domains make shorter keys and faster walks.
Prefer `std::uint16_t` over `unsigned short` so the depth is obvious.
@ -52,7 +54,8 @@ Three names cover every integer width:
in `GF(2)^N`. See `xor_wrapper` below.
`dpf::bit`, `dpf::twobit`, and `dpf::nyble` are packed output lanes of width
1, 2, and 4. They are not domains. A domain of that width is `modint<1>`,
1, 2, and 4. `dpf::gf2`, `gf22`, and `gf24` are the same widths in GF(2^k).
They are not domains. A domain of that width is `modint<1>`,
`modint<2>`, or `modint<4>` (or the matching `xint`).
**See also**\n
@ -70,7 +73,6 @@ and [Output types](@ref output_types) for the packed lanes.
<details class="type-note">
<summary>Extended-precision integer scalar types</summary>
The extended-precision (`128`-bit) integer scalar types provided as
compiler extensions by most major C++ compilers (e.g., `__int128` and `unsigned __int128`), including `g++` and
`clang++` when compiling for `64`-bit targets. (As these types are not
@ -93,6 +95,9 @@ to declare such 128-bit integers in compiler-independent way.
<details class="type-note">
<summary>dpf::modint&lt;Nbits&gt;</summary>
\htmlonly
<div class="eli5"><b>ELI5.</b> modint&lt;N&gt; is an integer modulo 2^N. The DPF depth is N, not the width of the integer stored underneath. Arithmetic uses that underlying integer and masks on read, so it is as fast as the raw word and still wraps at N bits.</div>
\endhtmlonly
Arbitrary-, yet fixed-bitlength unsigned integer types. `dpf::modint` is a
lightweight class template that adapts one of the above-mentioned integer
@ -127,7 +132,6 @@ Here are some examples of arithmetic operations with `dpf::modint`:
<details class="type-note">
<summary>dpf::bitstring&lt;Nbits&gt;</summary>
Arbitrary-, yet fixed- bitlength binary strings types. `dpf::bitstring` is
a class template that represents a binary string of any given length.
Compared with `dpf::modint`, a `dpf::bitstring` is well suited to cases
@ -162,7 +166,6 @@ shortest length -- and with the fastest evaluations -- possible.
<details class="type-note">
<summary>dpf::keyword&lt;Alphabet, N&gt;</summary>
Fixed-length strings over restricted alphabets. `dpf::keyword` is an alias
for the class template `dpf::basic_fixed_length_string`, which represents
a string of length `N` consisting solely of letters from
@ -175,6 +178,7 @@ internal representation. This produces representations that are
when `alphabet` comprises few elements.
For example
\code{cpp}
const char cstr[] = "7fffae02";
std::cout << (sizeof(cstr) - 1) * CHAR_BIT << "\n"; // prints 64
@ -218,6 +222,9 @@ The `dpf::alphabets` namespace for a catalog of predefined alphabets.
<details class="type-note">
<summary>dpf::xor_wrapper&lt;T&gt;</summary>
\htmlonly
<div class="eli5"><b>ELI5.</b> The group operation is XOR of the underlying word, not addition. A domain of this type walks the bits in GF(2)^N. A leaf of this type opens by XORing the two shares.</div>
\endhtmlonly
An element of `GF(2)^N` for `N=8*sizeof(T)`. `xor_wrapper` is a
lightweight class template that adapts "integer-like" types so that
@ -246,7 +253,6 @@ when the index is an ordinary integer of the same width.
<details class="type-note">
<summary>dpf::keyword2&lt;Pattern&gt;</summary>
A ranked string whose language is a static pattern. `dpf::keyword2` is the
replacement for `dpf::keyword`: the pattern is the type, and each accepted
string has one rank in `0 .. |L|-1`. That rank is the DPF input. The type
@ -290,6 +296,9 @@ auto [k0, k1] = dpf::make_dpf(alpha, std::uint64_t{1});
<details class="type-note">
<summary>grotto::fixedpoint&lt;FractionalBits, IntegralType&gt;</summary>
\htmlonly
<div class="eli5"><b>ELI5.</b> The value is an integer with a public split between integer bits and fraction bits. Leaf addition is integer addition of the raw word. It does not shift the binary point.</div>
\endhtmlonly
A fixed-point value stored in an integer backend. `FractionalBits` is the
number of bits after the binary point. `IntegralType` defaults to
@ -323,6 +332,9 @@ auto [k0, k1] = dpf::make_dpf(half, std::uint64_t{1});
<details class="type-note">
<summary>dpf::wildcard_value&lt;T&gt; as a domain</summary>
\htmlonly
<div class="eli5"><b>ELI5.</b> Keygen plants a random mask instead of the index. The parties later open mask − alpha, not alpha. Until that offset is reconstructed, evaluation throws.</div>
\endhtmlonly
A wildcard input defers the secret index. `make_dpf(dpf::wildcard_value<Input>{}, beta)` plants a random mask in each key's `offset_x`. The public value the parties later open is `mask - alpha`, not `alpha`.
@ -352,7 +364,6 @@ auto [k0, k1] = dpf::make_dpf(dpf::wildcard_value<std::uint8_t>{}, std::uint32_t
<details class="type-note">
<summary>Custom input type requirements</summary>
A type can be a DPF input when the library can walk its bits and, for interval
or full-domain evaluation, order its values.
@ -392,3 +403,7 @@ template <> struct mod_pow_2<input_type> {
}
\endcode
</details>
\htmlonly
<div class="tldr"><b>TL;DR.</b> Shorter domains make shorter keys. Use a fixed-width integer when it fits, modint or xint for any other width up to 256, and bitstring or keyword when the index is not a number. bit, twobit, and nyble are payloads, not domains. A wildcard index is filled in after keygen.</div>
\endhtmlonly

View file

@ -1,5 +1,5 @@
\htmlonly
<p class="hero-lead"><code>libdpf++</code> is a header-only C++17 library of distributed point functions: short keys that hide one secret index, then answer at public points as secret shares.</p>
<p class="hero-lead"><code>libdpf++</code> is a header-only C++17 library of distributed point functions: short keys that hide one secret index, then answer at public points as secret shares. The same tree ships a TLS party mesh, round scheduling, MPC on those leaf shares, leveled run logs, and paper-cost statistics — so readers do not have to dig the API to find them.</p>
<h2 id="features">What it does</h2>
<div class="feature-grid">
<a class="feature-card" href="verifiability.html"><span class="feature-kicker">Proofs</span><strong>Verifiability &amp; authenticity</strong><p>Honest correction seeds, a weight-1 sketch, and MACs on the leaves, on the same walk.</p></a>
@ -8,10 +8,19 @@
<a class="feature-card" href="multipoint_keys.html"><span class="feature-kicker">Many secrets</span><strong>Multipoint keys</strong><p>Pack many secret points into one cuckoo key. One batched proof covers the set.</p></a>
<a class="feature-card" href="multiparty.html"><span class="feature-kicker">Three parties</span><strong>Multiparty &amp; 3-server</strong><p>Any two of three open a Shamir key. Or an information-theoretic three-server DPF.</p></a>
<a class="feature-card" href="dealer_free.html"><span class="feature-kicker">No dealer</span><strong>Dealer-free keygen</strong><p>The parties already share the index. Doerner–Shelat, geneval, and IKNP finish the key.</p></a>
<a class="feature-card" href="beaver_triples.html"><span class="feature-kicker">Multiplication</span><strong>Beaver triples</strong><p>Authenticated products on leaf shares, including the ABY2.0 MAC check.</p></a>
<a class="feature-card" href="jet_and_ring.html"><span class="feature-kicker">After the offset</span><strong>Grotto</strong><p>Jets, polynomials, carry, and exact ring changes once a public offset is open.</p></a>
<a class="feature-card" href="jet_and_ring.html"><span class="feature-kicker">After the offset</span><strong>Grotto</strong><p>Jets, polynomials, carry, and exact ring changes once a public offset is open. Several LUTs share one comparison.</p></a>
<a class="feature-card" href="ppvc_manual.html"><span class="feature-kicker">Commitments</span><strong>Programmable vectors</strong><p>Bind a vector, then open one hidden coordinate or the sum.</p></a>
<a class="feature-card" href="applications.html"><span class="feature-kicker">Protocols</span><strong>Application sketches</strong><p>The DPF step of Duoram, keyword PIR, PSI, Prio, LLAMA, and the rest.</p></a>
<a class="feature-card" href="applications.html"><span class="feature-kicker">Sketches</span><strong>Application sketches</strong><p>The DPF step of Duoram, keyword PIR, PSI, Prio, LLAMA, and the rest.</p></a>
</div>
<h2 id="around-the-keys">Around the keys</h2>
<p class="hero-lead" style="font-size:1.05rem;margin-bottom:0.6rem">Live parties, composition, MPC on leaf shares, logging, and statistics — first-class, not buried in the reference.</p>
<div class="feature-grid">
<a class="feature-card" href="network_and_mpc.html"><span class="feature-kicker">Links</span><strong>Network &amp; party mesh</strong><p>TLS 1.3 party sessions, trio mesh, RoundSink rounds, lanes, and reconnect.</p></a>
<a class="feature-card" href="protocol_compose.html"><span class="feature-kicker">Schedule</span><strong>Protocol composition</strong><p>FSS walks next to Beaver opens on one RoundSink plan, with explicit reshares.</p></a>
<a class="feature-card" href="beaver_triples.html"><span class="feature-kicker">Multiplication</span><strong>Beaver triples</strong><p>Authenticated products on leaf shares, including the ABY2.0 MAC check.</p></a>
<a class="feature-card" href="arith_runtime.html"><span class="feature-kicker">Shares</span><strong>Arithmetic share runtime</strong><p>edaBits, truncate, share compare, matmul, and hidden shuffle beside FSS.</p></a>
<a class="feature-card" href="yao_leaf.html"><span class="feature-kicker">Circuits</span><strong>Yao on a leaf</strong><p>Split a leaf into bits, garble a netlist, share the answer back as a leaf.</p></a>
<a class="feature-card" href="experiment_costs.html"><span class="feature-kicker">Observability</span><strong>Logging &amp; statistics</strong><p>Leveled run logs, provenance banners, replayable seeds, and CSV wire / PRG / timing breakdowns.</p></a>
</div>
\endhtmlonly

View file

@ -158,3 +158,7 @@ grotto::offset_iterable shifted(knots.begin(), knots.end(), 10);
**Defined in**\n
@ref grotto/offset_iterable.hpp
\htmlonly
<div class="tldr"><b>TL;DR.</b> eval_interval and eval_full walk a subinterval; eval_sequence walks the listed points. indices_set_in, advice_bits_of, batch_of, tuple_as_zip, and rotated_by only change the step. None of them copy the buffer.</div>
\endhtmlonly

View file

@ -12,6 +12,10 @@ The same offset also drives [offset Horner](@ref offset_horner),
## Binomial jet {#offset_jet}
\htmlonly
<div class="eli5"><b>ELI5.</b> The dealer keys the binomial coefficients of (center + eta) up to a chosen degree. After eta is public, a dot with those shares is the monomial or the polynomial, with no further tree walk.</div>
\endhtmlonly
`make_offset_jet_keys(center, degree)` keys one incremental `gt` whose
payload is the vector of \f$\binom{\mathrm{center}}{k}\f$ in
\f$\mathbb{Z}/2^{64}\f$. After `eta` opens,
@ -61,6 +65,10 @@ one reciprocal after the shares are opened. Those are not separate APIs.
## Exact ring switch {#ring_switch}
\htmlonly
<div class="eli5"><b>ELI5.</b> An n-bit limb is rewritten into another modulus, a field, or a P-256 scalar by an exact map on the opened residue. The value does not go through floating point, and the map does not expand another key.</div>
\endhtmlonly
For an unsigned \f$n\f$-bit limb (\f$n\le 64\f$) with representatives in
\f$[0,2^n)\f$,
@ -117,6 +125,10 @@ See also [representation shift and twisted jets](@ref repr_and_twist).
## Offset Horner {#offset_horner}
\htmlonly
<div class="eli5"><b>ELI5.</b> Powers of a public center are already shared. Shifting them by the opened eta, the binomial way, evaluates the polynomial at the secret. The degree here is fixed in the template.</div>
\endhtmlonly
`make_offset_horner_keys<Input, Degree>(center)` keys one `gt` whose
payload is `center^m` for `m = 0 .. Degree`. `Degree` is at most 3
(`offset_horner_max_degree`). Pass `dpf::verifiable{}` for proof tokens.
@ -147,6 +159,10 @@ reusable key.
## Offset polynomial {#offset_poly}
\htmlonly
<div class="eli5"><b>ELI5.</b> Same shift as offset Horner, but the degree is an argument, so the number of powered shares is chosen when the keys are built.</div>
\endhtmlonly
`make_offset_poly_keys(center, degree)` is offset Horner at a runtime
degree, at most 16 (`offset_poly_max_degree`). One incremental `gt`
whose payload is the vector of powers. `offset_poly_eval<Party>` dots the shifted powers.
@ -176,6 +192,10 @@ auto opened = s0 + grotto::offset_poly_eval<1>(mat, knots, coeff, eta);
## Carry {#carry}
\htmlonly
<div class="eli5"><b>ELI5.</b> A carry across a shift is a short list of comparisons and bit corrections, not a generic circuit. The request names the source width, the shift, and the width of what comes out. Truncate, arithmetic shift, and sign-extend are the same plan with different output widths.</div>
\endhtmlonly
A `carry_request` names the source width `n`, the shift `s`, the output
width `out_n`, a `carry_mode` (`truncate_reduce`, `same_ring`, `extend`,
`window`), and a `sign_knowledge` (`unknown`, `nonnegative`, `negative`).
@ -210,6 +230,10 @@ auto clear = grotto::eval_carry_clear(keys.recipe, x0, x1);
## Prefix parity {#prefix_parity}
\htmlonly
<div class="eli5"><b>ELI5.</b> One walk of an existing key stops at the public endpoints and folds XOR or addition along that prefix. The fold is O(number of endpoints), not a new key per prefix.</div>
\endhtmlonly
`prefix_parities(key, endpoints)` walks a key to the sorted endpoints and
returns XOR shares of the prefix parities, plus the index of the first
endpoint on the wrap. `segment_parities` turns those into one share per
@ -241,6 +265,67 @@ auto signs = grotto::signed_prefix_parities(cmp0, ends);
**Defined in**\n
@ref grotto/prefix_parity.hpp
## Several LUTs, one comparison {#lut_union}
\htmlonly
<div class="eli5"><b>ELI5.</b> Stack the breakpoints of every table into one sorted list. One comparison and one prefix walk label the pieces of that list. Each table then sums the labels that fall inside its own intervals.</div>
\endhtmlonly
`make_lut_union_plan(luts, eta)` shifts every piecewise LUT by the public
`eta`, inserts the same domain-minimum and carry cuts as
[offset polynomial](@ref offset_poly), and sorts the union. A piece of one
LUT is a span of those union knots: `[begin, end)`, or
`[begin, end-of-union) ∪ [0, end)` when the piece wraps. The span stores
that piece's binomial shift by its public `kappa`. Spans of one LUT
partition the union.
The interactive plan is one comparison, whatever the number of LUTs and
whatever the number of union knots:
- `plan.comparisons` and `plan.prefix_walks` are 1.
- `plan.depth` and `plan.geneval_rounds()` are the bitlength of the input.
- `plan.degree` is the widest polynomial. The payload is
`1, center, …, center^degree`.
- `schedule_lut_union` records one `fss_cmp` of that depth. The slot is
`lut_union_slot_bytes`: one AES block, or `lanes * 8` when the power
vector is wider. Prefix parity of the union is local after that
comparison.
`lut_union_eval<Party>` reads one `make_offset_poly_keys` key of degree
at least `plan.degree` and returns a share per LUT. `geneval_lut_union`
opens that comparison from XOR shares of the center, as in
`geneval_offset_horner`. Pass `dpf::arith_input` when the shares add to
the center in the input group.
`piecewise_from_easy` and `piecewise_from_constant` adapt the cleartext
tables. An `easy_lut` denominator other than 1 is a rounding division, so
`piecewise_from_easy` rejects it. Powers that are zero on every piece are
dropped, and the shared payload stays only as wide as the widest remaining
degree.
\code{cpp}
grotto::piecewise_lut<std::uint8_t> low{{0, 10}, {{1, 0}, {0, 2}}};
grotto::piecewise_lut<std::uint8_t> high{{0, 4, 12}, {{3, 0}, {1, 1}, {9, 4}}};
const std::uint8_t center = 12;
const std::uint8_t eta = 3;
auto plan = grotto::make_lut_union_plan({low, high}, eta);
auto mat = grotto::make_offset_poly_keys<std::uint8_t>(center, plan.degree);
auto s0 = grotto::lut_union_eval<0>(mat, plan);
auto s1 = grotto::lut_union_eval<1>(mat, plan);
dpf::protocol::composer composer(0);
grotto::schedule_lut_union(composer, plan); // rounds == plan.depth
\endcode
**Code samples**\n
<div class="tabbed">
- <b class="tab-title">lut_union.cpp</b> \include{cpp} grotto/lut_union.cpp
</div>
**Defined in**\n
@ref grotto/lut_union.hpp
## Cleartext maps {#grotto_luts}
These functions take a raw fixed-point word (`n << fractional_bits`) and
@ -250,6 +335,10 @@ return a raw word. They do not build a DPF. The type
## Fixed-point product {#fixedpoint_mul}
\htmlonly
<div class="eli5"><b>ELI5.</b> The product lives in a ring wide enough for both fixed-point operands. One Beaver triple in that ring is the product; the binary point is placed by a public shift afterward.</div>
\endhtmlonly
`fixed_mul<IntegerBits, FractionalBits>(lhs, rhs)` multiplies two
`fixedpoint` values and keeps that many integer bits (including the sign)
and fraction bits. Bits below the fraction are floored. The product type
@ -267,6 +356,10 @@ auto prod = grotto::fixed_mul<16, 16>(q16{1.5}, q16{2.0});
## Lookup tables {#lookup_tables}
\htmlonly
<div class="eli5"><b>ELI5.</b> A cleartext approximation is replaced by a table addressed with the secret. Constant, easy, range, window, and principal tables differ in how many bits of the input they consume and how the correction is added.</div>
\endhtmlonly
Constant, easy, principal, range, and window tables are included from
`grotto.hpp`. The dyadic table comes in through `exact_steps.hpp`, which
`grotto.hpp` also includes.
@ -303,6 +396,8 @@ Constant, easy, principal, range, and window tables are included from
(`principal_precision`). Names: `ln`, `exp`, `sin`, `tanf`, `tang`,
`sinh`, `cosh`, `sqrt`, `coth`, `sec`, `gsec`, `csch`, `inv`, `rsqrt`,
`invsq`.
- **Wavelet.** Haar and bior(5,3) compressed tables:
[Wavelet lookup tables](@ref dwt_luts).
\code{cpp}
auto sign = grotto::make_exact_constant_lut<std::int32_t>(
@ -368,6 +463,69 @@ picks the piece and calls that Horner step.
@ref grotto/window_lut.hpp, @ref grotto/principal_lut.hpp,
@ref grotto/piecewise.hpp
## Wavelet lookup tables {#dwt_luts}
\htmlonly
<div class="eli5"><b>ELI5.</b> A wavelet step is a fixed linear combination. The LUT stores that combination so the signal stays in shares and never enters a floating-point routine. Haar and biorthogonal 5/3 are the two filters.</div>
\endhtmlonly
`make_haar_dwt_lut` and `make_bior53_dwt_lut` compress a real signal of
length \f$2^n\f$ and evaluate it as a fixed-point word. The construction
is the cleartext Haar and bior(5,3) lookup of Reis, Ugurbil, Wagh, Henry,
and de Vega, [ePrint 2025/013](@ref bib_wave), Equations (7) and (8).
`sample_dwt_signal(domain_bits, fractional_bits, f)` writes the grid
\f$i \cdot 2^{-f}\f$ for \f$i \in [0, 2^n)\f$.
Both builders run the depth-\f$j\f$ low-pass with the smooth edge
extension used for that paper's accuracy tables. PyWavelets calls these
filters `haar` and `bior2.2`; bior(5,3) is the same pair, named there by
vanishing moments. Building either table is \f$\Theta(N)\f$ arithmetic
and extra memory, \f$N = 2^n\f$.
Haar then multiplies the approximation coefficients by \f$2^{-j/2}\f$
and rounds down to \f$f\f$ fraction bits. On this grid that coefficient
is the mean of each block of \f$2^j\f$ samples. Evaluation reads
`coeff[raw >> j]`, one indexing step, \f$\Theta(1)\f$.
bior(5,3) multiplies by \f$2^{j/2}\f$ and rounds down the same way.
Smooth extension prepends two coefficients, so the bin `msb = raw >> j`
lives at index `msb + 2`, and the next tap at `msb + 3`, wrapping in the
stored vector. With `lsb = raw mod 2^j`,
\f[
y = \bigl\lfloor\bigl(c_{\mathrm{msb}+2}\,(2^j - \mathrm{lsb})
+ c_{\mathrm{msb}+3}\,\mathrm{lsb}\bigr) / 2^{2j}\bigr\rfloor.
\f]
That is Equation (8): the Lemma 6 weights \f$(2^j - \mathrm{lsb}_j)\f$
and \f$\mathrm{lsb}_j\f$, in integer arithmetic. Two multiplications and
a shift, \f$\Theta(1)\f$.
The paper's online protocols look these tables up under a DPF. Haar is
paired there with a deterministic Pika truncation; bior(5,3) is paired
with segment parity. Those protocols are not a key type here. The value
they open is `table(raw)`.
\code{cpp}
auto samples = grotto::sample_dwt_signal(6, 4, [](double x) {
return 1.0 / (1.0 + std::exp(-(x - 2.0)));
});
auto haar = grotto::make_haar_dwt_lut(samples, 4, 2);
auto bior = grotto::make_bior53_dwt_lut(samples, 4, 2);
auto h = haar(std::uint64_t{32});
auto b = bior(std::uint64_t{33});
\endcode
**Code samples**\n
<div class="tabbed">
- <b class="tab-title">dwt_lut.cpp</b> \include{cpp} grotto/dwt_lut.cpp
</div>
**Defined in**\n
@ref grotto/dwt_lut.hpp
## Closed form {#closed_form}
`eval_closed(closed::atanh, fractional_bits, raw)` composes
@ -423,3 +581,7 @@ for a functor type.
**Defined in**\n
@ref grotto/gadgets.hpp, @ref grotto/gadget_hints.hpp
\htmlonly
<div class="tldr"><b>TL;DR.</b> Open eta = x − r once. Jets and Horner turn that public distance into polynomial powers. Ring switch and carry move the integer. Prefix parity folds a key you already hold. Lookup tables, including the wavelet tables, replace cleartext math. A fixed-point product is one Beaver triple.</div>
\endhtmlonly

View file

@ -4,6 +4,7 @@
- \subpage output_type_examples
- \subpage evaluation_examples
- \subpage grotto_examples
- \subpage protocol_examples
- \subpage iteratable_examples
\page input_type_examples Domain samples
@ -17,24 +18,31 @@
- \subpage input_types_2custom_8cpp
\page "input_types_2integral_types_8cpp" input_types/integral_types.cpp
\include{cpp} input_types/integral_types.cpp
\page "input_types_2extended_types_8cpp" input_types/extended_types.cpp
\include{cpp} input_types/extended_types.cpp
\page "input_types_2modint_8cpp" input_types/modint.cpp
\include{cpp} input_types/modint.cpp
\page "input_types_2bitstring_8cpp" input_types/bitstring.cpp
\include{cpp} input_types/bitstring.cpp
\page "input_types_2keyword_8cpp" input_types/keyword.cpp
\include{cpp} input_types/keyword.cpp
\page "input_types_2xor_wrapper_8cpp" input_types/xor_wrapper.cpp
\include{cpp} input_types/xor_wrapper.cpp
\page "input_types_2custom_8cpp" input_types/custom.cpp
\include{cpp} input_types/custom.cpp
\page output_type_examples Payload samples
@ -42,30 +50,42 @@
- \subpage output_types_2integral_types_8cpp
- \subpage output_types_2extended_types_8cpp
- \subpage output_types_2bit_8cpp
- \subpage output_types_2gf2_8cpp
- \subpage output_types_2bitstring_8cpp
- \subpage output_types_2wildcard_8cpp
- \subpage output_types_2xor_wrapper_8cpp
- \subpage output_types_2custom_8cpp
\page "output_types_2integral_types_8cpp" output_types/integral_types.cpp
\include{cpp} output_types/integral_types.cpp
\page "output_types_2extended_types_8cpp" output_types/extended_types.cpp
\include{cpp} output_types/extended_types.cpp
\page "output_types_2bit_8cpp" output_types/bit.cpp
\include{cpp} output_types/bit.cpp
\page "output_types_2gf2_8cpp" output_types/gf2.cpp
\include{cpp} output_types/gf2.cpp
\page "output_types_2bitstring_8cpp" output_types/bitstring.cpp
\include{cpp} output_types/bitstring.cpp
\page "output_types_2wildcard_8cpp" output_types/wildcard.cpp
\include{cpp} output_types/wildcard.cpp
\page "output_types_2xor_wrapper_8cpp" output_types/xor_wrapper.cpp
\include{cpp} output_types/xor_wrapper.cpp
\page "output_types_2custom_8cpp" output_types/custom.cpp
\include{cpp} output_types/custom.cpp
\page evaluation_examples Evaluation samples
@ -84,52 +104,84 @@
- \subpage evaluation_2eval_dpf3_cmp_ic_8cpp
\page "evaluation_2eval_point_8cpp" evaluation/eval_point.cpp
\include{cpp} evaluation/eval_point.cpp
\page "evaluation_2eval_interval_8cpp" evaluation/eval_interval.cpp
\include{cpp} evaluation/eval_interval.cpp
\page "evaluation_2eval_full_8cpp" evaluation/eval_full.cpp
\include{cpp} evaluation/eval_full.cpp
\page "evaluation_2defer_eval_8cpp" evaluation/defer_eval.cpp
\include{cpp} evaluation/defer_eval.cpp
\page "evaluation_2eval_sequence_8cpp" evaluation/eval_sequence.cpp
\include{cpp} evaluation/eval_sequence.cpp
\page "evaluation_2memoizers_8cpp" evaluation/memoizers.cpp
\include{cpp} evaluation/memoizers.cpp
\page "evaluation_2output_buffers_8cpp" evaluation/output_buffers.cpp
\include{cpp} evaluation/output_buffers.cpp
\page "evaluation_2eval_inner_product_8cpp" evaluation/eval_inner_product.cpp
\include{cpp} evaluation/eval_inner_product.cpp
\page "evaluation_2buffered_prg_8cpp" evaluation/buffered_prg.cpp
\include{cpp} evaluation/buffered_prg.cpp
\page "evaluation_2eval_dpf3_point_8cpp" evaluation/eval_dpf3_point.cpp
\include{cpp} evaluation/eval_dpf3_point.cpp
\page "evaluation_2eval_dpf3_doerner_shelat_8cpp" evaluation/eval_dpf3_doerner_shelat.cpp
\include{cpp} evaluation/eval_dpf3_doerner_shelat.cpp
\page "evaluation_2eval_dpf3_cmp_ic_8cpp" evaluation/eval_dpf3_cmp_ic.cpp
\include{cpp} evaluation/eval_dpf3_cmp_ic.cpp
\page grotto_examples Grotto samples
- \subpage grotto_2jet_and_ring_8cpp
- \subpage grotto_2repr_and_twist_8cpp
- \subpage grotto_2lut_union_8cpp
- \subpage grotto_2dwt_lut_8cpp
\page "grotto_2jet_and_ring_8cpp" grotto/jet_and_ring.cpp
\include{cpp} grotto/jet_and_ring.cpp
\page "grotto_2repr_and_twist_8cpp" grotto/repr_and_twist.cpp
\include{cpp} grotto/repr_and_twist.cpp
\page "grotto_2lut_union_8cpp" grotto/lut_union.cpp
\include{cpp} grotto/lut_union.cpp
\page "grotto_2dwt_lut_8cpp" grotto/dwt_lut.cpp
\include{cpp} grotto/dwt_lut.cpp
\page protocol_examples Protocol composition samples
- \subpage protocol_2compose_schedule_8cpp
\page "protocol_2compose_schedule_8cpp" protocol/compose_schedule.cpp
\include{cpp} protocol/compose_schedule.cpp
\page iteratable_examples Iterable samples
- \subpage iterables_2setbit_index_iterable_8cpp
@ -140,19 +192,25 @@
- \subpage iterables_2zip_iterable_8cpp
\page "iterables_2setbit_index_iterable_8cpp" iterables/setbit_index_iterable.cpp
\include{cpp} iterables/setbit_index_iterable.cpp
\page "iterables_2advice_bit_iterable_8cpp" iterables/advice_bit_iterable.cpp
\include{cpp} iterables/advice_bit_iterable.cpp
\page "iterables_2parallel_bit_iterable_8cpp" iterables/parallel_bit_iterable.cpp
\include{cpp} iterables/parallel_bit_iterable.cpp
\page "iterables_2subinterval_iterable_8cpp" iterables/subinterval_iterable.cpp
\include{cpp} iterables/subinterval_iterable.cpp
\page "iterables_2subsequence_iterable_8cpp" iterables/subsequence_iterable.cpp
\include{cpp} iterables/subsequence_iterable.cpp
\page "iterables_2zip_iterable_8cpp" iterables/zip_iterable.cpp
\include{cpp} iterables/zip_iterable.cpp

View file

@ -1,5 +1,9 @@
# Multiparty & 3-server {#multiparty}
\htmlonly
<div class="eli5"><b>ELI5.</b> make_dpf3 is two ordinary spines per party and a Shamir payload, so any two parties open and the third key is independent of the value. make_it_dpf3 shares the whole truth table instead, and all three shares are required. Doerner–Shelat and geneval are the two-party, no-dealer alternatives.</div>
\endhtmlonly
Two-party keys are the default. The library also builds three-evaluator
`(2,3)` keys, information-theoretic three-server DPFs, and dealer-free
two-party keygen when the index is already shared.
@ -12,6 +16,13 @@ two-party keygen when the index is already shared.
| [dpf::make_it_dpf3](@ref dpf/it_dpf3.hpp) | Information-theoretic 3-server DPF ([ePrint 2023/028](@ref bib_itdpf)) | `it_dpf3.hpp` |
| [dpf::make_dpf_doerner_shelat](@ref dpf/doerner_shelat.hpp) | Two parties, shared index, no dealer for the point | `doerner_shelat.hpp` |
| [dpf::geneval_*](@ref dpf/geneval.hpp) | Answer shares for one query, no reusable key | `geneval.hpp` |
| [dpf::shamir::deal](@ref dpf/shamir.hpp) | (K,N) Shamir. `(2,3)` is the `make_dpf3` payload | `shamir.hpp` |
The payload inside `make_dpf3` is degree-1 Shamir on the points `1`, `2`,
and `3`. That access structure is `shamir::two_of_three`, the type
`shamir_share`. The same split takes other thresholds:
`shamir::deal<T, K, N>` and `shamir::share_secret`. `make_dpf3` stays
`(2,3)`. See [secret shares](@ref secret_shares) and `examples/mwe/shamir.cpp`.
```cpp
auto [k1, k2, k3] = dpf::make_dpf3(std::uint8_t{42}, dpf::fp61{7});

View file

@ -1,5 +1,9 @@
# Multipoint keys {#multipoint_keys}
\htmlonly
<div class="eli5"><b>ELI5.</b> Cuckoo hashing puts each secret point in one of a few buckets, with three hash functions and one point key per bucket. A query probes its three candidate buckets. Key size is linear in the number of points. The S&amp;P 2025 PCG packing, which would derive those buckets from one seed, is not implemented.</div>
\endhtmlonly
`make_multipoint(alphas, betas)` packs many secret points into cuckoo
buckets (de Castro–Polychroniadou, EUROCRYPT 2022, §4 / [ePrint 2021/580](@ref bib_vdpf)):
`κ = 3` hashes, one point key per bucket. The bucket count is linear in

View file

@ -0,0 +1,92 @@
# Network, parties, and MPC around the keys {#network_and_mpc}
\htmlonly
<div class="eli5"><b>ELI5.</b> The library is about DPFs first. The same headers also wire the parties together, schedule rounds, multiply and garble leaf shares, write leveled run logs, and emit paper-cost statistics — so a PIR or Duoram sketch does not start from bare sockets.</div>
\endhtmlonly
Keys and evaluation are the core story. This page is the map of everything
that sits *around* those keys when you run a real protocol: links, party
roles, composition, arithmetic and boolean MPC on leaf shares, logging, and
cost instrumentation. Open a linked page for the details; the [call index](@ref api_reference)
lists the headers.
## Network and party mesh
| Piece | Role |
| --- | --- |
| [party_session](@ref dpf/net/party_session.hpp) | One role joins a mesh: dial/accept, lanes, reconnect, dealer link |
| [trio](@ref dpf/net/trio.hpp) | Convenience `(2+1)` mesh (`p0`, `p1`, dealer `p2`) |
| [TLS 1.3](@ref dpf/net/tls.hpp) / [security](@ref dpf/net/security.hpp) | Default on peer links; `--encryption=off` for plaintext benches |
| [RoundSink](@ref dpf/net/round_sink.hpp) / [edge_mesh](@ref dpf/net/edge_mesh.hpp) | Batched exchange rounds on star / clique / dealer topologies |
| [stream arrays](@ref dpf/net/stream_array.hpp) | Sync or async TCP (and optional SCTP) byte lanes |
Processes may start in any order: lower ids accept, higher ids connect and
retry. After connect (or TLS handshake) both ends exchange a fixed hello so
mismatched party id, epoch, transport, or lane count fail at join time.
```cpp
// Conceptual shape — see party_session / trio headers for the full API.
dpf::net::session_options opt; // TLS on by default
dpf::net::party_session session(/*role*/, opt);
session.join(/*host:port table*/); // or in-process rendezvous
auto & sink = session.round_sink(/*peer*/);
```
**Go deeper:** [trio.hpp](@ref dpf/net/trio.hpp),
[party_session.hpp](@ref dpf/net/party_session.hpp),
[secure_channel.hpp](@ref dpf/net/secure_channel.hpp),
examples under `examples/protocol/`.
## Protocol composition
[dpf::protocol::composer](@ref protocol_compose) records FSS walks, Beaver
opens, and reshares as one RoundSink plan. Share domains are tagged
(`fss`, `a`, `b`, `rss`, `y`); party-count changes are explicit `reshare`.
Independent opens share a wave; dependency chains become successive waves.
Drive with `drive` / `drive_via_schedule`, or party helpers
`util::drive_composed` / `util::drive_composed_trio`. Named application
skeletons (PIR upload/answer, SUBLEQ, hushmap, …) live in
[app_plans.hpp](@ref dpf/app_plans.hpp).
**Go deeper:** [Protocol composition](@ref protocol_compose).
## MPC on leaf shares
The DPF answers the secret index. What you do *with* the opened (or still
shared) leaf is ordinary MPC in the same library:
| Tool | When you reach for it |
| --- | --- |
| [Beaver triples](@ref beaver_triples) | Products, dots, polynomials, optional MACs (ABY2.0) |
| [Arithmetic share runtime](@ref arith_runtime) | edaBits, truncate, share compare, matmul, hidden shuffle |
| [Yao on a leaf](@ref yao_leaf) | A boolean circuit the key does not contain; half-gates |
| [Dealer-free keygen](@ref dealer_free) | Doerner–Shelat / IKNP when nobody deals the pads |
| [Multiparty & 3-server](@ref multiparty) | `(2,3)` Shamir spines or IT three-server tables |
**Go deeper:** those capability pages, then the [guided tour](@ref guided_tour)
sections on Beaver, Yao, and running protocols.
## Logging and statistics
Observability is first-class, not an afterthought:
| Facility | Role |
| --- | --- |
| [Run log](@ref run_log) (`log.hpp` / `run_log.hpp`) | Leveled `key=value` lines to stderr, syslog, or a file; provenance banner (build, host, argv, env, config); link and trial events |
| [Statistics & CSVs](@ref statistics) (`experiment.hpp`) | Replayable master seeds, per-party streams, wire bytes, rounds, wall/CPU, symmetric-key blocks by purpose×primitive |
| [`prg::count`](@ref dpf/prg_count.hpp) | Thread-local expand / hash / harness counters (AES, ChaCha, LowMC) |
| [`thread_work`](@ref dpf/thread_work.hpp) | Charge compute-pool kernels back to the owning party |
In-tree `party/` drivers and `run_parties` / `app::run` / `app::run_measured`
call `app::start_logging` and drive the mesh end-to-end. Before any party
starts its clock, all parties wait at a start gate.
**Go deeper:** [Logging, statistics, and experiments](@ref experiment_costs),
[compose.hpp](@ref dpf/compose.hpp),
[experiment.hpp](@ref dpf/experiment.hpp),
[log.hpp](@ref dpf/log.hpp).
\htmlonly
<div class="tldr"><b>TL;DR.</b> Start with make_dpf and eval. When you need live parties, party_session / trio give TLS links and RoundSink rounds; the composer schedules FSS next to Beaver and Yao; start_logging and experiment stamp the run log and paper CSVs. The keys stay the product — the net, MPC, logging, and statistics stack is how you run and measure them.</div>
\endhtmlonly

View file

@ -1,3 +1,5 @@
An output type is the group element at the secret index.
Every output on one key has the same width, and the type is trivially copyable.
Leaf addition is the group operation.
@ -11,8 +13,9 @@ Leaf addition is the group operation.
| `vec<T, N>` | `N` lanes, no carry between them | [vec.hpp](@ref dpf/vec.hpp) |
| `wildcard_value<T>` | Filled in later | [wildcard.hpp](@ref dpf/wildcard.hpp) |
| `field64` / `field128` / `fp61` | Prime field | [fp61.hpp](@ref dpf/fp61.hpp) |
| `gf2` / `gf22` / `gf24` / `gf28` / `gf216` / `gf232` / `gf264` | GF(2^k) | [gf2.hpp](@ref dpf/gf2.hpp) |
| `p256` / `p256_scalar` | Curve point or scalar | [p256.hpp](@ref dpf/p256.hpp) |
| Shares | (2,2), (3,3), or (2,3) replicated | [secret shares](@ref secret_shares) |
| Shares | (2,2), (3,3), (2,3) replicated, (K,N) Shamir | [secret shares](@ref secret_shares) |
| `fixedpoint` | Fixed-point word | [fixedpoint.hpp](@ref grotto/fixedpoint.hpp) |
`bool` is an 8-bit integer. A one-bit payload is `dpf::bit`.
@ -27,7 +30,6 @@ The notes below are closed. Open one when you need the rules.
<details class="type-note">
<summary>Integer scalar types</summary>
Fixed-width integers (`uint32_t`, `int64_t`, and the other `psnip` widths)
are an additive group. Leaves use SIMD add, subtract, and multiply, including
`char`, `long long`, `char16_t`, `char32_t`, and `wchar_t` at 8, 16, 32, and
@ -46,6 +48,9 @@ add the underlying word; values wider than one AES block add with
<details class="type-note">
<summary>Secret shares</summary>
\htmlonly
<div class="eli5"><b>ELI5.</b> The wrapper is the same bits as the payload, plus a rule for opening. Subtractive shares open as share0 − share1, which is what a point leaf returns. Additive shares open as a sum, which is what a comparison returns. A replicated share gives two of the three additive pieces to each party, so any two parties suffice. A Shamir share is one point on a polynomial; any K of the N points open the constant term.</div>
\endhtmlonly
`dpf::additive_share<T, Party>` and `dpf::subtractive_share<T, Party>` are
(2,2) shares, layout-identical to `T` (`Party` is `0` or `1`).
@ -62,6 +67,38 @@ so the secret is `x_0 + x_1 + x_2`. Any two parties reconstruct.
`as_additive3()` is that party's (3,3) component. `add_replicated` folds a
(3,3) sharing in by updating both holders of each component.
`dpf::shamir::share<T, Party, K, N>` is a Shamir share over a field.
The secret is the constant term of a polynomial of degree `K - 1`.
Party `i` holds that polynomial at `x = i + 1`. Any `K` shares open it.
`make_shamir_shares<K, N>` and `shamir::deal` are the deterministic split.
`shamir::share_secret` draws the higher coefficients with `uniform_sample`.
`shamir::reconstruct` opens typed shares, or runtime `shamir::point_share`s.
The first `K` shares are the interpolating set and are not checked.
Each further share must lie on that polynomial. That catches a share that
does not belong. It does not name the bad share, and it does not correct it.
Exactly `K` shares accept any values. When `K = N` that is every share, so
a tampered full set is not detected. A complete set of shares of a different
secret is consistent and opens that secret. A zero leading coefficient is a
lower threshold than `K`.
`(2,3)` is `shamir::two_of_three`. `shamir::share<T, Party, 2, 3>` is
`dpf::shamir_share<T, Party>` (`sharing::shamir`).
`make_shamir_shares(secret, slope)` is that case with one coefficient.
`shamir3` is the same polynomial on `fp61`, with party indices `1`, `2`,
and `3`. `make_dpf3` uses that case. The field inverse is
`detail::shamir_field`. `fp61` and `gf2n` specialize it. For `gf2n`
the integers `1 .. N` are bit patterns, so `N` must be less than `2^k`
or a party lands on `0` or on another party's point. Addition in that
field is XOR, so a share added to itself opens `0`.
A complete program is `examples/mwe/shamir.cpp`.
\code{cpp}
const std::array<dpf::fp61, 2> coeff{{dpf::fp61{2}, dpf::fp61{3}}};
auto shares = dpf::make_shamir_shares<3, 5>(dpf::fp61{10}, coeff);
auto opened = dpf::shamir::reconstruct(
std::get<0>(shares), std::get<2>(shares), std::get<4>(shares));
\endcode
Creating from a plaintext puts the value on party 0. For a replicated share
that value is component `x_0`, which party 2 also stores as `next`.
@ -84,7 +121,8 @@ Conversions that keep the secret on one party are `a2b`, `b2a`, `a2fss`,
`fss2a`, `b2fss`, and `fss2b` (`a` additive, `b` subtractive, `fss` the leaf
share), plus `rss2y` and `y2rss` between a replicated share and its (3,3)
components (`y`). `s2y`, `y2s`, `s2rss`, and `rss2s` open a reconstructing
set and split again (`s` is (2,3) Shamir, `shamir_share`, points 1, 2, 3).
set and split again (`s` is the (2,3) case of (K,N) Shamir, `shamir_share`,
points 1, 2, 3).
A cast that changes how many parties hold the secret, including anything
that would need a garbled circuit, is not a conversion. `rss_mul` is the
replicated product: each party forms `x_i y_i + x_i y_{i+1} + x_{i+1} y_i`,
@ -94,7 +132,6 @@ and the three terms are an additive sharing of the product.
<details class="type-note">
<summary>Extended-precision integer scalar types</summary>
`simde_int128`, `simde_uint128`, `uint128_t`, and `uint256_t` are additive.
Their leaf arithmetic is ordinary addition of those integers.
</details>
@ -102,7 +139,6 @@ Their leaf arithmetic is ordinary addition of those integers.
<details class="type-note">
<summary>dpf::bit</summary>
A one-bit output. The group is XOR: `operator+` and `operator-` are both XOR,
and a leaf multiply is AND with an all-zero or all-one mask. Many `dpf::bit`
outputs are packed into each leaf, low bit first. `dpf::bit::zero` and
@ -123,7 +159,6 @@ auto [k0, k1] = dpf::make_dpf(std::uint8_t{3}, dpf::bit::one);
<details class="type-note">
<summary>dpf::twobit</summary>
A 2-bit output in the ring Z/4Z. Values are `0` through `3`
(`twobit::zero` .. `twobit::three`). Scalar `+` and `-` wrap modulo 4.
A leaf packs one lane every two bits, low lane in the low bits of the first
@ -144,7 +179,6 @@ auto [k0, k1] = dpf::make_dpf(std::uint8_t{3}, dpf::twobit::two);
<details class="type-note">
<summary>dpf::nyble</summary>
A 4-bit output in the ring Z/16Z. Values are `0` through `15`. Scalar `+`
and `-` wrap modulo 16. A leaf packs one lane every four bits, low nibble
first. Leaf addition is not XOR and is not a byte add: a carry must not
@ -165,7 +199,6 @@ auto [k0, k1] = dpf::make_dpf(std::uint8_t{3}, dpf::to_nyble(0xau));
<details class="type-note">
<summary>dpf::bitstring&lt;Nbits&gt;</summary>
A fixed string of bits in the XOR group. `operator+`, `operator-`, and leaf
addition are XOR. The leftmost character of a literal or of `to_string` is
the most significant bit, matching `0b` notation. Bits above `Nbits` are not
@ -176,7 +209,6 @@ part of the value. `dpf::bitN_t` is `bitstring<N>` for `N` from 1 through
<details class="type-note">
<summary>dpf::vec&lt;T, N&gt;</summary>
`N` lanes of an ordinary output `T`, stored with lane 0 in the least-significant
place. `T` is an integer, `modint`, `xint` / `xor_wrapper`, `fixedpoint`,
`twobit`, or `nyble`. `+`, `-`, and `*` on a `vec` run per lane and do not
@ -200,6 +232,9 @@ auto [k0, k1] = dpf::make_dpf(std::uint8_t{3}, beta);
<details class="type-note">
<summary>dpf::wildcard_value&lt;T&gt;</summary>
\htmlonly
<div class="eli5"><b>ELI5.</b> The leaf group is the group of T, but the correction word is withheld. Evaluation throws until assign writes it. The assign itself is a Beaver correction, not a new tree.</div>
\endhtmlonly
A placeholder for an output of type `T`. The leaf group is the group of `T`.
`make_dpf` accepts an empty `wildcard_value<T>{}` (or `dpf::wildcard<T>`, or
@ -245,6 +280,9 @@ A wildcard *input* is separate: it masks the secret index. See
<details class="type-note">
<summary>dpf::xor_wrapper&lt;T&gt;</summary>
\htmlonly
<div class="eli5"><b>ELI5.</b> The group operation is XOR of the underlying word, not addition. A domain of this type walks the bits in GF(2)^N. A leaf of this type opens by XORing the two shares.</div>
\endhtmlonly
An element of `GF(2)^n` for `n = 8 * sizeof(T)`, or `n = N` for
`dpf::xint<N>`. `operator+` and `operator-` are XOR, `operator*` is AND, and
@ -258,8 +296,42 @@ leaf scaling is AND.
</details>
<details class="type-note">
<summary>Prime fields and P-256</summary>
<summary>GF(2^k)</summary>
`dpf::gf2`, `gf22`, `gf24`, `gf28`, `gf216`, `gf232`, and `gf264` are
GF(2^k) for k = 1, 2, 4, 8, 16, 32, and 64. Leaf addition is XOR. Leaf
scaling is field multiplication. Widths below 8 bits pack one field
element per lane, low lane first, and addition XORs that lane. A negative
integer constructs the same element as its magnitude.
`shamir::deal` and `shamir::reconstruct` run over these fields. The
shareholder count `N` must be less than `2^k`: the point integers are
stored as bit patterns, and `2^k` itself is `0`. `gf2` can hold one
shareholder. `gf22` can hold three, which is a `(2,3)` sharing. A share
added to itself is `0`.
| Type | Modulus |
| --- | --- |
| `gf2` | `x + 1` |
| `gf22` | `x^2 + x + 1` |
| `gf24` | `x^4 + x + 1` |
| `gf28` | `x^8 + x^4 + x^3 + x + 1` (AES, `0x11B`) |
| `gf216` | `x^16 + x^5 + x^3 + x^2 + 1` |
| `gf232` | `x^32 + x^31 + x^28 + x^21 + 1` |
| `gf264` | `x^64 + x^63 + x^62 + x^53 + 1` |
```cpp
auto [k0, k1] = dpf::make_dpf(std::uint8_t{3}, dpf::gf28{0x1b});
auto [s0, s1, s2] = dpf::make_shamir_shares(dpf::gf28{0x1b}, dpf::gf28{0x5a});
auto opened = dpf::reconstruct(s0, s2);
```
**Defined in**\n
@ref dpf/gf2.hpp
</details>
<details class="type-note">
<summary>Prime fields and P-256</summary>
`dpf::field64` is GF(2^64 − 2^32 + 1), the same prime as libprio `Field64`.
`dpf::field128` is GF(340282366920938462946865773367900766209), the same
@ -294,6 +366,9 @@ recipes still use the integer limb channel.
<details class="type-note">
<summary>grotto::fixedpoint&lt;FractionalBits, IntegralType&gt;</summary>
\htmlonly
<div class="eli5"><b>ELI5.</b> The value is an integer with a public split between integer bits and fraction bits. Leaf addition is integer addition of the raw word. It does not shift the binary point.</div>
\endhtmlonly
A fixed-point output stored in an integer backend. `FractionalBits` is the
number of bits after the binary point. `IntegralType` defaults to
@ -318,7 +393,6 @@ auto [k0, k1] = dpf::make_dpf(std::uint8_t{3}, fp{1});
<details class="type-note">
<summary>Custom output type requirements</summary>
Specialize `dpf::leaf_arithmetic::add_t`, `subtract_t`, and `multiply_t` for
the exterior node type (`simde__m128i` for the default AES PRG, and
`simde__m256i` when that node is used). Each functor's call operator receives
@ -335,3 +409,7 @@ Outputs must be trivially copyable and standard layout. `dpf::utils::make_from_i
should build `T` from the integer `1` when tests or `make_default` need a
nonzero payload. See `test/tests/helpers/custom_output_type_small.hpp`.
</details>
\htmlonly
<div class="tldr"><b>TL;DR.</b> One key, one leaf width. Integers and modint add; xint, bit, bitstring, and GF(2^k) XOR; twobit and nyble wrap inside their lane. Typed shares remember whether opening adds or subtracts, and whether any two parties suffice. A wildcard leaf throws until it is assigned.</div>
\endhtmlonly

View file

@ -1,5 +1,9 @@
# Point-programmable vector commitments {#ppvc_manual}
\htmlonly
<div class="eli5"><b>ELI5.</b> The vector is bound before the coordinate is chosen. Each coordinate is a pair of 1-bit DPF roots, and both roots are committed with a Naor string. Opening one side of each pair writes the hidden coordinate, or the sum, onto a public index.</div>
\endhtmlonly
A point-programmable vector commitment binds a vector
`x` in `(Z/2^s Z)^n` and still lets one hidden coordinate be chosen
after the commitment is published.
@ -53,6 +57,10 @@ storage, so two expansions on one thread must not overlap.
## What an opening proves {#ppvc_verify}
\htmlonly
<div class="eli5"><b>ELI5.</b> verify recomputes the Naor string on each opened root and compares it to the commitment. A root that was not the committed one fails. The check does not reveal the other coordinates.</div>
\endhtmlonly
`verify` checks each opened root against its Naor string.
`accept` also checks the programmed statement: the rotated coordinate
when `mu` is 0, the column sum when `mu` is 1.
@ -89,3 +97,7 @@ The generator is `dpf::prg::aes128` unless another 128-bit PRG is named.
Naor's string commitment is Moni Naor, [Bit Commitment Using Pseudorandomness](@ref bib_naor), Journal of Cryptology 1991.
The point keys are the Boyle–Gilboa–Ishai construction named in
[DPF basics](@ref point_functions).
\htmlonly
<div class="tldr"><b>TL;DR.</b> Commit binds the vector before the coordinate is chosen, by committing both DPF roots. Open writes one coordinate or the sum onto a public index. verify checks those roots against the Naor strings.</div>
\endhtmlonly

View file

@ -1,5 +1,9 @@
# Programmability {#programmability}
\htmlonly
<div class="eli5"><b>ELI5.</b> A wildcard is a placeholder with the correction word not yet fixed. Assigning an input or a leaf is a Beaver multiplication against that placeholder: one blinded share is exchanged, then both parties write the same correction. An updatable leaf can be patched again the same way.</div>
\endhtmlonly
Fill in a secret index or payload after the key exists, rewrite an
updatable leaf, or commit to a vector before choosing which coordinate
to open.

View file

@ -7,6 +7,10 @@ with a public Pascal shift plus \f$\lambda^{\kappa}\f$.
## Representation shift {#offset_repr}
\htmlonly
<div class="eli5"><b>ELI5.</b> Offset Horner is the Pascal-matrix case of a shift-invariant recurrence. Representation shift is the same idea for a general linear recurrence: a public step count kappa advances the shared state without rebuilding the key.</div>
\endhtmlonly
Offset Horner is the unipotent (Pascal) case of a shift-invariant module.
Here the dealer keys an arbitrary state
\f$S_c\in(\mathbb{Z}/2^{64})^d\f$ at the hidden center. After `eta` opens, each
@ -55,6 +59,10 @@ squaring, at most 63 squarings) and applies it on every refined piece,
## Twisted jets {#offset_twist}
\htmlonly
<div class="eli5"><b>ELI5.</b> The dealer keys one comparison whose payload is the vector of twisted powers c^m λ^c, including the dyadic case c = 1/2. After eta opens, the parties scale that vector. They do not re-expand the tree.</div>
\endhtmlonly
The dealer keys one comparison whose payload is the vector of twisted powers
\f$c^{m}\lambda^{c}\f$ in \f$\mathbb{Z}/2^{64}\f$. After `eta` opens, the segment
walk returns those shares on the hot piece. A public binomial shift of the
@ -112,5 +120,10 @@ auto half_keys = grotto::make_offset_twist_keys<std::uint8_t>(
center, 2, grotto::twist_half);
\endcode
Offset Horner, offset polynomials, carry, prefix parity, and the cleartext
Offset Horner, offset polynomials, a union of several LUTs on one
comparison, carry, prefix parity, and the cleartext
LUTs are on [jet and ring](@ref jet_and_ring).
\htmlonly
<div class="tldr"><b>TL;DR.</b> Both start from the opened offset eta = x − r. Representation shift advances a linear recurrence by a public step count. Twisted jets scale a vector of powers that already includes the constant factor.</div>
\endhtmlonly

View file

@ -1,5 +1,9 @@
# Verifiability & authenticity {#verifiability}
\htmlonly
<div class="eli5"><b>ELI5.</b> The proof rides on the same walk as the payload. verifiable checks that the correction seeds were the honest ones. extractable checks that the path is weight 1, so a second programmed point fails. A MAC checks the leaf share after it is opened.</div>
\endhtmlonly
Prove that a DPF walk used honest correction seeds, or that a weight-1
sketch over the path is consistent. The tags ride along as extra
`make_dpf` arguments; eval still returns the usual leaf share.

View file

@ -17,11 +17,13 @@ Types in full are on [Input types](@ref input_types) and [Output types](@ref out
- The parties already share `alpha` and want a reusable key: [dpf::make_dpf_doerner_shelat](@ref dpf/doerner_shelat.hpp).
- The parties want the answer and no key: [dpf::geneval_point](@ref dpf/geneval.hpp), [dpf::geneval_interval](@ref dpf/geneval.hpp), [dpf::geneval_cmp](@ref dpf/geneval.hpp).
- Three evaluators: [dpf::make_dpf3](@ref dpf/dpf3.hpp) and [dpf::make_dpf3_doerner_shelat](@ref dpf/dpf3_ds.hpp).
- A Shamir secret for any threshold, not only (2,3): [shamir::deal](@ref dpf/shamir.hpp) and `examples/mwe/shamir.cpp`.
- One point, a comparison, or a public interval: [dpf::gt](@ref dpf/dcf.hpp) and the other predicates, and [dpf::ic](@ref dpf/interval.hpp).
- Many secret points: [dpf::make_multipoint](@ref dpf/multipoint.hpp).
- A vector whose hidden coordinate is chosen after the commitment: [point-programmable vector commitments](@ref ppvc_manual).
- Several lanes at one leaf: [dpf::vec](@ref dpf/vec.hpp).
- A payload filled in later: [dpf::wildcard_value](@ref dpf/wildcard.hpp).
- A proof the key is well formed: [dpf::verifiable](@ref dpf/verifiable.hpp).
- A leaf share must enter a bit circuit the key does not compute: [b2y](@ref yao_leaf) and the netlist on that page. A comparison, an interval, or a public-offset polynomial stays a key.
The call index, with one line each, is the [API reference](@ref api_reference).

124
doc/pages/yao.md Normal file
View file

@ -0,0 +1,124 @@
# A boolean function of a leaf {#yao_leaf}
\htmlonly
<div class="eli5"><b>ELI5.</b> The key still answers the point, the comparison, and the interval. When the leaf you already hold has to go through a bit circuit the key does not contain, split that leaf into XOR bits, garble the circuit, and share the result back in the leaf's own type.</div>
\endhtmlonly
Party 0 garbles. Party 1 evaluates. Free-XOR, half-gates
([ePrint 2014/756](@ref bib_halfgates)). One table row is 32 bytes.
Outputs are XOR shares of the output bits: the garbler's share is the
permute bit, the evaluator's share is the color of the label it holds.
The circuit this library is built around is the zero-key AES-MMO block
`prg::aes128::eval` already uses on every tree expand. A seed or a leaf
can enter that block without being opened. The same netlist is a hand-built
straight line of XOR, AND, XNOR, and NOT. Inputs are declared first.
| Leaf you hold | Into the netlist | Back out |
| --- | --- | --- |
| Point leaf (subtractive) | `b2y` | `y2b` |
| Comparison leaf (additive) | `a2y` | `y2a` |
| `fss_share` | `fss2y` | `y2fss` |
| Replicated, parties 0 and 1 garble | `rss2y` | `y2rss` |
Bits are least-significant first, one byte each, 0 or 1. `width` 0 means
the whole ring on the way in, and the vector length on the way out.
Both parties' shares are arguments. The edaBit open of `x - r` and the
daBit open of `b ⊕ r` are resolved inside the call, the same way
`edabit::a2b_gmw_pair` does. Those masked values are uniform. The integer
stays shared.
`rss2y` does not wake party 2. Party 0 already holds `x0` and `x1`. Party 1
already holds `x2`. `r0.next` must equal `r1.own`. Top-level `dpf::rss2y`
and `dpf::y2rss` are the local (3,3) casts and are different functions.
```cpp
auto [k0, k1] = dpf::make_dpf(std::uint8_t{42}, std::uint32_t{0x6b});
auto s0 = *dpf::eval_point(k0, std::uint8_t{42});
auto s1 = *dpf::eval_point(k1, std::uint8_t{42});
auto [y0, y1] = dpf::yao::b2y(s0, s1, 8);
dpf::yao::netlist n;
dpf::yao::bit in[8];
for (int i = 0; i < 8; ++i)
in[i] = n.shared_in();
n.out(n.and_(in[0], in[1]));
dpf::yao::session garbler;
auto out0 = garbler.eval(0, n, y0.data(), link); // party 1 passes y1
auto [z0, z1] = dpf::yao::y2b<std::uint32_t>(out0, out1, 1);
// reconstruct(z0, z1) == (0x6b & 1) & ((0x6b >> 1) & 1)
```
`session` keeps the IKNP base OT. The first `eval` that needs a choice
label runs Chou–Orlandi. Later evals on that session only extend. Do not
interleave `iknp::sample` on the same channel. Party 0 is the garbler for
the life of the session. Tables are one-time. The model is semi-honest.
## What stays a key {#yao_not}
A public query against a secret point is a comparison key. An interval is
an interval key. A polynomial in a public offset is Grotto, one comparison
and a local dot. A product of two leaf shares is a Beaver triple. A word
mux is one bit×ring inject. `cost_pass` still chooses among a DCF mask, an
edaBit MSB, and a full adder for those.
Garble when the AND depth is the cost and the circuit is this shape: the
MMO block (5120 ANDs, one message, about 160 KiB of tables), the AES-128
block under a shared key (6400 ANDs, key schedule included), or a netlist
you built because the leaf bits are the input. The Boyar–Peralta S-box
inside those blocks is 32 ANDs. `correction_level` is that hash as one garble: eight MMO blocks, lanes
`0..3` on each seed, 40960 ANDs, XOR shares of the four-block digest.
The level and the prefix share are packed into the inputs
(`pack_correction_level`); the netlist itself does not change per level.
`party/oblivious_hash.hpp` still walks the same S-box as GMW AND layers,
80 opens per level.
## A secret branch {#yao_stack}
The netlist above is a straight line. A secret if/else, or a secret choice
among a few blocks, still belongs on that leaf circuit: the bits came from
`b2y` / `a2y` / `rss2y`, and the answer goes back with `y2b` / `y2a` /
`y2rss`. It is not a reason to open the leaf.
`yao::eval_if` stacks the two branches
([Heath and Kolesnikov, CRYPTO 2020](@ref bib_stacked)). Each branch is
garbled from the hash of the control label for that semantic bit. The
generator XORs the materials. The evaluator rebuilds the inactive branch
from a seed under the control label and XORs it out. Transmitted AND rows
follow the heavier branch, plus four translation rows per output bit so the
garbler's share does not depend on the branch.
`yao::eval_one_hot` is the same stack for `k` netlists, `k` from 2 to 8
([Heath and Kolesnikov, CCS 2021](@ref bib_onehot)). The index is
`index_p0 XOR index_p1`. The demux row for the selector color carries the
inactive seeds.
Both parties pass the same netlists. Branch inputs use the same layout as
`eval_pair`: a shared entry is that party's XOR share, `priv0` is read on
party 0, `priv1` on party 1. The reconstructed bit is `share0[i] XOR share1[i]`.
`stack_blocks` is the stacked material. `naive_blocks` is the sum of the
branches. A comparison or a public-offset polynomial still stays on the key.
## Calls {#yao_calls}
| Call | What it does |
| --- | --- |
| `yao::netlist` | XOR, AND, XNOR, NOT, XOR with a public bit. `n_and()` is the row count |
| `yao::eval_plain` | The same wires in the clear |
| `yao::eval_local` / `eval_pair` | Garble and evaluate in one process |
| `yao::session::eval` | Garble on the peer channel. Party 0 sends the tables |
| `yao::a2y` `b2y` `fss2y` `rss2y` | Ring share to LSB-first XOR bits |
| `yao::y2a` `y2b` `y2fss` `y2rss` | Those bits back to a ring share |
| `yao::aes_mmo` | One zero-key MMO block on a session. `pos` is public |
| `yao::correction_level` | Eight of those blocks: the correction-seed hash for one tree level |
| `yao::aes128` | AES-128 of a shared block under a shared key |
| `yao::eval_if` | Stacked if/else. Rows follow the heavier branch ([CRYPTO 2020](@ref bib_stacked)) |
| `yao::eval_one_hot` | One stack over `k` branches ([CCS 2021](@ref bib_onehot)) |
**Go deeper:** [a secret branch](@ref yao_stack), [yao.hpp](@ref dpf/yao.hpp),
[yao_stack.hpp](@ref dpf/yao_stack.hpp), [yao_share.hpp](@ref dpf/yao_share.hpp),
[yao_aes.hpp](@ref dpf/yao_aes.hpp), [F_Yao](@ref yao.hpp),
[F_YaoShare](@ref yao_share.hpp), [half-gates](@ref bib_halfgates),
[edaBits](@ref bib_edabits). The walk that still uses GMW for the hash is
[the dealer-free tour](@ref tour_ds).

View file

@ -386,6 +386,8 @@ details.type-note > p:first-of-type {
justify-content: space-between;
align-items: center;
gap: 1rem;
width: 100%;
box-sizing: border-box;
margin: 0 0 1.1rem;
padding: 0.45rem 0 0.7rem;
border-bottom: 1px solid var(--separator-color);
@ -416,3 +418,28 @@ pre.fragment {
background: var(--fragment-background);
border: 1px solid var(--separator-color);
}
.eli5, .tldr {
margin: 0.85rem 0 1rem;
padding: 0.7rem 0.95rem;
border-radius: 8px;
line-height: 1.45;
}
.eli5 {
border-left: 3px solid var(--primary-color);
background: var(--fragment-background);
}
.tldr {
border-top: 1px solid var(--separator-color);
background: var(--page-background-color);
}
.eli5 b, .tldr b {
display: inline-block;
margin-right: 0.4rem;
color: var(--primary-color);
font-size: 0.75rem;
letter-spacing: 0.06em;
}

View file

@ -0,0 +1,139 @@
# Stream-array / network layer report
Interactive MPC/FSS that is prep plus message rounds runs on
`net::async_stream_array` (in-process memory, unix sockets, TCP mux, parallel
TCP, SCTP) through `async_round_sink`, `compose_async`, and the N-party runner.
Every decision the network layer makes is an explicit, per-object setting; the
library reads no environment variables.
## Configuration
| Setting | Type / where |
|---------|--------------|
| Wire policy: window, max frame, chunk size, coalescing (frames and bytes), inbox compaction, socket options | `net::wire_policy` (`net/policy.hpp`), passed to every backend constructor; `set_window_bytes` at run time |
| Socket options: `TCP_NODELAY`, quickack, keepalive (idle/interval/count), `SO_SNDBUF`/`SO_RCVBUF`, `SCTP_NODELAY` | `net::socket_options` inside `wire_policy` |
| Setup deadlines: join, connect, accept, handshake, drain | `net::deadlines` |
| Lanes and framing | `drive_options::n_lanes` (`0` = 8, `lanes_one_per_round`), `framing_mode {automatic, always, never}` |
| Wait budgets | `drive_options::wait_timeout` per round wait, `edge_timeout[edge]`, `round_timeout[round]`; the clock restarts whenever an instance advances, so a long healthy run never times out |
| Kernels | `drive_options::workers` (compute pool) and `pump` (the sink's `io_context`, serviced while a kernel runs) |
| Harness | `app::run_config`: transport, host, lanes, framing, instances, wire policy, pipeline credit, wait budget, compute threads, warmup, trials, CPU pins, deadlines. `from_env()` (`DPF_<KEY>`) and `apply_args()` (`--key=value`) share one parser; unknown keys and bad values throw |
`pipeline_credit` only lets a session submit rounds whose bytes do not depend on
the peer, and only while that edge's window has room (`RoundSink::can_send_ahead`).
Compose plans are fully dependent; overlap them with `instances`.
## Backends
| Backend | Notes |
|---------|-------|
| `async_dual_memory_hub` / `make_async_dual_memory_stream_pair` | One `io_context` per side; per-writer window; closing a side still delivers accepted writes, then EOF |
| `async_mux_stream_array` | One TCP socket; writes are chunked and round-robined across lanes so a small write is not queued behind a large one; up to `coalesce_frames` / `coalesce_bytes` per syscall; reads land directly in a waiting reader's buffer |
| `basic_async_parallel_stream_array` (TCP and unix) | One socket per lane with its own strand, queue, and window (`lane_window_bytes`); `accept_parallel_tcp` / `connect_parallel_tcp` use one port and a lane-index handshake |
| `async_sctp_stream_array` (Linux + libsctp) | Lane `i` = SCTP stream `i`; per-stream chunked round-robin sends straight from the owned payload; window; `sctp_available()` checks support up front |
| `mux_stream_array` (synchronous) | Same wire format as the async mux (interoperates with it); a poll-based pump drains reads while writing, so two large cross flushes do not deadlock |
All backends report `stream_stats` (wire bytes including headers, payload
bytes, frames, write calls, buffered/unread bytes, last activity, error).
`close()` aborts; destruction is graceful and drains accepted writes.
## Round sink
`async_round_sink(streams, slots, instances, sink_options)`:
* Plan-shape hello (rounds, slot widths, lanes, instances, framing, epoch); a
mismatch fails both ends with a message naming the field.
* Framed lanes carry `{round, nbytes}`; partial instance prefixes are ready as
they land. Unframed lanes carry one round each.
* `flush_round` waits while the lane's window is full, up to `drain_timeout`.
* `sink_options::reconnect` supplies a replacement link after a transport error
(for example `party_session::reconnector(peer)`). Both ends exchange what
they received and resend the rest, so a plan resumes mid-run
(`sink_stats::resumes`, `resent_bytes`).
* Separate out and in links (`async_round_sink(out, in, ...)`) for the RSS ring.
## Sessions and multi-party runs
| Surface | Where |
|---------|-------|
| `party_session`: static `host:port` table (any start order), connect with retry to a deadline, per-edge transport (mux / parallel / SCTP) and wire policy, handshake that names mismatches, per-edge stats and epochs, `reconnect` / `reconnector` | `net/party_session.hpp` |
| Dealer link in its own failure domain with its own epoch, policy, and `io_context`; `dealer_session` serves several parties | `net/party_session.hpp` |
| `run_parties` (2 or 3 parties, dealer and RSS-ring edges) on every transport; `run_two_party` / `run_three_party` | `party_runner.hpp`, `launch.hpp` |
| One process per party: `parse_node_args` (`--party=i --peers=host:port,...` plus any `run_config` key) and `run_node` | `party_runner.hpp`, `examples/protocol/party_node.cpp` |
| Prep over a dealer session on the configured transport | `session::ship_prep(demand, cfg)` |
## Link security
Party links (party to party, and the dealer) run TLS 1.3 through OpenSSL on
every socket edge unless `encryption=off`. Each party has a raw Ed25519 key
(`identity = FILE`, made with `examples/tools/dpf_keygen.cpp`; 44 characters of
base64, like a WireGuard key). There are no certificate files and no CA: the
certificate TLS needs is generated in memory from the key, and peers check the
key inside it.
| Configured | Effect |
|------------|--------|
| no keys at all | links are encrypted to a fresh key per run; nobody is authenticated; each side logs `security.no_identity` and `security.unauthenticated` |
| `peer.N = KEY` on party M | M authenticates N; N still does not authenticate M unless it holds M's key |
| a held key that does not match | the link fails at setup with both keys in the message |
| `dealer_key = KEY` | the party authenticates the dealer (the dealer uses `peer.N` for parties) |
| SCTP edge while encrypted | refused at setup (no TLS over SCTP streams); `encryption=off` allows it |
The session hello (party id, epoch, transport, lanes) travels inside TLS and
carries whether the sender authenticated the receiver, so every `link.up` record
shows `encryption=TLSv1.3/<cipher>`, `auth=key|none`, `peer_key`, and
`peer_verified_us`.
Client links (`dpf/net/client_link.hpp`): a `client_listener` presents
`server_cert`/`server_key` (PEM, CA-issued), or `server_identity`, or, with
neither, the built-in development certificate. `connect_server` always verifies
the server unless `client_verify=off`: a pinned key (`client_pin`), a CA chain
for the host name (`client_ca = FILE|system`, `client_server_name`), or, when
neither is configured, the development certificate. That certificate's private
key is public, so the default pairing works out of the box and both ends log
that it provides no security. A server may check client keys
(`server_client_pin`, with `client_identity` on the client).
All of it is `run_config` keys, as flags, `DPF_*` variables, or a file of
`key = value` lines (`--config=FILE`, `DPF_CONFIG`; relative paths resolve
against the file):
```text
# p0.conf
identity = p0.key
peer.1 = file:p1.pub
dealer_key = 8s3b...=
transport = mux
```
## Harness
`app::exercise_parties(plans, inputs, kernels, ex, cfg)` runs each party's plan
from fresh inputs for `warmup` untimed and `trials` timed repetitions and
records the median. `experiment` writes `config.csv` (every `run_config`
setting), `trials.csv`, and `wire.csv` (party 0's link counters) next to the
existing tables. `experiment_bench` takes every setting as a flag.
## Remaining
1. QUIC: fits the `async_stream_array` interface; not implemented.
2. Synchronous entry points (`tcp_pair`, `tcp_pair_mux`, `join_*_tcp_mesh`, `trio`) go through the same `peer_security` / `party_session` framework as async edges (TLS 1.3 by default). Sync mux is a blocking face over `async_stream_array`.
3. SCTP links cannot be encrypted.
4. The synchronous `sctp_stream_array` face still throws; use `async_sctp_stream_array`.
5. Non-Linux SCTP is not supported; constructors throw and `sctp_available()` is false.
## Verification
`net_control_test` (27 tests) covers the controls above: config parsing, framing
modes, hello mismatches, partial prefixes, window gating, per-edge and per-round
budgets, a healthy run longer than its wait budget, mux round-robin and graceful
close, per-lane parallel windows, sync/async mux interop, connect deadlines,
static tables, transport mismatch, parallel and SCTP edges, mid-plan reconnect,
dealer epochs, every transport through `run_parties`, a 3-party dealer and ring
plan, separate processes from a static table, harness trials and CSVs, and prep
over a dealer session. `share_runtime_test` (91 tests) covers the rest of the stack.
`security_test` covers key files, TLS in every authentication combination, a
recording relay that sees only ciphertext, wrong keys, mixed encryption
settings, the SCTP refusal, missing-key logging, dealer keys, every client-link
mode (development default, pins, CA chain and host name, verify off, client
keys), config files, and two processes that authenticate each other from key
files.

View file

@ -0,0 +1,113 @@
#include <array>
#include <cstddef>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// BitMore, the DPF query only (Hafiz and Henry, PoPETs 2019 §5.2).
// ell = 2^L servers. The client samples L independent 1-bit DPFs at the
// secret row. Server j, whose label bits are j_{L-1} ... j_0, receives
// key j_e of DPF e and expands it. Concatenating those bits per row is
// the query string the information-theoretic response then consumes.
//
// `dpf::pack_bit_columns(keys...)` runs the full-domain bit walk once per
// key and writes lane e = key e into one integer per row, so the server
// loop reads `symbol[row]` instead of unpacking one int per bit.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/bitmore.cpp
namespace
{
constexpr int bits_l = 2;
constexpr int nservers = 1 << bits_l;
constexpr std::size_t nrows = 256;
} // namespace
int main()
{
constexpr std::uint8_t alpha = 42;
// L independent 1-bit DPFs at the secret row; keep both parties' keys.
auto [e0k0, e0k1] = dpf::make_dpf(alpha, dpf::bit::one);
auto [e1k0, e1k1] = dpf::make_dpf(alpha, dpf::bit::one);
// Server j reads key (j>>e)&1 of DPF e. Pack those L bits per row into
// one digit with pack_bit_columns; lane e is DPF e.
std::array<std::vector<std::uint64_t>, nservers> symbol{};
symbol[0] = dpf::pack_bit_columns(e0k0, e1k0); // parties (0,0)
symbol[1] = dpf::pack_bit_columns(e0k1, e1k0); // parties (1,0)
symbol[2] = dpf::pack_bit_columns(e0k0, e1k1); // parties (0,1)
symbol[3] = dpf::pack_bit_columns(e0k1, e1k1); // parties (1,1)
std::array<std::uint64_t, nservers> at_alpha{};
for (std::size_t row = 0; row < nrows; ++row)
{
if (row == alpha)
{
for (int j = 0; j < nservers; ++j)
at_alpha[static_cast<std::size_t>(j)] =
symbol[static_cast<std::size_t>(j)][row];
continue;
}
for (int j = 1; j < nservers; ++j)
{
if (symbol[static_cast<std::size_t>(j)][row]
!= symbol[0][row])
{
std::cerr << "bitmore off-row\n";
return 1;
}
}
}
// On the secret row the server digits are a translate of the server
// ids: symbol(j) = symbol(0) XOR j. That is a permutation of 0 .. ell-1.
for (int j = 0; j < nservers; ++j)
{
const std::uint64_t expect = at_alpha[0] ^ static_cast<std::uint64_t>(j);
if (at_alpha[static_cast<std::size_t>(j)] != expect)
{
std::cerr << "bitmore secret row\n";
return 1;
}
}
// L = 1 is the 2-server member of the same family: XOR the rows each
// party's bit selects, read off the packed digit's low bit.
std::vector<std::uint64_t> records(nrows);
records[alpha] = 99;
records[7] = 3;
auto [q0, q1] = dpf::make_dpf(alpha, dpf::bit::one);
const auto s0 = dpf::pack_bit_columns(q0);
const auto s1 = dpf::pack_bit_columns(q1);
std::uint64_t a0 = 0;
std::uint64_t a1 = 0;
for (std::size_t i = 0; i < nrows; ++i)
{
if (s0[i] & 1u)
a0 ^= records[i];
if (s1[i] & 1u)
a1 ^= records[i];
}
if ((a0 ^ a1) != records[alpha])
{
std::cerr << "bitmore two-server\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("bitmore",
dpf::protocol::bitmore_fan_plan(0, 4, 6), 6))
return rc;
}
std::cout << (a0 ^ a1) << "\n";
return 0;
}

View file

@ -0,0 +1,79 @@
#include <cstddef>
#include <cstdint>
#include <iostream>
#include <type_traits>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// 3-party Duoram, the DPF steps only (Vadapalli, Henry, Goldberg, USENIX
// Security 2023). Online update: expand with `leaf_later`, rotate value and
// control together, then `apply_leaf_correction` once F is known.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/duoram3.cpp
namespace
{
constexpr std::size_t n = 256;
using output_t = simde_uint128;
template <typename Key>
output_t leaf_cw_of(const Key & key)
{
using exterior = typename Key::exterior_node;
return dpf::extract_leaf<exterior, output_t>(key.template leaf<0>(), 0);
}
} // namespace
int main()
{
constexpr std::uint8_t r = 10;
constexpr std::uint8_t i_star = 42;
constexpr unsigned shift = static_cast<unsigned>(i_star - r);
const output_t message = output_t{7};
std::vector<output_t> memory(n);
memory[i_star] = output_t{100};
memory[r] = output_t{5};
auto [u0, u1] = dpf::make_dpf(r, output_t{1});
const auto read = dpf::reconstruct(
dpf::eval_full_inner_product(dpf::paired, u0, memory, dpf::rotate{shift}),
dpf::eval_full_inner_product(dpf::paired, u1, memory, dpf::rotate{shift}));
if (read != memory[i_star])
{
std::cerr << "duoram read\n";
return 1;
}
auto [w0, w1] = dpf::make_dpf(r, message);
const output_t F = leaf_cw_of(w0);
std::vector<output_t> d0 = memory;
std::vector<output_t> d1(n);
std::vector<std::uint8_t> t0(n), t1(n);
dpf::eval_full_add_into(d0, t0, w0, dpf::leaf_later{}, dpf::rotate{shift});
dpf::eval_full_add_into(d1, t1, w1, dpf::leaf_later{}, dpf::rotate{shift});
dpf::apply_leaf_correction(d0, t0, F);
dpf::apply_leaf_correction(d1, t1, F);
if ((d0[i_star] - d1[i_star]) != memory[i_star] + message
|| (d0[r] - d1[r]) != memory[r])
{
std::cerr << "duoram update\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("duoram3",
dpf::protocol::duoram_update_plan(0), 8))
return rc;
}
std::cout << static_cast<unsigned long long>(d0[i_star] - d1[i_star]) << "\n";
return 0;
}

View file

@ -0,0 +1,222 @@
#include <algorithm>
#include <cstdint>
#include <cstdlib>
#include <iostream>
#include <stdexcept>
#include <string>
#include <utility>
#include <vector>
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
#include "dpf/bench_cells.hpp"
#include "dpf/experiment.hpp"
#include "dpf/party_runner.hpp"
#include "dpf/run_log.hpp"
// Every cell is a compose plan driven by `run_parties` on one `run_config`:
// the party mesh over in-process async memory, unix sockets, TCP mux, parallel
// TCP, or SCTP. `memory` and `stream` are the older paired sinks; this harness
// uses the mesh, so those two names run as async. Every `run_config` key is a
// flag (`--transport=mux --lanes=4 --window=262144 --warmup=2 --trials=9`) or
// the matching `DPF_*` variable; flags win. Lanes default to 1 here. Each cell
// runs `warmup` untimed and `trials` timed repetitions and records every
// party's trial times, party 0's and the slowest party's medians, the
// configuration, and party 0's wire counters in the CSVs. Every repetition
// draws from streams derived from the cell's master, so replaying the master
// replays the cell.
//
// c++ -std=c++17 -march=native -pthread -I include -I thirdparty \
// examples/applications/experiment_bench.cpp -lsctp
// DPF_EXPERIMENT_DIR=/tmp/libdpf_bench ./a.out --transport=mux --trials=5
//
// The first block is the DPF plans. The second block is the work that is not
// a key: word gadgets, stacked branches, short tables, and a hidden reorder
// of a column the parties already share.
namespace
{
dpf::app::run_config mesh_config(int argc, char ** argv, std::string & replaced)
{
dpf::app::run_config cfg;
cfg.n_lanes = 1;
cfg.merge_env();
const auto rest = cfg.apply_args(argc, argv);
if (!rest.empty())
throw std::invalid_argument("unexpected argument '" + rest.front()
+ "' (flags are --key=value)");
if (cfg.kind == dpf::net::transport::memory_sink
|| cfg.kind == dpf::net::transport::memory_stream)
{
std::cout << "transport "
<< dpf::net::transport_name(cfg.kind)
<< " is a paired sink; the battery uses the party mesh (async)\n";
replaced = dpf::net::transport_name(cfg.kind);
cfg.kind = dpf::net::transport::async_memory;
}
return cfg;
}
int measure_plans(const char * name, std::vector<dpf::protocol::plan> plans,
std::uint64_t run_id, const std::string & dir, const dpf::app::run_config & cfg,
dpf::protocol::cell_fn cell)
{
if (plans.empty() || plans[0].rounds() == 0)
{
std::cerr << name << " has no exchange rounds\n";
return 1;
}
dpf::experiment ex(name, "p0");
ex.set_run_id(run_id);
ex.ingest_plan(plans[0]);
ex.set_config(cfg.describe());
dpf::app::parties_result result;
const std::size_t total = cfg.warmup + std::max<std::size_t>(1, cfg.trials);
try
{
for (std::size_t t = 0; t < total; ++t)
{
std::vector<dpf::app::party_values> values(plans.size());
const bool last = t + 1 == total;
result = dpf::app::run_parties(plans, values, {}, cfg,
last ? &ex : nullptr, cell, &ex);
if (t >= cfg.warmup)
ex.add_trial(result.party0_wall_ns, result.party_wall_ns);
}
}
catch (const std::exception & err)
{
std::cerr << name << " flow: " << err.what() << "\n";
return 1;
}
const auto wire = result.wire.empty() ? dpf::net::stream_stats{} : result.wire[0];
dpf::experiment::wire_counts w;
w.bytes_out = wire.bytes_out;
w.bytes_in = wire.bytes_in;
w.payload_out = wire.payload_out;
w.payload_in = wire.payload_in;
w.frames_out = wire.frames_out;
w.frames_in = wire.frames_in;
w.write_calls = wire.write_calls;
ex.set_wire(w);
ex.write_csv(dir);
std::cout << name << " parties=" << plans.size()
<< " run_id=" << run_id
<< " rounds=" << ex.interactive_rounds()
<< " bytes=" << ex.plan_bytes_out()
<< " wire_out=" << wire.bytes_out
<< " wire_in=" << wire.bytes_in
<< " payload_out=" << wire.payload_out
<< " median_ns=" << ex.median_trial_ns()
<< " slowest_median_ns=" << ex.slowest_median_ns()
<< " trials=" << ex.trials().size()
<< " prg_evals=" << ex.prg_evals()
<< " seed=" << ex.seed_hex() << "\n";
return 0;
}
} // namespace
int main(int argc, char ** argv)
{
const char * env = std::getenv("DPF_EXPERIMENT_DIR");
const std::string dir = (env && env[0] != '\0') ? env
: "/tmp/libdpf_experiment_bench";
dpf::app::run_config cfg;
std::string replaced;
try
{
cfg = mesh_config(argc, argv, replaced);
dpf::app::start_logging(cfg);
}
catch (const std::exception & err)
{
std::cerr << "experiment_bench: " << err.what() << "\n";
return 2;
}
if (!replaced.empty())
DPF_LOG(warning, "config.override").kv("key", "transport")
.kv("requested", replaced).kv("used", "async")
.kv("detail", "paired sinks cannot carry the party mesh");
std::cout << cfg.summary() << "\n";
struct named
{
const char * name;
int parties;
dpf::protocol::plan (*make)(std::size_t party);
};
const named dpf_plans[] = {
{"keyword_pir", 2, [](std::size_t p) {
return dpf::protocol::keyword_pir_compose_plan(p, 8);
}},
{"express", 2, [](std::size_t p) {
return dpf::protocol::mailbox_write_fused_plan(p);
}},
{"subleq", 2, [](std::size_t p) {
return dpf::protocol::subleq_instruction_plan(p);
}},
{"pika", 2, [](std::size_t p) {
return dpf::protocol::pika_lookup_plan(p);
}},
{"duoram3", 2, [](std::size_t p) {
return dpf::protocol::duoram_update_plan(p);
}},
{"poplar", 2, [](std::size_t p) {
return dpf::protocol::poplar_prefix_plan(p);
}},
{"poplar_fan4", 2, [](std::size_t p) {
return dpf::protocol::poplar_prefix_fan_plan(p, 4);
}},
{"ledger23", 2, [](std::size_t p) {
return dpf::protocol::ledger23_append_plan(p);
}},
{"bitmore", 2, [](std::size_t p) {
return dpf::protocol::bitmore_fan_plan(p);
}},
{"floram", 2, [](std::size_t p) {
return dpf::protocol::floram_ds_plan(p);
}},
{"fss_point", 2, [](std::size_t p) {
return dpf::protocol::fss_point_plan(p);
}},
{"fss_cmp", 2, [](std::size_t p) {
return dpf::protocol::fss_cmp_plan(p);
}},
{"range_count", 2, [](std::size_t p) {
return dpf::protocol::range_count_plan(p);
}},
{"psi_cuckoo", 2, [](std::size_t p) {
return dpf::protocol::psi_cuckoo_plan(p, {0, 2, 5, 7});
}},
{"idpf_agg", 2, [](std::size_t p) {
return dpf::protocol::idpf_agg_plan(p, 8);
}},
};
std::uint64_t run_id = 0;
for (const auto & row : dpf_plans)
{
std::vector<dpf::protocol::plan> plans;
plans.reserve(static_cast<std::size_t>(row.parties));
for (int p = 0; p < row.parties; ++p)
plans.push_back(row.make(static_cast<std::size_t>(p)));
if (int rc = measure_plans(row.name, std::move(plans), run_id++, dir, cfg,
nullptr))
return rc;
}
for (const auto & cell : dpf::bench::battery())
{
std::vector<dpf::protocol::plan> plans;
plans.reserve(static_cast<std::size_t>(cell.parties));
for (int p = 0; p < cell.parties; ++p)
plans.push_back(dpf::bench::plan_for(static_cast<std::size_t>(p), cell.id));
if (int rc = measure_plans(cell.name, std::move(plans), run_id++, dir, cfg,
&dpf::bench::run_cell))
return rc;
}
std::cout << "wrote CSVs under " << dir << "\n";
return 0;
}

View file

@ -0,0 +1,84 @@
#include <cstddef>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Express, the mailbox write (Eskandarian, Corrigan-Gibbs, Zaharia, Boneh,
// USENIX Security 2021 §3.1). Two servers hold XOR shares of every mailbox
// row. The client sends one `blob` DPF key each. Each server adds its
// expansion into its share and folds the one-hot audit in the same walk.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/express.cpp
namespace
{
constexpr std::size_t nboxes = 256;
constexpr std::size_t row_bytes = 100;
using row_t = dpf::blob<row_bytes>;
} // namespace
int main()
{
constexpr std::uint8_t address = 9;
row_t message{};
for (std::size_t i = 0; i < row_bytes; ++i)
message.bytes[i] = static_cast<unsigned char>(i + 1);
auto [k0, k1] = dpf::make_dpf(address, message);
std::vector<row_t> box0(nboxes), box1(nboxes);
// One-pass caller fold: count how many non-zero shares each server sees.
std::size_t hot0 = 0, hot1 = 0;
dpf::eval_full_add_into(box0, k0, [&](std::size_t, const row_t & s) {
if (s != row_t{})
++hot0;
});
dpf::eval_full_add_into(box1, k1, [&](std::size_t, const row_t & s) {
if (s != row_t{})
++hot1;
});
(void)hot0;
(void)hot1;
if ((box0[address] ^ box1[address]) != message)
{
std::cerr << "express mailbox\n";
return 1;
}
if ((box0[0] ^ box1[0]) != row_t{})
{
std::cerr << "express neighbor\n";
return 1;
}
// fp61 one-hot audit on a parallel extractable key (same walk shape).
auto [a0, a1] = dpf::make_dpf(address, dpf::fp61{1}, dpf::extractable{});
std::vector<dpf::fp61> challenge(nboxes);
for (std::size_t i = 0; i < nboxes; ++i)
challenge[i] = dpf::fp61{static_cast<std::uint64_t>(i + 1)};
std::vector<dpf::fp61> audit0(nboxes), audit1(nboxes);
dpf::sketch_share s0{}, s1{};
dpf::eval_full_add_into(audit0, a0, dpf::sketch(s0, challenge));
dpf::eval_full_add_into(audit1, a1, dpf::sketch(s1, challenge));
if (!dpf::sketch_verify(s0, s1))
{
std::cerr << "express audit\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("express",
dpf::protocol::mailbox_write_fused_plan(0, 8), 8))
return rc;
}
std::cout << static_cast<unsigned>(message.bytes[0]) << "\n";
return 0;
}

View file

@ -0,0 +1,61 @@
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Floram, the FSS read and write (Doerner and shelat, CCS 2017). Both
// parties see the memory. The address is secret-shared, so keygen is
// Doerner–Shelat rather than a dealer who knows the index. The read is
// the inner product of a unit key with that memory. The write adds a
// payload key, built from the same address shares, into subtractive
// copies of the array.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/floram.cpp
int main()
{
constexpr std::size_t n = 256;
constexpr std::uint8_t address = 42;
constexpr std::uint8_t a0 = 0x15;
constexpr std::uint8_t a1 = static_cast<std::uint8_t>(address ^ a0);
constexpr std::uint64_t message = 9;
std::vector<std::uint64_t> memory(n);
memory[address] = 100;
memory[7] = 3;
auto [r0, r1] = dpf::make_dpf_doerner_shelat(a0, a1, std::uint64_t{1});
const auto word = dpf::reconstruct(
dpf::eval_full_inner_product(dpf::paired, r0, memory),
dpf::eval_full_inner_product(dpf::paired, r1, memory));
if (word != memory[address])
{
std::cerr << "floram read\n";
return 1;
}
auto [w0, w1] = dpf::make_dpf_doerner_shelat(a0, a1, message);
std::vector<std::uint64_t> s0 = memory;
std::vector<std::uint64_t> s1(n);
dpf::eval_full_add_into(s0, w0);
dpf::eval_full_add_into(s1, w1);
if (s0[address] - s1[address] != memory[address] + message
|| s0[7] - s1[7] != memory[7])
{
std::cerr << "floram write\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("floram",
dpf::protocol::floram_ds_plan(0), 40))
return rc;
}
std::cout << word << "\n";
return 0;
}

View file

@ -0,0 +1,38 @@
#include <cstdint>
#include <iostream>
#include "dpf/online_session.hpp"
#include "dpf/prep_source.hpp"
// Hushmap KHM ADD: dealer tape + two online opens on async round sinks,
// then a real prep shipment (deal_views over async streams).
//
// c++ -std=c++17 -pthread -I include -I thirdparty \
// examples/applications/hushmap_add.cpp
int main()
{
constexpr std::size_t layers = 3;
try
{
dpf::session::drive_hushmap_add(layers);
dpf::prep::demand d;
d.ring_triples = static_cast<std::uint32_t>(layers);
const auto shipped = dpf::session::ship_prep(d);
std::uint8_t a0[8]{}, b0[8]{}, c0[8]{};
std::uint8_t a1[8]{}, b1[8]{}, c1[8]{};
auto c0p = shipped.party0;
auto c1p = shipped.party1;
c0p.take_ring(a0, b0, c0);
c1p.take_ring(a1, b1, c1);
std::cout << "hushmap_add layers=" << layers
<< " rounds=" << (layers + 2)
<< " prep_bytes=" << shipped.bytes0 << "\n";
}
catch (const std::exception & ex)
{
std::cerr << "hushmap_add: " << ex.what() << "\n";
return 1;
}
return 0;
}

View file

@ -0,0 +1,52 @@
#include <algorithm>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Max and k-th order statistic over secret uint16 values. Each value is
// one incremental DPF with a unit payload on every prefix length. Servers
// resume only the live prefixes with eval_until (ePrint 2024/1190).
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/idpf_agg.cpp
int main()
{
const std::vector<std::uint16_t> values{12, 80, 3, 80, 40};
using key0_t = decltype(dpf::make_dpf(std::uint16_t{0},
dpf::idpf_ones<16>()).first);
using key1_t = decltype(dpf::make_dpf(std::uint16_t{0},
dpf::idpf_ones<16>()).second);
std::vector<key0_t> k0;
std::vector<key1_t> k1;
for (auto v : values)
{
auto [a, b] = dpf::make_dpf(v, dpf::idpf_ones<16>());
k0.push_back(std::move(a));
k1.push_back(std::move(b));
}
const auto opened_max = dpf::idpf_agg_max(k0, k1);
const auto opened_k2 = dpf::idpf_agg_kth(k0, k1, 2);
auto sorted = values;
std::sort(sorted.begin(), sorted.end(), std::greater<>{});
if (opened_max != sorted[0] || opened_k2 != sorted[1])
{
std::cerr << "idpf_agg " << opened_max << " " << opened_k2 << "\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("idpf_agg",
dpf::protocol::idpf_agg_plan(0, 16), 16))
return rc;
}
std::cout << opened_max << "\n";
return 0;
}

View file

@ -0,0 +1,49 @@
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Three-server index PIR with the information-theoretic DPF
// (ePrint 2023/028). The database is public and replicated. Each server
// dots its additive share with the table; the three dots sum to the record.
// make_dpf3 remains the computational (2,3) Shamir key.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/it_pir3.cpp
int main()
{
constexpr std::size_t n = 256;
constexpr std::uint8_t index = 42;
std::vector<std::uint64_t> database(n);
for (std::size_t i = 0; i < n; ++i)
database[i] = i * i + 1;
auto [k0, k1, k2] = dpf::make_it_dpf3(index, 1);
const auto s0 = dpf::eval_it_dpf3_inner_product(k0, database);
const auto s1 = dpf::eval_it_dpf3_inner_product(k1, database);
const auto s2 = dpf::eval_it_dpf3_inner_product(k2, database);
const auto opened = s0 + s1 + s2;
if (opened != database[index])
{
std::cerr << "it_pir3\n";
return 1;
}
{
constexpr std::size_t query_bytes =
dpf::it_dpf3_key::domain_size * sizeof(std::uint64_t);
if (int rc = dpf::app::run_measured("it_pir3",
dpf::protocol::n_server_pir_plan(0, 3, query_bytes,
sizeof(std::uint64_t)),
2))
return rc;
}
std::cout << opened << "\n";
return 0;
}

View file

@ -0,0 +1,60 @@
#include <cstdint>
#include <iostream>
#include <string>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Two-server keyword PIR, the DPF step (Gilboa and Ishai, EUROCRYPT 2014).
// `eval_sequence_xor` folds the selected records into one XOR accumulator
// without materializing a bit vector.
//
// c++ -std=c++17 -march=native -pthread -I include -I thirdparty \
// examples/applications/keyword_pir.cpp
// DPF_EXPERIMENT_DIR=/tmp/kw ./a.out # optional CSV dump
namespace
{
using keyword = dpf::keyword<3, dpf::alphabets::lowercase_alpha>;
} // namespace
int main()
{
const std::vector<keyword> dict{keyword{"bat"}, keyword{"cat"},
keyword{"dog"}, keyword{"pig"}};
const std::vector<int> records{56, 12, 34, 78};
auto [k0, k1] = dpf::make_dpf(keyword{"bat"}, dpf::bit::one);
const int selected = dpf::eval_sequence_xor(k0, dict.begin(), dict.end(),
records)
^ dpf::eval_sequence_xor(k1, dict.begin(), dict.end(), records);
if (selected != 56)
{
std::cerr << "keyword hit\n";
return 1;
}
auto [m0, m1] = dpf::make_dpf(keyword{"rat"}, dpf::bit::one);
const int missing = dpf::eval_sequence_xor(m0, dict.begin(), dict.end(),
records)
^ dpf::eval_sequence_xor(m1, dict.begin(), dict.end(), records);
if (missing != 0)
{
std::cerr << "keyword miss\n";
return 1;
}
{
constexpr std::size_t depth = dpf::utils::bitlength_of<keyword>::value;
if (int rc = dpf::app::run_measured("keyword_pir",
dpf::protocol::keyword_pir_compose_plan(0, depth), 2))
return rc;
}
std::cout << selected << "\n";
return 0;
}

View file

@ -0,0 +1,85 @@
#include <array>
#include <cstddef>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// A (2,3) ledger, the DPF step. Three servers replicate a ledger as
// Shamir-style (2-of-3) shares (this group's dpf3 / VDPF+ construction). Each
// append writes one point (slot -> amount) into all three shares; any two
// servers reconstruct a slot. A verifiable proof (verify_dpf3) rejects an
// append that is not a single well-formed point before it is applied.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/ledger23.cpp
namespace
{
using dpf::fp61;
constexpr std::size_t nslots = 256; // uint8 slot domain
fp61 open2(fp61 a, fp61 b) // any two of three shares reconstruct
{
return dpf::shamir3::reconstruct(dpf::shamir3::share{1, a},
dpf::shamir3::share{2, b});
}
} // namespace
int main()
{
std::vector<fp61> l1(nslots), l2(nslots), l3(nslots);
// Each append is a verified (2,3) point key.
const std::pair<std::uint8_t, std::uint64_t> entries[] = {
{5, 100}, {40, 25}, {5, 7}};
for (auto [slot, amount] : entries)
{
auto [k1, k2, k3] = dpf::make_dpf3(slot, fp61{amount}, dpf::verifiable{});
if (!dpf::verify_dpf3(dpf::prove_dpf3(k1, slot),
dpf::prove_dpf3(k2, slot), dpf::prove_dpf3(k3, slot)))
{
std::cerr << "ledger append audit\n";
return 1;
}
// Fold the (2,3) expansion into each party's ledger shares.
dpf::eval_full_add_into(l1, k1);
dpf::eval_full_add_into(l2, k2);
dpf::eval_full_add_into(l3, k3);
}
// Slot 5 got two credits (100 + 7); slot 40 got 25; the rest are 0.
if (open2(l1[5], l2[5]) != fp61{107}
|| open2(l1[40], l2[40]) != fp61{25}
|| open2(l1[0], l2[0]).raw() != 0)
{
std::cerr << "ledger balance\n";
return 1;
}
// A proof that does not come from the same append is rejected: mixing one
// party's token from an independent key triple fails verification.
auto [b1, b2, b3] = dpf::make_dpf3(std::uint8_t{9}, fp61{1}, dpf::verifiable{});
auto [c1, c2, c3] = dpf::make_dpf3(std::uint8_t{9}, fp61{1}, dpf::verifiable{});
if (dpf::verify_dpf3(dpf::prove_dpf3(b1, std::uint8_t{9}),
dpf::prove_dpf3(b2, std::uint8_t{9}),
dpf::prove_dpf3(c3, std::uint8_t{9})))
{
std::cerr << "ledger accepted a bad append\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("ledger23",
dpf::protocol::ledger23_append_plan(0), 2))
return rc;
}
std::cout << open2(l1[5], l2[5]).raw() << "\n";
return 0;
}

View file

@ -0,0 +1,98 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
#include "grotto/carry.hpp"
#include "grotto/carry_plan.hpp"
// LLAMA, the FSS gates that touch a DPF (Gupta, Kumaraswamy, Chandran,
// and Gupta, ePrint 2022/793). Width gates call `grotto::sign_extend` and
// `grotto::truncate_reduce`. A degree-0 spline is one interval key per piece.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/llama.cpp
int main()
{
constexpr std::uint8_t r = 20;
constexpr std::uint8_t knot = 16;
const std::uint8_t threshold = static_cast<std::uint8_t>(knot + r);
auto [c0, c1] = dpf::make_dpf(threshold, dpf::gt(std::uint64_t{1}));
const auto above = dpf::reconstruct(
dpf::eval_point(dpf::cmp, c0, static_cast<std::uint8_t>(30 + r)),
dpf::eval_point(dpf::cmp, c1, static_cast<std::uint8_t>(30 + r)));
const auto below = dpf::reconstruct(
dpf::eval_point(dpf::cmp, c0, static_cast<std::uint8_t>(4 + r)),
dpf::eval_point(dpf::cmp, c1, static_cast<std::uint8_t>(4 + r)));
if (above != 1 || below != 0)
{
std::cerr << "llama comparison\n";
return 1;
}
auto [lo0, lo1] = dpf::make_dpf(r, dpf::ic(std::uint8_t{0}, std::uint8_t{15},
std::uint64_t{2}));
auto [hi0, hi1] = dpf::make_dpf(r, dpf::ic(std::uint8_t{16}, std::uint8_t{31},
std::uint64_t{5}));
auto piece = [&](std::uint8_t x) {
const std::uint8_t x_hat = static_cast<std::uint8_t>(x + r);
const auto lo = dpf::reconstruct(
dpf::eval_point(dpf::ic, lo0, x_hat),
dpf::eval_point(dpf::ic, lo1, x_hat));
const auto hi = dpf::reconstruct(
dpf::eval_point(dpf::ic, hi0, x_hat),
dpf::eval_point(dpf::ic, hi1, x_hat));
return lo + hi;
};
if (piece(4) != 2 || piece(20) != 5 || piece(40) != 0)
{
std::cerr << "llama spline\n";
return 1;
}
// Truncate-reduce: drop 3 low bits of an 8-bit opening.
{
auto keys = grotto::make_truncate_reduce_keys(8, 3);
const std::uint64_t x0 = 0x05, x1 = 0x03;
const std::uint64_t opened = (x0 + x1 + keys.rin) & 0xffu;
const auto y0 = grotto::truncate_reduce(keys, 0, opened);
const auto y1 = grotto::truncate_reduce(keys, 1, opened);
const auto got = (y0.value + y1.value) & 0x1fu;
const auto want = grotto::eval_carry_clear(keys.recipe, x0, x1);
if (got != want)
{
std::cerr << "llama truncate_reduce\n";
return 1;
}
}
// Sign-extend: 8 → 16 bits.
{
auto keys = grotto::make_sign_extend_keys(8, 16);
const std::uint64_t x0 = 0x80, x1 = 0;
const std::uint64_t opened = (x0 + x1 + keys.rin) & 0xffu;
const std::uint64_t msb_high = 1; // 0x80 is negative
const auto y0 = grotto::sign_extend(keys, 0, opened, msb_high);
const auto y1 = grotto::sign_extend(keys, 1, opened, msb_high);
const auto got = (y0.value + y1.value) & 0xffffu;
const auto want = grotto::carry_extend_clear(x0, x1, 8, 16);
if (got != want)
{
std::cerr << "llama sign_extend\n";
return 1;
}
}
{
if (int rc = dpf::app::run_measured("llama",
dpf::protocol::range_count_plan(0), 8))
return rc;
}
std::cout << above << " " << piece(20) << "\n";
return 0;
}

View file

@ -0,0 +1,94 @@
#include <cstddef>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Mastic, the DPF step (private weighted heavy-hitters / attribute-based
// metrics). Each client keys an incremental DPF whose payload is its weight
// instead of a plain 1. Servers sum the weighted prefix shares at each depth
// and keep the heavy prefixes. This is Poplar's prefix walk with a weight
// payload; VIDPF path-consistency is `verify_idpf_path` below.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/mastic.cpp
namespace
{
// Weighted counts of all 2^length prefixes, summed over the two clients.
template <typename Tag, typename A0, typename A1, typename Kb0, typename Kb1>
std::vector<std::uint64_t> weighted_level(Tag tag, std::size_t nprefix,
const A0 & a0, const A1 & a1, const Kb0 & b0, const Kb1 & b1)
{
auto [ba0, ia0] = dpf::eval_prefixes(tag, a0);
auto [ba1, ia1] = dpf::eval_prefixes(tag, a1);
auto [bb0, ib0] = dpf::eval_prefixes(tag, b0);
auto [bb1, ib1] = dpf::eval_prefixes(tag, b1);
std::vector<std::uint64_t> w(nprefix);
for (std::size_t p = 0; p < nprefix; ++p)
w[p] = dpf::reconstruct(ba0[p], ba1[p])
+ dpf::reconstruct(bb0[p], bb1[p]);
return w;
}
std::vector<dpf::fp61> path_challenges(std::size_t n)
{
std::vector<dpf::fp61> rs(n);
for (auto & r : rs)
r = dpf::uniform_sample<dpf::fp61>();
return rs;
}
} // namespace
int main()
{
// Two clients report strings 0xA0 and 0xB0 (both begin "101"), with
// weights 5 and 3. A heavy-hitter threshold of 6 should keep prefix 101.
auto [a0, a1] = dpf::make_dpf(std::uint8_t{0xA0},
dpf::idpf(std::uint64_t{5}, std::uint64_t{5}, std::uint64_t{5}));
auto [b0, b1] = dpf::make_dpf(std::uint8_t{0xB0},
dpf::idpf(std::uint64_t{3}, std::uint64_t{3}, std::uint64_t{3}));
// One-time VIDPF path check per client (weight-1 / parent consistency).
const auto rs = path_challenges(dpf::path_sketch_challenge_count(3));
if (!dpf::verify_idpf_path<3>(a0, a1, rs)
|| !dpf::verify_idpf_path<3>(b0, b1, rs))
{
std::cerr << "mastic path sketch\n";
return 1;
}
// Length 1: prefix "1" carries the full weight 8; "0" carries 0.
const auto lvl1 = weighted_level(dpf::out<0, 1>, 2, a0, a1, b0, b1);
if (lvl1[1] != 8 || lvl1[0] != 0)
{
std::cerr << "mastic level1\n";
return 1;
}
// Length 3: prefix 101 (=5) is the heavy hitter with weight 8.
const auto lvl3 = weighted_level(dpf::out<2, 3>, 8, a0, a1, b0, b1);
constexpr std::uint64_t threshold = 6;
std::size_t heavy = 0, nheavy = 0;
for (std::size_t p = 0; p < lvl3.size(); ++p)
if (lvl3[p] >= threshold) { heavy = p; ++nheavy; }
if (nheavy != 1 || heavy != 0b101 || lvl3[0b101] != 8)
{
std::cerr << "mastic heavy\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("mastic",
dpf::protocol::poplar_prefix_plan(0), 8))
return rc;
}
std::cout << lvl3[0b101] << "\n";
return 0;
}

View file

@ -0,0 +1,72 @@
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Pika, the lookup (Wagh, PoPETs 2022, Fig. 1). Party P2 is the dealer.
// P2 keys a unit DPF at a fresh index r and shares r. P0 and P1 open
// x = r - a, and take the inner product of the DPF with the table rotated
// by x. A word payload of 1 reconstructs to +1. A 1-bit payload lifts to
// +1 or -1; the dealer reads that sign off Gen's final control bit.
//
// The rotation is folded into the walk with `dpf::rotate{s}` (no rotated
// copy of the table), and the sign is recorded at keygen with
// `dpf::unit_sign` (no evaluator-side eval_point).
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/pika.cpp
int main()
{
constexpr std::size_t n = 256;
constexpr std::uint8_t r = 50;
constexpr std::uint8_t a0 = 10;
constexpr std::uint8_t a1 = 7;
constexpr std::uint8_t a = static_cast<std::uint8_t>(a0 + a1);
constexpr std::uint8_t x = static_cast<std::uint8_t>(r - a);
std::vector<std::uint64_t> table(n);
for (std::size_t i = 0; i < n; ++i)
table[i] = static_cast<std::uint64_t>(i) * i;
// Lookup: DPF at r dotted with the table read at (i - x) mod n, i.e.
// rotated by s = (n - x) mod n. The walk applies the offset; no copy.
auto [k0, k1] = dpf::make_dpf(r, std::uint64_t{1});
const std::size_t s = (n - static_cast<std::size_t>(x)) % n;
const auto value = dpf::reconstruct(
dpf::eval_full_inner_product(dpf::paired, k0, table, dpf::rotate{s}),
dpf::eval_full_inner_product(dpf::paired, k1, table, dpf::rotate{s}));
if (value != table[a])
{
std::cerr << "pika lookup\n";
return 1;
}
// The early-stop bit leaf. The dealer, who sees both keys, records a
// sign of +1 or -1 at r via `unit_sign`; the evaluators never open r.
int w0 = 0, w1 = 0;
auto [b0, b1] = dpf::make_dpf(r, dpf::bit::one, dpf::unit_sign{w0, w1});
const int sign = w0 - w1;
if (sign != 1 && sign != -1)
{
std::cerr << "pika sign\n";
return 1;
}
if (dpf::reconstruct(*dpf::eval_point(k0, r), *dpf::eval_point(k1, r)) != 1)
{
std::cerr << "pika unit\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("pika",
dpf::protocol::pika_lookup_plan(0, 8, 3), 5))
return rc;
}
std::cout << value << "\n";
return 0;
}

View file

@ -0,0 +1,55 @@
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Three-server index PIR. The database is public and replicated. The
// secret index is one (2,3) point key: each server holds one share and
// dots it with the database. Any two of those dots reconstruct the
// record. This is the library's three-evaluator key (ePrint 2024/1658),
// the same sharing the ledger appends with.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/pir3.cpp
int main()
{
constexpr std::size_t n = 256;
constexpr std::uint8_t index = 42;
std::vector<dpf::fp61> database(n);
for (std::size_t i = 0; i < n; ++i)
database[i] = dpf::fp61{static_cast<std::uint64_t>(i * i + 1)};
auto [k1, k2, k3] = dpf::make_dpf3(index, dpf::fp61{1});
const auto s1 = dpf::eval_full_inner_product(k1, database);
const auto s2 = dpf::eval_full_inner_product(k2, database);
const auto s3 = dpf::eval_full_inner_product(k3, database);
const auto opened = dpf::shamir3::reconstruct(
dpf::as_share(k1, s1), dpf::as_share(k2, s2));
const auto opened_13 = dpf::shamir3::reconstruct(
dpf::as_share(k1, s1), dpf::as_share(k3, s3));
if (opened != database[index] || opened_13 != database[index])
{
std::cerr << "pir3\n";
return 1;
}
{
// Client uploads one (2,3) key (two point-key spines) to each server.
constexpr std::size_t depth = 8;
constexpr std::size_t query_bytes = 2 * (16 + depth * 16);
if (int rc = dpf::app::run_measured("pir3",
dpf::protocol::n_server_pir_plan(0, 3, query_bytes,
sizeof(dpf::fp61)),
2))
return rc;
}
std::cout << opened.raw() << "\n";
return 0;
}

View file

@ -0,0 +1,42 @@
#include <cstdint>
#include <iostream>
#include <memory>
#include <vector>
#include "dpf/online_session.hpp"
// PIRsona BitMore fetch on a split-io async star (L=1 → 2 servers).
//
// c++ -std=c++17 -pthread -I include -I thirdparty \
// examples/applications/pirsona_fetch.cpp
int main()
{
constexpr std::size_t L = 1;
constexpr std::size_t n = 1u << L;
auto seeds = std::make_shared<std::vector<std::vector<std::uint8_t>>>(n);
auto answers = std::make_shared<std::vector<std::vector<std::uint8_t>>>(n);
for (std::size_t i = 0; i < n; ++i)
{
(*seeds)[i].assign(16 * L, static_cast<std::uint8_t>(i + 1));
(*answers)[i].assign(8, static_cast<std::uint8_t>(0x40 + i));
}
try
{
auto client = dpf::protocol::pirsona_bitmore_fetch(L, 16, 8, seeds, answers);
const std::vector<std::size_t> slots{16u * L, 8u};
dpf::session::drive_async_star(n, slots, std::move(client),
[&](std::size_t i) {
return dpf::protocol::star_server_reply_rounds(16 * L, 8,
(*answers)[i]);
});
std::cout << "pirsona_fetch rounds=" << (2 * n) << " servers=" << n
<< "\n";
}
catch (const std::exception & ex)
{
std::cerr << "pirsona_fetch: " << ex.what() << "\n";
return 1;
}
return 0;
}

View file

@ -0,0 +1,97 @@
#include <array>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// PRAC, the DPF steps that are not Duoram (Sasy, Vadapalli, and Goldberg,
// ePrint 2023/1897). Binary search builds the path with `make_dpf` /
// `extend`, one prefix at a time. Heapify uses a wide `vec` leaf.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/prac.cpp
namespace
{
constexpr std::size_t n = 256;
constexpr std::size_t bitlen = 8;
using beta_t = std::uint64_t;
using input_t = std::uint8_t;
using wide3 = dpf::vec<beta_t, 3>;
template <typename Key0, typename Key1>
beta_t open_prefix(const Key0 & k0, const Key1 & k1, std::size_t level,
input_t x)
{
auto one = [&](auto lvl) {
return dpf::reconstruct(
*dpf::eval_point(dpf::out<lvl.value>, k0, x),
*dpf::eval_point(dpf::out<lvl.value>, k1, x));
};
switch (level)
{
case 0: return one(std::integral_constant<std::size_t, 0>{});
case 1: return one(std::integral_constant<std::size_t, 1>{});
default: throw std::logic_error("prac: level");
}
}
} // namespace
int main()
{
constexpr std::array<std::uint64_t, 8> memory{
1, 3, 5, 7, 9, 11, 13, 15};
constexpr std::uint64_t needle = 10;
// Search path bits (MSB first): 1, then 0 → prefix 0b10......
constexpr input_t path = 0x80;
auto [p0, p1] = dpf::make_dpf(path, dpf::at<1>(beta_t{1}));
auto [q0, q1] = dpf::extend(p0, p1, path, dpf::at<2>(beta_t{1}));
constexpr std::uint64_t stride2[] = {memory[1], memory[5]};
constexpr std::uint64_t stride4[] = {memory[0], memory[2], memory[4], memory[6]};
const auto sel1 = open_prefix(q0, q1, 0, path);
const auto sel2 = open_prefix(q0, q1, 1, path);
const auto at_5 = sel1 * stride2[1];
const auto at_4 = sel2 * stride4[2];
if (memory[3] != 7 || at_5 != 11 || at_4 != 9
|| sel1 != 1 || sel2 != 1)
{
std::cerr << "prac search " << at_5 << " " << at_4 << "\n";
return 1;
}
constexpr unsigned answer = 0b101;
if (answer != 5 || memory[answer] < needle)
{
std::cerr << "prac index\n";
return 1;
}
// Heapify: one wide leaf of three lanes at the answer index.
wide3 payload{};
payload.lanes = {1, 2, 3};
auto [h0, h1] = dpf::make_dpf(static_cast<input_t>(answer), payload);
const auto w0 = *dpf::eval_point(h0, static_cast<input_t>(answer));
const auto w1 = *dpf::eval_point(h1, static_cast<input_t>(answer));
const auto opened = dpf::reconstruct(w0, w1);
if (opened.lanes[0] != 1 || opened.lanes[1] != 2 || opened.lanes[2] != 3)
{
std::cerr << "prac heapify\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("prac",
dpf::protocol::poplar_prefix_plan(0), 8))
return rc;
}
std::cout << "prac ok\n";
return 0;
}

View file

@ -0,0 +1,88 @@
#include <array>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Prio's frequency count, with the one-hot vector replaced by a DPF, and
// the prefix walk Poplar uses for heavy hitters (Boneh, Boyle,
// Corrigan-Gibbs, Gilboa, Ishai). Classic Prio proves an encoding with a
// SNIP; this file is only the DPF-shaped encoding.
// field64 is libprio's Field64.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/prio.cpp
namespace
{
constexpr int nbins = 256;
} // namespace
int main()
{
// Histogram. Each client sends one unit DPF at a secret bin.
// Each server adds the expansion into its running share with
// `eval_full_add_into` (no separate expansion buffer). The opened bin
// is the count.
const std::array<std::uint8_t, 4> bins{3, 3, 7, 3};
std::vector<dpf::field64> h0(nbins);
std::vector<dpf::field64> h1(nbins);
for (std::uint8_t bin : bins)
{
auto [k0, k1] = dpf::make_dpf(bin, dpf::field64{1});
dpf::eval_full_add_into(h0, k0);
dpf::eval_full_add_into(h1, k1);
}
// Leaf shares are subtractive, so the opened bin is share0 - share1.
const dpf::field64 c3 = h0[3] - h1[3];
const dpf::field64 c7 = h0[7] - h1[7];
const dpf::field64 c0 = h0[0] - h1[0];
if (c3.raw() != 3 || c7.raw() != 1 || c0.raw() != 0)
{
std::cerr << "prio histogram\n";
return 1;
}
// Heavy-hitter prefixes. idpf plants a 1 on each prefix length.
// Length 1 is the high bit. 0xA0 and 0xB0 share 101; they split at bit 4.
constexpr std::uint8_t left = 0xA0;
constexpr std::uint8_t right = 0xB0;
auto [a0, a1] = dpf::make_dpf(left,
dpf::idpf(std::uint64_t{1}, std::uint64_t{1}, std::uint64_t{1}));
auto [b0, b1] = dpf::make_dpf(right,
dpf::idpf(std::uint64_t{1}, std::uint64_t{1}, std::uint64_t{1}));
auto one = [](auto tag, auto k0, auto k1, std::uint8_t node) {
return dpf::reconstruct(*dpf::eval_point(tag, k0, node),
*dpf::eval_point(tag, k1, node));
};
auto count = [&](auto tag, std::uint8_t node) {
return one(tag, a0, a1, node) + one(tag, b0, b1, node);
};
// out<0> is prefix length 1, out<1> length 2, out<2> length 3.
const auto high = count(dpf::out<0, 1>, std::uint8_t{0x80});
const auto low = count(dpf::out<0, 1>, std::uint8_t{0x00});
const auto shared = count(dpf::out<2, 3>, std::uint8_t{0xA0});
const auto split = count(dpf::out<2, 3>, std::uint8_t{0x80});
if (high != 2 || low != 0 || shared != 2 || split != 0)
{
std::cerr << "prio prefixes " << high << " " << low << " " << shared
<< " " << split << "\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("prio",
dpf::protocol::poplar_prefix_plan(0), 8))
return rc;
}
std::cout << c3.raw() << "\n";
return 0;
}

View file

@ -0,0 +1,75 @@
#include <cstdint>
#include <cstring>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Private set intersection, the DPF step (Kolesnikov, Kumaresan, Rosulek,
// and Trieu, CCS 2016). The sender keeps a puncturable-PRF master. Each
// receiver element is a puncture; the servers evaluate the punctured key
// and test whether the tag sits in the sender's image. No full-domain table.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/psi.cpp
namespace
{
using domain_t = std::uint8_t;
bool blocks_eq(simde__m128i a, simde__m128i b)
{
return std::memcmp(&a, &b, sizeof(a)) == 0;
}
bool in_image(simde__m128i tag, const std::vector<simde__m128i> & image)
{
for (auto v : image)
if (blocks_eq(v, tag))
return true;
return false;
}
} // namespace
int main()
{
auto master = dpf::make_pprf_master<domain_t>();
const domain_t sender[] = {4, 10, 42};
const domain_t receiver[] = {42, 7};
std::vector<simde__m128i> image;
for (auto x : sender)
image.push_back(dpf::pprf_eval(master, x));
auto punctured_hit = dpf::puncture(master, receiver[0]);
auto punctured_miss = dpf::puncture(master, receiver[1]);
// Off-path points agree with the master; the programmed leaf at alpha
// matches the master leaf (sender who keeps the master set it).
const auto hit = dpf::pprf_eval(punctured_hit, receiver[0]);
const auto miss_off = dpf::pprf_eval(punctured_miss, domain_t{0});
const auto master_miss_off = dpf::pprf_eval(master, domain_t{0});
if (!blocks_eq(hit, dpf::pprf_eval(master, receiver[0]))
|| !in_image(hit, image)
|| in_image(dpf::pprf_eval(master, receiver[1]), image)
|| !blocks_eq(miss_off, master_miss_off))
{
std::cerr << "psi\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("psi",
dpf::protocol::psi_cuckoo_plan(0, {0, 1, 0}), 9))
return rc;
}
std::cout << "1\n";
return 0;
}

View file

@ -0,0 +1,61 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// A private range count. Each secret value is one comparison key. The
// public interval is [lo, hi). The comparison opens to 1 at a query q
// when q is strictly above the secret value, so the two endpoints
// differ by 1 exactly on lo <= v < hi. The count is the sum of those
// bits. The servers never see a value, and the interval is public.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/range_count.cpp
namespace
{
std::uint64_t inside(std::uint8_t value, std::uint8_t lo, std::uint8_t hi)
{
auto [k0, k1] = dpf::make_dpf(value, dpf::gt(std::uint64_t{1}));
const auto above_lo = dpf::reconstruct(
dpf::eval_point(dpf::cmp, k0, lo),
dpf::eval_point(dpf::cmp, k1, lo));
const auto above_hi = dpf::reconstruct(
dpf::eval_point(dpf::cmp, k0, hi),
dpf::eval_point(dpf::cmp, k1, hi));
// eval(q) = 1 iff q > value, so eval(hi) - eval(lo) = 1{lo <= value < hi}.
return above_hi - above_lo;
}
} // namespace
int main()
{
constexpr std::uint8_t lo = 10;
constexpr std::uint8_t hi = 20;
const std::uint8_t values[] = {3, 10, 12, 19, 20, 40};
std::uint64_t count = 0;
for (auto v : values)
count += inside(v, lo, hi);
// 10, 12, and 19. 3 and 40 are outside. 20 is the open end.
if (count != 3 || inside(10, lo, hi) != 1 || inside(20, lo, hi) != 0
|| inside(9, lo, hi) != 0)
{
std::cerr << "range count " << count << "\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("range_count",
dpf::protocol::range_count_plan(0), 8))
return rc;
}
std::cout << count << "\n";
return 0;
}

View file

@ -0,0 +1,79 @@
#include <cstddef>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Sabre, the mailbox write with a fast audit (Vadapalli, Storrier, and Henry,
// S&P 2022). Sender-anonymous messaging: two servers hold subtractive shares
// of every mailbox and the client sends one key each. Like Express the write
// is a full-domain add; unlike Express the audit is a *verifiable* DPF proof
// (Boyle et al. once-per-node fold), a constant-size token per party that
// opens to accept iff the key is a single honest point.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/sabre.cpp
namespace
{
constexpr std::size_t nboxes = 256;
} // namespace
int main()
{
constexpr std::uint8_t address = 17;
constexpr std::uint64_t message = 42;
auto [k0, k1] = dpf::make_dpf(address, message, dpf::verifiable{});
// Each server folds the write into its mailbox shares (one full walk).
std::vector<std::uint64_t> box0(nboxes, 0), box1(nboxes, 0);
dpf::eval_full_add_into(box0, k0);
dpf::eval_full_add_into(box1, k1);
if (box0[address] - box1[address] != message)
{
std::cerr << "sabre mailbox\n";
return 1;
}
if (box0[0] - box1[0] != 0)
{
std::cerr << "sabre neighbor\n";
return 1;
}
// Fast audit: a full-domain VDPF proof. Each party folds a constant-size
// token; the tokens open to accept an honest single-point write.
dpf::proof_token pi0{}, pi1{};
dpf::prove_full(k0, dpf::prove(pi0));
dpf::prove_full(k1, dpf::prove(pi1));
if (!dpf::verify(pi0, pi1))
{
std::cerr << "sabre audit\n";
return 1;
}
// A proof folded over a mismatched pair of points (the shape a malformed,
// multi-point write produces) fails the same check.
dpf::proof_token bad0{}, bad1{};
dpf::prove_interval(k0, std::uint8_t{0}, std::uint8_t{7}, dpf::prove(bad0));
dpf::prove_interval(k1, std::uint8_t{8}, std::uint8_t{15}, dpf::prove(bad1));
if (dpf::verify(bad0, bad1))
{
std::cerr << "sabre audit accepted a mismatch\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("sabre",
dpf::protocol::mailbox_write_fused_plan(0, 8), 8))
return rc;
}
std::cout << (box0[address] - box1[address]) << "\n";
return 0;
}

View file

@ -0,0 +1,65 @@
#include <cstddef>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Splinter, the DPF query step (Wang, Yun, Goldwasser, Vaikuntanathan, and
// Zeldovich, NSDI 2017). Private queries on public data with two-server FSS.
// The client's private selector is a unit DPF at a secret attribute value.
// Each server dots that selector with a public aggregate column, so the
// answer is the SUM (or COUNT) for the private key without either server
// learning which key was asked.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/splinter.cpp
int main()
{
constexpr std::size_t domain = 256; // attribute values
// Public data, pre-aggregated by attribute: group_sum[v] is the SUM of a
// value column over the rows whose attribute equals v.
std::vector<std::uint64_t> group_sum(domain);
std::vector<std::uint64_t> group_cnt(domain, 1);
for (std::size_t v = 0; v < domain; ++v)
group_sum[v] = (v * 37 + 11) % 1000;
constexpr std::uint8_t secret_key = 88; // the private WHERE value
// One selector key per server. reconstruct = the two servers' shares.
auto [k0, k1] = dpf::make_dpf(secret_key, std::uint64_t{1});
// SELECT SUM(value) WHERE attribute = secret_key.
const auto sum = dpf::reconstruct(
dpf::eval_full_inner_product(dpf::paired, k0, group_sum),
dpf::eval_full_inner_product(dpf::paired, k1, group_sum));
if (sum != group_sum[secret_key])
{
std::cerr << "splinter sum\n";
return 1;
}
// SELECT COUNT(*) WHERE attribute = secret_key is the same selector on an
// all-ones column.
const auto cnt = dpf::reconstruct(
dpf::eval_full_inner_product(dpf::paired, k0, group_cnt),
dpf::eval_full_inner_product(dpf::paired, k1, group_cnt));
if (cnt != 1)
{
std::cerr << "splinter count\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("splinter",
dpf::protocol::fss_point_plan(0), 8))
return rc;
}
std::cout << sum << "\n";
return 0;
}

View file

@ -0,0 +1,132 @@
#include <cstdint>
#include <iostream>
#include <iterator>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// MPC SUBLEQ, the DPF steps of one instruction (Jiang and Henry).
// Offline: expand wildcard unit keys with `defer_eval_full` before the
// addresses are known. Online: assign each address into `offset_x`, read
// by rotating the prepaid buffer (no second AES pass), write by scaling
// the same `e_B` view, and branch with a path evaluation of `x ≤ 0`.
//
// Instruction fetch is the same prepaid unit dotted against three sliding
// windows of D; this listing starts after (A, B, C) are already shares.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/subleq.cpp
namespace
{
using addr_t = std::uint8_t;
using word_t = std::uint32_t;
constexpr std::size_t n = 256;
template <typename Key0, typename Key1, typename T>
void assign_input(Key0 & k0, Key1 & k1, T alpha)
{
const T a0 = static_cast<T>(0x12);
const T a1 = static_cast<T>(alpha - a0);
const auto s0 = k0.offset_x.compute_and_get_share(a0);
const auto s1 = k1.offset_x.compute_and_get_share(a1);
k0.offset_x.reconstruct(s1);
k1.offset_x.reconstruct(s0);
}
/// Opened unit · public memory over `[0, n)`.
template <typename View0, typename View1>
word_t dot_prefix(View0 && v0, View1 && v1, const std::vector<word_t> & mem)
{
word_t acc = 0;
auto it0 = std::begin(v0);
auto it1 = std::begin(v1);
for (std::size_t i = 0; i < n; ++i, ++it0, ++it1)
acc += dpf::reconstruct(*it0, *it1) * mem[i];
return acc;
}
template <typename View0, typename View1>
void add_scaled_prefix(std::vector<word_t> & mem, View0 && v0, View1 && v1,
word_t scale)
{
auto it0 = std::begin(v0);
auto it1 = std::begin(v1);
for (std::size_t i = 0; i < n; ++i, ++it0, ++it1)
mem[i] += scale * dpf::reconstruct(*it0, *it1);
}
} // namespace
int main()
{
// Two's-complement words in a uint32_t container (same bits as int32_t).
constexpr addr_t A = 3;
constexpr addr_t B = 7;
constexpr addr_t C = 2;
constexpr addr_t pc = 0;
std::vector<word_t> D(n);
D[A] = 5;
D[B] = 3; // after SUBLEQ: D[B] = 3 - 5 = -2 ≤ 0 → pc' = C
// --- Offline: wildcard unit keys, full-domain expand at identity -----
auto [kA0, kA1] = dpf::make_dpf(dpf::wildcard_value<addr_t>{}, word_t{1});
auto [kB0, kB1] = dpf::make_dpf(dpf::wildcard_value<addr_t>{}, word_t{1});
auto bufA0 = dpf::make_output_buffer_for_full(kA0);
auto bufA1 = dpf::make_output_buffer_for_full(kA1);
auto bufB0 = dpf::make_output_buffer_for_full(kB0);
auto bufB1 = dpf::make_output_buffer_for_full(kB1);
auto defA0 = dpf::defer_eval_full(kA0, bufA0);
auto defA1 = dpf::defer_eval_full(kA1, bufA1);
auto defB0 = dpf::defer_eval_full(kB0, bufB0);
auto defB1 = dpf::defer_eval_full(kB1, bufB1);
// --- Online: open addresses, rotate prepaid unit vectors -------------
assign_input(kA0, kA1, A);
assign_input(kB0, kB1, B);
const word_t DA = dot_prefix(defA0.get(), defA1.get(), D);
const word_t DB = dot_prefix(defB0.get(), defB1.get(), D);
if (DA != D[A] || DB != D[B])
{
std::cerr << "subleq read\n";
return 1;
}
const word_t x = static_cast<word_t>(DB - DA); // wraps to -2 as uint32_t
// Write D[B] ← D[B] - D[A] by adding (-DA) · e_B. The protocol Beavers
// the scale; the opened -DA stands in here.
add_scaled_prefix(D, defB0.get(), defB1.get(), static_cast<word_t>(-DA));
if (D[B] != static_cast<word_t>(3 - 5) || D[A] != 5)
{
std::cerr << "subleq write\n";
return 1;
}
// Branch: path eval only — never expand the word-domain key.
// `leq` at knot 0, evaluated at x: 1 iff x ≤ 0 in signed order.
auto [kZ0, kZ1] = dpf::make_dpf(std::int32_t{0}, dpf::leq(std::uint64_t{1}));
const auto b = dpf::reconstruct(
dpf::eval_point(dpf::cmp, kZ0, static_cast<std::int32_t>(x)),
dpf::eval_point(dpf::cmp, kZ1, static_cast<std::int32_t>(x)));
const addr_t pc_next = b ? C : static_cast<addr_t>(pc + 3);
if (b != 1 || pc_next != C)
{
std::cerr << "subleq branch\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("subleq",
dpf::protocol::subleq_instruction_plan(0), 8))
return rc;
}
std::cout << static_cast<std::int32_t>(D[B]) << " "
<< static_cast<unsigned>(pc_next) << "\n";
return 0;
}

View file

@ -0,0 +1,69 @@
#include <cstddef>
#include <cstdint>
#include <iostream>
#include <vector>
#include "dpf.hpp"
#include "dpf/app_flow.hpp"
#include "dpf/app_plans.hpp"
// Waldo, the FSS steps (Dauterman, Fang, Crooks, and Popa, S&P 2022). A
// private time-series database. The store is append-only: a new event writes
// a fresh point and never updates an old one. A range/threshold aggregate
// uses the comparison (DCF) channel: the parties dot the per-timestamp
// comparison shares with a public value column, so a SUM over the timestamps
// past a *secret* threshold reveals neither the threshold nor the matches.
//
// c++ -std=c++17 -march=native -I include -I thirdparty \
// examples/applications/waldo.cpp
int main()
{
constexpr std::size_t horizon = 256; // timestamp domain
// Append-only writes. Each event is a unit DPF at its timestamp; the
// servers fold it into their subtractive value shares. Never an update.
std::vector<std::uint64_t> col0(horizon, 0), col1(horizon, 0);
const std::pair<std::uint8_t, std::uint64_t> events[] = {
{30, 5}, {90, 8}, {200, 3}};
for (auto [ts, val] : events)
{
auto [k0, k1] = dpf::make_dpf(ts, val);
dpf::eval_full_add_into(col0, k0);
dpf::eval_full_add_into(col1, k1);
}
if (col0[90] - col1[90] != 8 || col0[30] - col1[30] != 5)
{
std::cerr << "waldo append\n";
return 1;
}
// Public per-timestamp magnitudes (metadata the response consumes).
std::vector<std::uint64_t> magnitude(horizon, 0);
for (auto [ts, val] : events)
magnitude[ts] = val;
// Private-threshold aggregate: SUM of magnitudes at timestamps > T, with
// T secret. Key a gt comparison at T and dot its per-timestamp shares
// with the public magnitude column in one comparison walk.
constexpr std::uint8_t secret_T = 50;
auto [c0, c1] = dpf::make_dpf(secret_T, dpf::gt(std::uint64_t{1}));
const auto h0 = dpf::eval_full_inner_product(dpf::cmp, c0, magnitude);
const auto h1 = dpf::eval_full_inner_product(dpf::cmp, c1, magnitude);
const auto after = dpf::reconstruct_cmp_halves(h0, h1).raw();
// Timestamps 90 and 200 are past T=50: 8 + 3 = 11.
if (after != 11)
{
std::cerr << "waldo threshold aggregate " << after << "\n";
return 1;
}
{
if (int rc = dpf::app::run_measured("waldo",
dpf::protocol::fss_cmp_plan(0), 8))
return rc;
}
std::cout << after << "\n";
return 0;
}

View file

@ -0,0 +1,57 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
/// Pre-assign full-domain expand, then rotate after the input wildcard opens.
int main()
{
using input_type = std::uint8_t;
using output_type = std::uint64_t;
const output_type beta = 7;
const input_type alpha = 42;
const input_type from = 40;
const input_type to = 50;
auto [k0, k1] = dpf::make_dpf(dpf::wildcard_value<input_type>{}, beta);
//! [defer-eval]
auto buf0 = dpf::make_output_buffer_for_full(k0);
auto buf1 = dpf::make_output_buffer_for_full(k1);
auto deferred0 = dpf::defer_eval_interval(k0, from, to, buf0);
auto deferred1 = dpf::defer_eval_interval(k1, from, to, buf1);
// Parties open mask - alpha into offset_x (local demo of the exchange).
const input_type a0 = 0x12;
const input_type a1 = static_cast<input_type>(alpha - a0);
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
k0.offset_x.reconstruct(sh1);
k1.offset_x.reconstruct(sh0);
auto view0 = deferred0.get();
auto view1 = deferred1.get();
//! [defer-eval]
auto it0 = std::begin(view0);
auto it1 = std::begin(view1);
for (input_type x = from; x <= to; ++x, ++it0, ++it1)
{
const output_type got = dpf::reconstruct(*it0, *it1);
const output_type expect = (x == alpha) ? beta : 0;
if (got != expect)
{
std::cerr << "defer_eval\n";
return 1;
}
}
if (it0 != std::end(view0) || it1 != std::end(view1))
{
std::cerr << "defer_eval length\n";
return 1;
}
std::cout << dpf::reconstruct(*std::begin(view0), *std::begin(view1))
<< "\n";
return 0;
}

View file

@ -0,0 +1,44 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
/// Three-party comparison and interval-containment keys (one DCF share each).
int main()
{
using Input = std::uint8_t;
const Input thresh = 100;
const std::uint64_t beta = 5;
//! [eval-dpf3-cmp]
auto [c1, c2, c3] = dpf::make_dpf3_cmp(thresh, beta);
// Parties 1 and 3 hold the k0 half; party 2 holds k1. Open any complementary pair.
const dpf::fp61 hot =
dpf::reconstruct_cmp_halves(dpf::eval_point(c1, Input{10}),
dpf::eval_point(c2, Input{10}));
const dpf::fp61 cold =
dpf::reconstruct_cmp_halves(dpf::eval_point(c3, Input{200}),
dpf::eval_point(c2, Input{200}));
//! [eval-dpf3-cmp]
if (hot.raw() != beta || cold.raw() != 0)
{
std::cerr << "dpf3 cmp\n";
return 1;
}
//! [eval-dpf3-ic]
auto [i1, i2, i3] = dpf::make_dpf3_ic(Input{10}, Input{20}, Input{40}, beta);
// Interval is relative to the public shift `r`; x=35 is on for (20,40)@r=10.
const dpf::fp61 inside =
dpf::reconstruct_cmp_halves(dpf::eval_point(i1, Input{35}),
dpf::eval_point(i2, Input{35}));
//! [eval-dpf3-ic]
if (inside.raw() != beta)
{
std::cerr << "dpf3 ic\n";
return 1;
}
std::cout << hot.raw() << "\n";
return 0;
}

View file

@ -0,0 +1,36 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
/// Dual-spine Doerner–Shelat (2,3) keygen: XOR shares of `α`, same clear `β`
/// as honest-dealer `make_dpf3(α, β)`.
int main()
{
using Input = std::uint8_t;
const Input alpha = 0x2a;
const Input x0 = 0x13;
const Input x1 = static_cast<Input>(alpha ^ x0);
const dpf::fp61 beta{99};
//! [eval-dpf3-ds]
auto [d1, d2, d3] = dpf::make_dpf3(alpha, beta);
auto [s1, s2, s3] = dpf::make_dpf3_doerner_shelat(x0, x1, beta);
const dpf::fp61 dealer = dpf::reconstruct(
dpf::as_share(d1, dpf::eval_point(d1, alpha)),
dpf::as_share(d2, dpf::eval_point(d2, alpha)),
dpf::as_share(d3, dpf::eval_point(d3, alpha)));
const dpf::fp61 dual = dpf::reconstruct(
dpf::as_share(s1, dpf::eval_point(s1, alpha)),
dpf::as_share(s2, dpf::eval_point(s2, alpha)),
dpf::as_share(s3, dpf::eval_point(s3, alpha)));
//! [eval-dpf3-ds]
if (dealer != beta || dual != beta)
{
std::cerr << "dealer vs dual-spine disagree\n";
return 1;
}
std::cout << dual.raw() << "\n";
return 0;
}

View file

@ -0,0 +1,42 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
/// Three-evaluator point DPF (ePrint 2024/1658 Fig. 3). Each key is a Shamir
/// share; open with any two (or all three) via `dpf::reconstruct`.
int main()
{
using Input = std::uint8_t;
const Input alpha = 42;
const dpf::fp61 beta{7};
//! [eval-dpf3-point]
auto [k1, k2, k3] = dpf::make_dpf3(alpha, beta);
const dpf::fp61 y1 = dpf::eval_point(k1, alpha);
const dpf::fp61 y2 = dpf::eval_point(k2, alpha);
const dpf::fp61 y3 = dpf::eval_point(k3, alpha);
const dpf::fp61 opened = dpf::reconstruct(
dpf::as_share(k1, y1), dpf::as_share(k2, y2), dpf::as_share(k3, y3));
//! [eval-dpf3-point]
if (opened != beta)
{
std::cerr << "dpf3 at the programmed input\n";
return 1;
}
const dpf::fp61 z1 = dpf::eval_point(k1, Input{41});
const dpf::fp61 z2 = dpf::eval_point(k2, Input{41});
const dpf::fp61 z3 = dpf::eval_point(k3, Input{41});
if (dpf::reconstruct(dpf::as_share(k1, z1), dpf::as_share(k2, z2),
dpf::as_share(k3, z3))
.raw()
!= 0)
{
std::cerr << "dpf3 off the programmed input\n";
return 1;
}
std::cout << opened.raw() << "\n";
return 0;
}

View file

@ -0,0 +1,102 @@
#include <array>
#include <cstdint>
#include <iostream>
#include <tuple>
#include <vector>
#include "dpf.hpp"
/// Fused inner product: do not materialize the DPF vector.
/// A scalar weight vector dots with one output. A row of a tuple or
/// `std::array` dots with several outputs, including an ancestor slot
/// and the leaf, read off one path.
int main()
{
using In = std::uint8_t;
//! [eval-inner-product-scalar]
// Trivial: sum_x DPF(x) * w[x] over a short interval.
const In alpha = 42;
const std::uint64_t beta = 7;
auto [k0, k1] = dpf::make_dpf(alpha, beta);
const In from = 40;
const In to = 50;
std::vector<std::uint64_t> w(to - from + 1);
for (std::size_t i = 0; i < w.size(); ++i)
w[i] = i + 1;
const auto s0 = dpf::eval_inner_product(dpf::paired, k0, from, to, w);
const auto s1 = dpf::eval_inner_product(dpf::paired, k1, from, to, w);
//! [eval-inner-product-scalar]
if (dpf::reconstruct(s0, s1) != beta * w[alpha - from])
{
std::cerr << "scalar interval\n";
return 1;
}
//! [eval-inner-product-full]
// Trivial full domain. Only α contributes.
std::vector<std::uint64_t> wall(256, 1);
const auto f0 = dpf::eval_full_inner_product(dpf::paired, k0, wall);
const auto f1 = dpf::eval_full_inner_product(dpf::paired, k1, wall);
//! [eval-inner-product-full]
if (dpf::reconstruct(f0, f1) != beta)
{
std::cerr << "full\n";
return 1;
}
//! [eval-inner-product-paired]
// Two outputs on the same leaf. rows[i] = {weight for output 0, output 1}.
auto [p0, p1] = dpf::make_dpf(In{9}, std::uint32_t{3}, std::uint32_t{5});
std::vector<std::array<std::uint32_t, 2>> rows;
for (In x = 8;; ++x)
{
rows.push_back({std::uint32_t{1}, std::uint32_t{x}});
if (x == 10)
break;
}
const auto a0 = dpf::eval_inner_product<0, 1>(dpf::paired, p0, In{8}, In{10}, rows);
const auto a1 = dpf::eval_inner_product<0, 1>(dpf::paired, p1, In{8}, In{10}, rows);
//! [eval-inner-product-paired]
// x=9 is hot: output0 * 1 + output1 * 9.
if (dpf::reconstruct(a0, a1) != std::uint64_t{3} * 1u + std::uint64_t{5} * 9u)
{
std::cerr << "paired leaf\n";
return 1;
}
//! [eval-inner-product-ancestor]
// Prefix slot at<4> and the full-domain leaf, one path per point.
// 0x2a and 0x2b share the high nibble 0x2, so both see payload 5 there.
// 0x10 is a different nibble. Only 0x2a is hot on the leaf.
auto [h0, h1] = dpf::make_dpf(In{0x2a}, dpf::at<4>(std::uint8_t{5}), std::uint8_t{9});
const std::vector<In> pts{0x10, 0x2a, 0x2b};
const std::vector<std::tuple<std::uint32_t, std::uint32_t>> hw{
{1u, 0u}, {1u, 1u}, {2u, 4u}};
const auto q0 = dpf::eval_sequence_inner_product<0, 1>(h0, pts.begin(), pts.end(), hw);
const auto q1 = dpf::eval_sequence_inner_product<0, 1>(h1, pts.begin(), pts.end(), hw);
//! [eval-inner-product-ancestor]
// 0x10 is off. 0x2a: 5*1 + 9*1. 0x2b: prefix still 5, leaf 0, times (2, 4).
const std::uint64_t ancestor_expect = 5u * 1u + 9u * 1u + 5u * 2u;
if (dpf::reconstruct(q0, q1) != ancestor_expect)
{
std::cerr << "ancestor\n";
return 1;
}
//! [eval-inner-product-recipe]
const auto recipe = dpf::make_sequence_recipe<decltype(h0)>(pts.begin(), pts.end());
const auto r0 = dpf::eval_sequence_inner_product<0, 1>(
h0, recipe, pts.begin(), pts.end(), hw);
const auto r1 = dpf::eval_sequence_inner_product<0, 1>(
h1, recipe, pts.begin(), pts.end(), hw);
//! [eval-inner-product-recipe]
if (dpf::reconstruct(r0, r1) != ancestor_expect)
{
std::cerr << "recipe\n";
return 1;
}
std::cout << dpf::reconstruct(s0, s1) << "\n";
return 0;
}

View file

@ -0,0 +1,58 @@
#include <cmath>
#include <cstdint>
#include <iostream>
#include <vector>
#include "grotto.hpp"
// Haar and bior(5,3) lookup tables (Reis, Ugurbil, Wagh, Henry, de Vega,
// PoPETs 2025, ePrint 2025/013). The grid is sigmoid on [0, 4), stored as
// Q4.4. Depth 2 keeps the top 4 bits of a 6-bit index.
//
// c++ -std=c++17 -march=native -I include -I thirdparty examples/grotto/dwt_lut.cpp
int main()
{
//! [dwt-lut]
constexpr unsigned domain_bits = 6;
constexpr unsigned fractional_bits = 4;
constexpr unsigned depth = 2;
auto samples = grotto::sample_dwt_signal(domain_bits, fractional_bits,
[](double x) {
return 1.0 / (1.0 + std::exp(-(x - 2.0)));
});
auto haar = grotto::make_haar_dwt_lut(samples, fractional_bits, depth);
auto bior = grotto::make_bior53_dwt_lut(samples, fractional_bits, depth);
// Haar is the mean of each block of 2^depth samples, then quantized.
const std::uint64_t raw = 32;
double block = 0;
for (unsigned k = 0; k < 4; ++k)
block += samples[(raw & ~std::uint64_t{3}) + k];
const auto haar_expect = static_cast<std::int64_t>(
std::floor(block / 4.0 * 16.0));
// bior(5,3), lsb = 0: only the first tap, at index msb+2, divided by 2^j.
const std::uint64_t msb = raw >> depth;
const auto c0 = bior.coeff[(msb + 2) % bior.coeff.size()];
const auto bior_at_32 = c0 / 4;
// lsb = 1: both taps, weights (2^j - lsb) and lsb, then divide by 2^{2j}.
const auto c1 = bior.coeff[(msb + 3) % bior.coeff.size()];
const auto bior_at_33 = (c0 * 3 + c1) / 16;
//! [dwt-lut]
if (haar(raw) != haar_expect || haar(raw) != 8)
{
std::cerr << "haar lut\n";
return 1;
}
if (bior(raw) != bior_at_32 || bior(33) != bior_at_33 || bior(raw) != 8)
{
std::cerr << "bior lut\n";
return 1;
}
std::cout << haar(raw) << " " << bior(raw) << " " << bior(33) << "\n";
return 0;
}

View file

@ -0,0 +1,98 @@
#include <cstdint>
#include <iostream>
#include <vector>
#include "grotto.hpp"
/// Binomial jet readouts and an exact ring switch from one public offset.
int main()
{
//! [jet-and-ring]
// --- Binomial jet -------------------------------------------------------
// After eta opens, the jet is the binomial basis at the wrapped
// center+kappa. Degree 3 leaves room for a degree-2 hockey-stick prefix.
const std::uint8_t center = 12;
const std::uint8_t eta = 3;
const std::uint8_t point = static_cast<std::uint8_t>(center + eta); // 15
const std::size_t degree = 3;
auto jet_keys = grotto::make_offset_jet_keys<std::uint8_t>(center, degree);
// f(t) = 4 + 2 C(t,1) + C(t,2) (padded to degree 3).
const std::vector<std::uint64_t> poly{4, 2, 1};
std::vector<std::uint64_t> coeff = poly;
coeff.push_back(0);
const std::vector<std::uint8_t> knots{0};
const auto j0 = grotto::offset_jet_shares<0>(jet_keys, knots, eta);
const auto j1 = grotto::offset_jet_shares<1>(jet_keys, knots, eta);
std::vector<std::uint64_t> jet(degree + 1);
for (std::size_t k = 0; k <= degree; ++k)
jet[k] = j0[k] + j1[k];
const std::uint64_t value = grotto::offset_jet_dot(coeff, jet);
const std::uint64_t diff = grotto::offset_jet_dot(
grotto::offset_jet_difference_coeff(coeff), jet);
const std::uint64_t prefix = grotto::offset_jet_dot(
grotto::offset_jet_prefix_coeff(poly), jet);
// Padé / Newton are public dots against the same jet, then one reciprocal
// after the shares are opened. For a seed p(t)/p'(t):
// auto num = offset_jet_dot(coeff, jet);
// auto den = offset_jet_dot(offset_jet_difference_coeff(coeff), jet);
// // open num, den; one masked reciprocal; Newton: t - num/den.
// --- Exact ring switch --------------------------------------------------
// Same public-offset pattern: eta = x - r, then x lands in the residue.
const std::uint8_t r = 200;
const std::uint8_t x = 44;
const std::uint8_t ring_eta = static_cast<std::uint8_t>(x - r); // 100, wraps
using Z = grotto::zn64<1009>;
auto ring = grotto::make_ring_switch_keys<Z>(r);
const Z x_mod = grotto::ring_switch_eval<0>(ring, ring_eta)
+ grotto::ring_switch_eval<1>(ring, ring_eta);
auto field = grotto::make_ring_switch_keys<dpf::field128>(r);
const dpf::field128 x_field = grotto::ring_switch_eval<0>(field, ring_eta)
+ grotto::ring_switch_eval<1>(field, ring_eta);
//! [jet-and-ring]
auto c = [](std::uint64_t t, unsigned k) {
return grotto::offset_jet_binom(t, k);
};
const std::uint64_t expect_v =
4 + 2 * c(point, 1) + c(point, 2);
if (value != expect_v)
{
std::cerr << "jet value\n";
return 1;
}
const std::uint64_t expect_fx1 =
4 + 2 * c(static_cast<std::uint8_t>(point + 1), 1)
+ c(static_cast<std::uint8_t>(point + 1), 2);
if (diff != expect_fx1 - expect_v)
{
std::cerr << "jet difference\n";
return 1;
}
std::uint64_t expect_p = 0;
for (std::uint8_t i = 0; i < point; ++i)
expect_p += 4 + 2 * c(i, 1) + c(i, 2);
if (prefix != expect_p)
{
std::cerr << "jet prefix\n";
return 1;
}
if (x_mod.raw() != static_cast<std::uint64_t>(x) % 1009)
{
std::cerr << "ring zn64\n";
return 1;
}
if (x_field != dpf::field128{x})
{
std::cerr << "ring field128\n";
return 1;
}
std::cout << value << " " << diff << " " << prefix << " "
<< x_mod.raw() << "\n";
return 0;
}

View file

@ -0,0 +1,43 @@
#include <cstdint>
#include <iostream>
#include <vector>
#include "grotto.hpp"
/// Two piecewise LUTs, one comparison, one prefix walk of the union.
int main()
{
//! [lut-union]
grotto::piecewise_lut<std::uint8_t> low{{0, 10}, {{1, 0}, {0, 2}}};
grotto::piecewise_lut<std::uint8_t> high{{0, 4, 12}, {{3, 0}, {1, 1}, {9, 4}}};
const std::uint8_t center = 12;
const std::uint8_t eta = 3;
auto plan = grotto::make_lut_union_plan({low, high}, eta);
auto mat = grotto::make_offset_poly_keys<std::uint8_t>(center, plan.degree);
auto s0 = grotto::lut_union_eval<0>(mat, plan);
auto s1 = grotto::lut_union_eval<1>(mat, plan);
dpf::protocol::composer composer(0);
grotto::schedule_lut_union(composer, plan);
//! [lut-union]
if (plan.comparisons != 1 || plan.prefix_walks != 1)
{
std::cerr << "plan shape\n";
return 1;
}
if (composer.default_plan().rounds() != plan.depth)
{
std::cerr << "geneval rounds\n";
return 1;
}
const std::uint64_t opened[2] = {s0[0] + s1[0], s0[1] + s1[1]};
if (opened[0] != grotto::offset_poly_clear<std::uint8_t>(center, low.knots, low.coeff, eta)
|| opened[1] != grotto::offset_poly_clear<std::uint8_t>(center, high.knots, high.coeff, eta))
{
std::cerr << "opened value\n";
return 1;
}
return 0;
}

View file

@ -0,0 +1,132 @@
#include <cstdint>
#include <iostream>
#include <vector>
#include "grotto.hpp"
namespace
{
std::uint64_t pow_u64(std::uint64_t base, std::uint64_t exp)
{
std::uint64_t acc = 1;
while (exp != 0)
{
if (exp & 1u)
acc *= base;
base *= base;
exp >>= 1;
}
return acc;
}
} // namespace
/// Representation shift (Fibonacci / geometric / CRC) and twisted monomials.
int main()
{
//! [repr-and-twist]
// --- Representation shift: Fibonacci checkpoint ----------------------
// Dealer keys S_c = (F_{c+1}, F_c). After eta opens, each party applies
// the public companion-matrix power M^kappa to its share of S_c.
const std::uint8_t center = 10;
const std::uint8_t eta = 5;
const std::uint8_t point = static_cast<std::uint8_t>(center + eta); // 15
const auto fib_state = grotto::offset_repr_fibonacci_state(center);
const auto M = grotto::offset_repr_fibonacci_matrix();
auto fib_keys = grotto::make_offset_repr_keys<std::uint8_t>(center, fib_state);
const std::vector<std::uint8_t> knots{0};
const auto f0 = grotto::offset_repr_eval<0>(fib_keys, M, knots, eta);
const auto f1 = grotto::offset_repr_eval<1>(fib_keys, M, knots, eta);
const std::vector<std::uint64_t> S{f0[0] + f1[0], f0[1] + f1[1]};
// Geometric twin: 1x1 matrix [lambda] advances lambda^c by lambda^kappa.
const std::uint64_t lambda_geo = 3;
const auto G = grotto::offset_repr_geometric_matrix(lambda_geo);
auto geo_keys = grotto::make_offset_repr_keys<std::uint8_t>(
center, {pow_u64(lambda_geo, center)});
const auto g0 = grotto::offset_repr_eval<0>(geo_keys, G, knots, eta);
const auto g1 = grotto::offset_repr_eval<1>(geo_keys, G, knots, eta);
const std::uint64_t geo = g0[0] + g1[0];
// Clear CRC-32 jump documents the GF(2) twin (XOR shares, not additive).
const std::uint32_t crc_seed = 0x12345678u;
const std::uint32_t crc_jumped = grotto::offset_repr_crc32_jump(crc_seed, 64);
// --- Twisted monomials: (a0 + a1 x + a2 x^2) * lambda^x -------------
const std::uint64_t lambda = 3;
const std::size_t degree = 2;
auto twist_keys = grotto::make_offset_twist_keys<std::uint8_t>(
center, degree, lambda);
// h(x) = (2 + 5x + x^2) * 3^x
const std::vector<std::uint64_t> coeff{2, 5, 1};
const std::uint64_t twisted =
grotto::offset_twist_eval<0>(twist_keys, knots, coeff, eta)
+ grotto::offset_twist_eval<1>(twist_keys, knots, coeff, eta);
// Dyadic decay: masked carry shift. The sum of the shares is the shifted value.
auto half_keys = grotto::make_offset_twist_keys<std::uint8_t>(
center, degree, grotto::twist_half);
const std::uint64_t half =
grotto::offset_twist_eval<0>(half_keys, knots, coeff, eta)
+ grotto::offset_twist_eval<1>(half_keys, knots, coeff, eta);
// Closed form sum_{k=1}^n k * lambda^k from the same twisted table.
const std::uint64_t ag = grotto::offset_twist_arithmetico_geometric(point, lambda);
//! [repr-and-twist]
const auto expect_S = grotto::offset_repr_fibonacci_state(point);
if (S != expect_S)
{
std::cerr << "fibonacci state\n";
return 1;
}
if (geo != pow_u64(lambda_geo, point))
{
std::cerr << "geometric\n";
return 1;
}
std::uint64_t expect_t = 0;
std::uint64_t xp = 1;
for (std::uint64_t c : coeff)
{
expect_t += c * xp;
xp *= point;
}
expect_t *= pow_u64(lambda, point);
if (twisted != expect_t)
{
std::cerr << "twisted poly\n";
return 1;
}
std::uint64_t expect_h = 0;
xp = 1;
for (std::uint64_t c : coeff)
{
expect_h += c * xp;
xp *= point;
}
expect_h >>= point;
if (half != expect_h)
{
std::cerr << "twist half\n";
return 1;
}
std::uint64_t expect_ag = 0;
for (std::uint64_t k = 1; k <= point; ++k)
expect_ag += k * pow_u64(lambda, k);
if (ag != expect_ag)
{
std::cerr << "arithmetico-geometric\n";
return 1;
}
if (crc_jumped == 0 && crc_seed != 0)
{
// Jump can legally land on zero; only used as a smoke output.
}
std::cout << S[1] << " " << geo << " " << twisted << " " << half << " "
<< ag << " " << crc_jumped << "\n";
return 0;
}

131
examples/mwe/chooser.html Normal file
View file

@ -0,0 +1,131 @@
<div class="chooser">
<p class="q">Who knows the secret index?</p>
<input type="radio" name="holder" id="h-dealer">
<label class="choice" for="h-dealer"><strong>A dealer</strong><span>knows alpha and beta, and hands each party a key</span></label>
<input type="radio" name="holder" id="h-share">
<label class="choice" for="h-share"><strong>The two parties</strong><span>already share alpha. They want a key they can reuse</span></label>
<input type="radio" name="holder" id="h-answer">
<label class="choice" for="h-answer"><strong>The two parties</strong><span>already share alpha. They want the answer, not a key</span></label>
<input type="radio" name="holder" id="h-three">
<label class="choice" for="h-three"><strong>Three evaluators</strong><span>any two of them can open the value</span></label>
<div class="branch branch-dealer">
<p class="q">What should be nonzero?</p>
<input type="radio" name="dealer-what" id="d-point">
<label class="choice" for="d-point"><strong>One point</strong><span>beta at alpha, zero elsewhere</span></label>
<input type="radio" name="dealer-what" id="d-cmp">
<label class="choice" for="d-cmp"><strong>A comparison</strong><span>1 where x is above alpha</span></label>
<input type="radio" name="dealer-what" id="d-ic">
<label class="choice" for="d-ic"><strong>A public interval</strong><span>beta on a span of the unmasked input</span></label>
<div class="result result-point">
<h3>dpf::make_dpf</h3>
<p>Dealer point key. Leaf shares are subtractive, so reconstruct subtracts them. This prints <code>7 0</code>.</p>
<p><a href="incremental_8hpp.html">dpf/incremental.hpp</a></p>
<button type="button" class="mwe-copy">Copy</button>
<pre class="mwe"><code>#include &lt;cstdint&gt;
#include &lt;iostream&gt;
#include "dpf.hpp"
int main()
{
const std::uint8_t alpha = 42;
const std::uint64_t beta = 7;
auto [k0, k1] = dpf::make_dpf(alpha, beta);
const std::uint64_t at = dpf::reconstruct(
*dpf::eval_point(k0, alpha),
*dpf::eval_point(k1, alpha));
const std::uint64_t off = dpf::reconstruct(
*dpf::eval_point(k0, std::uint8_t{0}),
*dpf::eval_point(k1, std::uint8_t{0}));
std::cout &lt;&lt; at &lt;&lt; " " &lt;&lt; off &lt;&lt; "\n";
return (at == beta &amp;&amp; off == 0) ? 0 : 1;
}</code></pre>
<p class="mwe-cmd"><code>c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/point.cpp</code></p>
</div>
<div class="result result-cmp">
<h3>dpf::gt</h3>
<p>One comparison on the same key. Comparison shares are additive, so reconstruct adds them. This prints <code>1 0</code>.</p>
<p><a href="dcf_8hpp.html">dpf/dcf.hpp</a></p>
<button type="button" class="mwe-copy">Copy</button>
<pre class="mwe"><code>#include &lt;cstdint&gt;
#include &lt;iostream&gt;
#include "dpf.hpp"
int main()
{
const std::uint8_t alpha = 40;
auto [k0, k1] = dpf::make_dpf(alpha, dpf::gt(std::uint64_t{1}));
const auto above = dpf::reconstruct(
dpf::eval_point(dpf::cmp, k0, std::uint8_t{50}),
dpf::eval_point(dpf::cmp, k1, std::uint8_t{50}));
const auto below = dpf::reconstruct(
dpf::eval_point(dpf::cmp, k0, std::uint8_t{10}),
dpf::eval_point(dpf::cmp, k1, std::uint8_t{10}));
std::cout &lt;&lt; above &lt;&lt; " " &lt;&lt; below &lt;&lt; "\n";
return (above == 1 &amp;&amp; below == 0) ? 0 : 1;
}</code></pre>
<p class="mwe-cmd"><code>c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/compare.cpp</code></p>
</div>
<div class="result result-ic">
<h3>dpf::ic</h3>
<p>The secret is a mask. The public interval is on <code>x - r</code>. This prints <code>9 0</code>: 14 - 10 = 4, which sits in [3, 5].</p>
<p><a href="interval_8hpp.html">dpf/interval.hpp</a></p>
<button type="button" class="mwe-copy">Copy</button>
<pre class="mwe"><code>#include &lt;cstdint&gt;
#include &lt;iostream&gt;
#include "dpf.hpp"
int main()
{
const std::uint8_t r = 10;
const std::uint64_t beta = 9;
auto [k0, k1] = dpf::make_dpf(r, dpf::ic(std::uint8_t{3}, std::uint8_t{5}, beta));
const auto inside = dpf::reconstruct(
dpf::eval_point(dpf::ic, k0, std::uint8_t{14}),
dpf::eval_point(dpf::ic, k1, std::uint8_t{14}));
const auto outside = dpf::reconstruct(
dpf::eval_point(dpf::ic, k0, std::uint8_t{0}),
dpf::eval_point(dpf::ic, k1, std::uint8_t{0}));
std::cout &lt;&lt; inside &lt;&lt; " " &lt;&lt; outside &lt;&lt; "\n";
return (inside == beta &amp;&amp; outside == 0) ? 0 : 1;
}</code></pre>
<p class="mwe-cmd"><code>c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/interval.cpp</code></p>
</div>
</div>
<div class="branch branch-share">
<h3>dpf::make_dpf_doerner_shelat</h3>
<p>Each party holds an XOR share of alpha. A pad dealer supplies the tape and does not learn alpha. The key is reusable. This prints <code>7</code>.</p>
<p><a href="doerner__shelat_8hpp.html">dpf/doerner_shelat.hpp</a></p>
<button type="button" class="mwe-copy">Copy</button>
<pre class="mwe"><code>#include &lt;cstdint&gt;
#include &lt;iostream&gt;
#include "dpf.hpp"
int main()
{
const std::uint8_t alpha = 42;
const std::uint64_t beta = 7;
const std::uint8_t x0 = 7;
const std::uint8_t x1 = static_cast&lt;std::uint8_t&gt;(alpha ^ x0);
auto root = []() { return dpf::uniform_sample&lt;simde__m128i&gt;(); };
struct pad {
simde__m128i block() { return dpf::uniform_sample&lt;simde__m128i&gt;(); }
std::uint8_t bit() {
return static_cast&lt;std::uint8_t&gt;(dpf::uniform_sample&lt;unsigned&gt;() &amp; 1u);
}
};
dpf::ds_randomness&lt;decltype(root), pad&gt; rng{root, {}};
auto keys = dpf::make_dpf_doerner_shelat(x0, x1, rng, beta);
const std::uint64_t opened = dpf::reconstruct(
*dpf::eval_point(keys.first, alpha),
*dpf::eval_point(keys.second, alpha));
std::cout &lt;&lt; opened &lt;&lt; "\n";
return opened == beta ? 0 : 1;
}</code></pre>
<p class="mwe-cmd"><code>c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/shared_index.cpp</code></p>
</div>
<div class="branch branch-answer">
<h3>dpf::geneval_point</h3>
<p>Same share convention as the reusable key, and the parties open the value along the query. The call does not hand back a key you can evaluate again. The openings are on <a href="geneval_8hpp.html">dpf/geneval.hpp</a>: <code>geneval_point</code>, <code>geneval_interval</code>, <code>geneval_sequence</code>, <code>geneval_full</code>, and <code>geneval_cmp</code>.</p>
</div>
<div class="branch branch-three">
<h3>dpf::make_dpf3</h3>
<p>Three evaluators, any two open. Spines are Shamir shares in <code>fp61</code>. The dealerless form is <code>make_dpf3_doerner_shelat</code>. Comparisons for that setting are <code>make_dpf3_cmp</code> and <code>make_dpf3_ic</code>.</p>
<p><a href="dpf3_8hpp.html">dpf/dpf3.hpp</a> · <a href="dpf3__cmp_8hpp.html">dpf/dpf3_cmp.hpp</a></p>
</div>
</div>

24
examples/mwe/compare.cpp Normal file
View file

@ -0,0 +1,24 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
// Complete program. Comparison shares are additive: reconstruct is share0 + share1.
// gt(1) is 1 where x > alpha and 0 elsewhere.
//
// c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/compare.cpp
int main()
{
const std::uint8_t alpha = 40;
auto [k0, k1] = dpf::make_dpf(alpha, dpf::gt(std::uint64_t{1}));
const auto above = dpf::reconstruct(
dpf::eval_point(dpf::cmp, k0, std::uint8_t{50}),
dpf::eval_point(dpf::cmp, k1, std::uint8_t{50}));
const auto below = dpf::reconstruct(
dpf::eval_point(dpf::cmp, k0, std::uint8_t{10}),
dpf::eval_point(dpf::cmp, k1, std::uint8_t{10}));
std::cout << above << " " << below << "\n";
return (above == 1 && below == 0) ? 0 : 1;
}

28
examples/mwe/interval.cpp Normal file
View file

@ -0,0 +1,28 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
// Complete program. The secret is a mask r. The public interval is [p, q]
// on the unmasked input x - r. ic returns `beta` inside that interval.
//
// c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/interval.cpp
int main()
{
const std::uint8_t r = 10;
const std::uint8_t p = 3;
const std::uint8_t q = 5;
const std::uint64_t beta = 9;
auto [k0, k1] = dpf::make_dpf(r, dpf::ic(p, q, beta));
// x = 14 means x - r = 4, which is inside [3, 5].
const auto inside = dpf::reconstruct(
dpf::eval_point(dpf::ic, k0, std::uint8_t{14}),
dpf::eval_point(dpf::ic, k1, std::uint8_t{14}));
const auto outside = dpf::reconstruct(
dpf::eval_point(dpf::ic, k0, std::uint8_t{0}),
dpf::eval_point(dpf::ic, k1, std::uint8_t{0}));
std::cout << inside << " " << outside << "\n";
return (inside == beta && outside == 0) ? 0 : 1;
}

24
examples/mwe/point.cpp Normal file
View file

@ -0,0 +1,24 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
// Complete program. Leaf shares are subtractive: reconstruct is share0 - share1.
//
// c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/point.cpp
int main()
{
const std::uint8_t alpha = 42;
const std::uint64_t beta = 7;
auto [k0, k1] = dpf::make_dpf(alpha, beta);
const std::uint64_t at = dpf::reconstruct(
*dpf::eval_point(k0, alpha),
*dpf::eval_point(k1, alpha));
const std::uint64_t off = dpf::reconstruct(
*dpf::eval_point(k0, std::uint8_t{0}),
*dpf::eval_point(k1, std::uint8_t{0}));
std::cout << at << " " << off << "\n";
return (at == beta && off == 0) ? 0 : 1;
}

43
examples/mwe/shamir.cpp Normal file
View file

@ -0,0 +1,43 @@
#include <array>
#include <cstdint>
#include <iostream>
#include <tuple>
#include <type_traits>
#include "dpf.hpp"
// (K,N) Shamir. The secret is the constant term of a degree K-1 polynomial.
// Party i holds that polynomial at x = i+1. Any K shares open it. (2,3) is
// shamir_share: make_shamir_shares(secret, slope) is deal<T, 2, 3>.
// fp61 and gf2n both open. For gf2n, N must be less than 2^k.
//
// c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/shamir.cpp
int main()
{
using F = dpf::fp61;
const F secret{10};
// p(x) = 10 + 2x + 3x^2. Parties 0, 2, and 4 are enough.
const std::array<F, 2> coeff{{F{2}, F{3}}};
auto shares = dpf::make_shamir_shares<3, 5>(secret, coeff);
const F opened = dpf::shamir::reconstruct(
std::get<0>(shares), std::get<2>(shares), std::get<4>(shares));
const F slope{3};
auto [s0, s1, s2] = dpf::make_shamir_shares(secret, slope);
static_assert(std::is_same_v<decltype(s0), dpf::shamir::share<F, 0, 2, 3>>);
const F opened23 = dpf::reconstruct(s1, s2);
const bool on_line = s0.raw() == secret + slope * F{1}
&& s2.raw() == secret + slope * F{3};
using G = dpf::gf28;
auto [g0, g1, g2] = dpf::make_shamir_shares(G{0x1b}, G{0x5a});
const G gopen = dpf::reconstruct(g0, g2);
const G gopen13 = dpf::reconstruct(g1, g2);
std::cout << opened.raw() << " " << opened23.raw() << " "
<< static_cast<unsigned>(gopen.raw()) << "\n";
return (opened == secret && opened23 == secret && on_line
&& gopen == G{0x1b} && gopen13 == G{0x1b})
? 0 : 1;
}

View file

@ -0,0 +1,36 @@
#include <cstdint>
#include <iostream>
#include "dpf.hpp"
// Complete program. The parties already hold XOR shares of alpha.
// The pad stream is local here; a real protocol would draw it from a dealer
// who never sees alpha.
//
// c++ -std=c++17 -march=native -I include -I thirdparty examples/mwe/shared_index.cpp
int main()
{
const std::uint8_t alpha = 42;
const std::uint64_t beta = 7;
const std::uint8_t x0 = 7;
const std::uint8_t x1 = static_cast<std::uint8_t>(alpha ^ x0);
auto root = []() { return dpf::uniform_sample<simde__m128i>(); };
struct pad
{
simde__m128i block() { return dpf::uniform_sample<simde__m128i>(); }
std::uint8_t bit()
{
return static_cast<std::uint8_t>(dpf::uniform_sample<unsigned>() & 1u);
}
};
dpf::ds_randomness<decltype(root), pad> rng{root, {}};
auto keys = dpf::make_dpf_doerner_shelat(x0, x1, rng, beta);
const std::uint64_t opened = dpf::reconstruct(
*dpf::eval_point(keys.first, alpha),
*dpf::eval_point(keys.second, alpha));
std::cout << opened << "\n";
return opened == beta ? 0 : 1;
}

View file

@ -0,0 +1,10 @@
#include <iostream>
#include "dpf.hpp"
int main()
{
auto [k0, k1] = dpf::make_dpf(std::uint8_t{3}, dpf::gf28{0x1b});
auto y0 = *dpf::eval_point(k0, std::uint8_t{3});
auto y1 = *dpf::eval_point(k1, std::uint8_t{3});
std::cout << dpf::reconstruct(y0, y1) << "\n";
}

View file

@ -0,0 +1,154 @@
#include <atomic>
#include <cstdint>
#include <cstring>
#include <iostream>
#include <random>
#include <thread>
#include <vector>
#include "dpf/launch.hpp"
#include "dpf/net/client_link.hpp"
#include "dpf/run_log.hpp"
// A client splits a secret into two additive shares and sends share i to party
// i over a client link; the two parties then open the sum over their own
// party link. Both links are TLS 1.3; each end logs how it authenticated the
// other.
//
// c++ -std=c++17 -march=native -pthread -I include -I thirdparty \
// examples/protocol/client_shares.cpp -lsctp -lssl -lcrypto -o client_shares
// ./client_shares # development certificate (logged)
// ./dpf_keygen srv.key # prints srv's public key
// ./client_shares --server_identity=srv.key --client_pin=<srv public key>
// ./client_shares --client_verify=off # accepts any server (logged)
//
// With --server_identity and no --client_pin the client refuses the server:
// clients always verify unless told not to.
int main(int argc, char ** argv)
{
try
{
auto cfg = dpf::app::run_config::from_env();
for (const auto & extra : cfg.apply_args(argc, argv))
throw std::invalid_argument("unknown argument " + extra);
if (cfg.kind != dpf::net::transport::mux && cfg.kind != dpf::net::transport::parallel)
cfg.kind = dpf::net::transport::mux;
dpf::app::start_logging(cfg);
// Each party accepts one client and keeps the share it sends.
std::atomic<unsigned short> ports[2] = {{0}, {0}};
std::uint64_t shares[2] = {0, 0};
std::string errors[2];
std::vector<std::thread> parties;
for (int p = 0; p < 2; ++p)
parties.emplace_back([&, p] {
try
{
asio::io_context io;
dpf::net::client_listener l(io, cfg.server, 1, cfg.policy, cfg.limits);
ports[p].store(l.listen());
auto c = l.accept();
bool done = false;
std::error_code ec;
c.link->async_read(0, &shares[p], 8, [&](const std::error_code & e) {
ec = e;
done = true;
});
while (!done)
{
if (io.stopped())
io.restart();
io.run_one();
}
if (ec)
throw std::system_error(ec, "reading the client's share");
}
catch (const std::exception & e)
{
errors[p] = e.what();
ports[p].store(1);
}
});
// The client: one fresh share per party, each on its own verified link.
const std::uint64_t secret = 42;
std::random_device rd;
const std::uint64_t r = (static_cast<std::uint64_t>(rd()) << 32) | rd();
const std::uint64_t mine[2] = {r, secret - r};
std::string client_error;
for (int p = 0; p < 2 && client_error.empty(); ++p)
{
while (ports[p].load() == 0)
std::this_thread::yield();
try
{
asio::io_context io;
auto c = dpf::net::connect_server(io, cfg.host, ports[p].load(), cfg.client,
1, cfg.policy, cfg.limits);
std::cout << "client -> party " << p << ": " << c.security.protocol << " "
<< c.security.cipher << ", server auth=" << c.security.peer_auth
<< "\n";
bool done = false;
c.link->async_write(0, &mine[p], 8, [&](const std::error_code &) {
done = true;
});
while (!done)
{
if (io.stopped())
io.restart();
io.run_one();
}
c.link.reset();
io.poll();
}
catch (const std::exception & e)
{
client_error = e.what();
}
}
if (!client_error.empty())
{
// Unblock the listeners that are still waiting for this client.
for (int p = 0; p < 2; ++p)
{
std::error_code ec;
asio::io_context io;
asio::ip::tcp::socket s(io);
if (ports[p].load() > 1)
s.connect({asio::ip::make_address("127.0.0.1"), ports[p].load()}, ec);
}
}
for (auto & t : parties)
t.join();
if (!client_error.empty())
throw std::runtime_error("client: " + client_error);
for (const auto & e : errors)
if (!e.empty())
throw std::runtime_error("party: " + e);
// The parties open the sum over their party link.
dpf::protocol::composer c0(0), c1(1);
auto x0 = c0.input(dpf::protocol::domain::a, 8);
auto x1 = c1.input(dpf::protocol::domain::a, 8);
auto o0 = c0.exchange(x0);
(void)c1.exchange(x1);
auto p0 = c0.schedule();
auto p1 = c1.schedule();
dpf::app::party_values v0(p0.nodes().size()), v1(p1.nodes().size());
v0[x0.id].assign(8, 0);
v1[x1.id].assign(8, 0);
std::memcpy(v0[x0.id].data(), &shares[0], 8);
std::memcpy(v1[x1.id].data(), &shares[1], 8);
(void)dpf::run_two_party(p0, p1, v0, v1, {}, cfg);
std::uint64_t open = 0;
std::memcpy(&open, v0[o0.id].data(), 8);
std::cout << "parties opened " << open << " (share 0 = " << shares[0] << ")\n";
return open == secret ? 0 : 1;
}
catch (const std::exception & e)
{
std::cerr << "client_shares: " << e.what() << "\n";
return 1;
}
}

View file

@ -0,0 +1,285 @@
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <iostream>
#include <thread>
#include <vector>
#include "dpf.hpp"
#include "dpf/net/stream_array.hpp"
#include "dpf/protocol_factory.hpp"
// Protocol composition schedules (compose.hpp). Records FSS walks, ABY
// products, and RSS refreshes on one RoundSink plan — the shapes Express,
// Sabre, Pika early-stop, Poplar prefixes, and Duoram scale use by hand.
//
// The final block drives a composed open on the stream framework
// (drive_both_on_streams). The identical schedule also runs over the truly
// asynchronous backends (dpf::net::async_stream_array + async_round_sink, or
// dpf::async::overlapped_byte_protocol) and, on Linux, over real SCTP
// (async_sctp_stream_array). The experiment_bench harness picks the transport
// via DPF_TRANSPORT=memory|stream|async|mux|parallel|sctp.
//
// c++ -std=c++17 -march=native -I include -I thirdparty
// examples/protocol/compose_schedule.cpp
namespace
{
using dpf::protocol::composer;
using dpf::protocol::domain;
using dpf::protocol::effect;
namespace opcodes = dpf::protocol::opcodes;
} // namespace
int main()
{
// Express / Sabre: audit rides in the last CW flush (depth exchanges, not
// depth+1).
{
composer naive(0);
auto seed = naive.input(domain::fss, 16);
auto sketch = naive.input(domain::a, 8);
auto leaf = naive.fss_point(seed, 4, 16);
auto dep = naive.compute(opcodes::user_base + 1, {leaf, sketch},
domain::a, 8);
(void)naive.exchange(dep);
const auto naive_ex = naive.schedule().exchange_waves();
composer fused(0);
auto seed_f = fused.input(domain::fss, 16);
auto sketch_f = fused.input(domain::a, 8);
auto wr = fused.fss_point_fused(seed_f, 4, 16, sketch_f);
const auto fused_ex = fused.schedule().exchange_waves();
if (fused_ex >= naive_ex || fused.domain_of(wr.trailer_open) != domain::a)
{
std::cerr << "compose fuse\n";
return 1;
}
std::cout << "express_fuse " << naive_ex << "->" << fused_ex << "\n";
}
// Pika / small-output PIR: BGI Remark 3.4 early-stop drops ν CW rounds.
{
composer full(0);
auto seed = full.input(domain::fss, 16);
(void)full.fss_point(seed, 8, 16);
composer early(0);
auto seed_e = early.input(domain::fss, 16);
(void)early.fss_point_early_stop(seed_e, 8, /*early_stop=*/3, 16);
const auto a = full.schedule().exchange_waves();
const auto b = early.schedule().exchange_waves();
if (b != a - 3)
{
std::cerr << "compose early_stop\n";
return 1;
}
std::cout << "early_stop " << a << "->" << b << "\n";
}
// Poplar: prefix share after each CW, same exchange-wave depth.
{
composer c(0);
auto seed = c.input(domain::fss, 16);
auto wr = c.level_walk_prefixes(seed, 5, 16, /*prefix_bytes=*/8);
auto p = c.schedule();
if (wr.at_level.size() != 5 || p.exchange_waves() != 5)
{
std::cerr << "compose prefixes\n";
return 1;
}
std::cout << "prefixes " << p.exchange_waves() << "\n";
}
// DCF block_width: variable CW slot sizes.
{
composer c(0);
auto seed = c.input(domain::fss, 16);
const std::vector<std::size_t> slots = {16, 4, 4, 16};
(void)c.level_walk_sized(seed, slots);
auto p = c.schedule();
if (p.value_bytes_of(p.wave(1).exchanges[0].id) != 4)
{
std::cerr << "compose sized\n";
return 1;
}
std::cout << "sized_slots ok\n";
}
// RSS product → neighbor y-exchange → RSS (one round).
{
composer c(0);
auto x = c.input(domain::rss, 16);
auto y = c.input(domain::rss, 16);
auto z = c.rss_product_replicated(x, y);
if (c.domain_of(z) != domain::rss || c.schedule().exchange_waves() != 1)
{
std::cerr << "compose rss\n";
return 1;
}
std::cout << "rss_refresh 1\n";
}
// Duoram write-scale: FSS leaf feeds ABY; beaver waits for the leaf wave.
{
composer c(0);
auto seed = c.input(domain::fss, 16);
auto leaf = c.fss_point(seed, 4, 16);
auto scale = c.input(domain::a, 8);
auto scaled = c.aby_product<std::uint64_t>(leaf, scale);
auto p = c.schedule();
if (p.wave_of(scaled) < p.wave_of(leaf))
{
std::cerr << "compose duoram_scale\n";
return 1;
}
std::cout << "duoram_scale wave " << p.wave_of(scaled) << "\n";
}
// Round-aware ABY: sign×linear stays one online round.
{
composer c(0);
auto & s = c.aby<std::uint64_t>();
auto sgn = s.input();
auto x = s.input();
auto a0 = s.input();
auto a1 = s.input();
auto lin = s(sgn * (a1 * x + a0));
if (s.round_of(lin) != 1)
{
std::cerr << "compose aby_rounds\n";
return 1;
}
std::cout << "aby_rounds 1\n";
}
// Doerner–Shelat: 5 opens per level; OH adds 80 AND-layers / level.
{
composer c(0);
auto seed = c.input(domain::fss, 16);
auto tip = c.level_walk_ds(seed, 2, 16);
if (c.schedule().exchange_waves() != 10 || tip.id == seed.id)
{
std::cerr << "compose ds_walk\n";
return 1;
}
std::cout << "ds_walk 10\n";
composer c_oh(0);
auto tip_oh = c_oh.level_walk_ds(c_oh.input(domain::fss, 16), 1, 16,
/*oh=*/true);
(void)tip_oh;
const auto want = dpf::net::compose_ds_slot_bytes(1, 16, true).size();
if (c_oh.schedule().exchange_waves() != want)
{
std::cerr << "compose ds_oh\n";
return 1;
}
std::cout << "ds_oh " << want << "\n";
}
// Adaptive idpf: one packed L‖R open per step.
{
composer c(0);
auto seed = c.input(domain::fss, 16);
auto f = c.begin_adaptive_prefix(seed);
f = c.step_adaptive_prefix(f, 16, 8);
f = c.retain_adaptive_prefix(f, 0);
if (c.schedule().exchange_waves() != 1)
{
std::cerr << "compose adaptive\n";
return 1;
}
std::cout << "adaptive 1\n";
}
// default_plan: rounds == exchange_waves (sink-aligned).
{
composer c(0);
auto leaf = c.fss_point(c.input(domain::fss, 16), 4, 16);
(void)leaf;
auto p = c.default_plan();
if (p.rounds() != p.exchange_waves()
|| p.rounds() != p.slot_bytes_all().size() || p.rounds() != 4)
{
std::cerr << "compose default_plan\n";
return 1;
}
std::cout << "default_plan 4\n";
}
// Multipoint buckets: CW waves pack; answers one open.
{
composer c(0);
auto mr = c.multipoint_fan(2,
[&](std::size_t) { return c.input(domain::fss, 16); }, 3, 16, 8);
(void)mr;
if (c.schedule().exchange_waves() != 4)
{
std::cerr << "compose multipoint\n";
return 1;
}
std::cout << "multipoint 4\n";
}
// Multi-lane ABY: independent barriers, one wave.
{
composer c(0);
auto z0 = c.aby_product<std::uint64_t>(c.input(domain::a, 8),
c.input(domain::a, 8), 0);
auto z1 = c.aby_product<std::uint64_t>(c.input(domain::a, 8),
c.input(domain::a, 8), 1);
if (c.schedule().exchange_waves() != 1
|| c.schedule().effect_count(effect::exchange) != 2
|| z0.id == z1.id)
{
std::cerr << "compose multilane\n";
return 1;
}
std::cout << "multilane 1\n";
}
// Prepaid defer + rotate: zero online FSS rounds.
{
composer c(0);
auto buf = c.defer_expand(c.input(domain::fss, 16), 4, 16);
auto rot = c.rotate_share(buf, 7);
if (c.schedule().exchange_waves() != 0 || rot.id == buf.id)
{
std::cerr << "compose defer\n";
return 1;
}
std::cout << "defer 0\n";
}
// stream_array: drive a compose open through drive_both_on_streams.
{
composer c0(0);
composer c1(1);
auto x0 = c0.input(domain::a, 8);
auto x1 = c1.input(domain::a, 8);
auto e0 = c0.exchange(x0);
auto e1 = c1.exchange(x1);
auto p0 = c0.schedule();
auto p1 = c1.schedule();
std::vector<std::vector<std::uint8_t>> v0(p0.nodes().size()),
v1(p1.nodes().size());
const std::uint64_t a = 3, b = 5;
v0[x0.id].assign(8, 0);
v1[x1.id].assign(8, 0);
std::memcpy(v0[x0.id].data(), &a, 8);
std::memcpy(v1[x1.id].data(), &b, 8);
dpf::protocol::drive_both_on_streams(p0, p1, v0, v1);
std::uint64_t open0 = 0, open1 = 0;
std::memcpy(&open0, v0[e0.id].data(), 8);
std::memcpy(&open1, v1[e1.id].data(), 8);
if (open0 != 8 || open1 != 8)
{
std::cerr << "drive_plan_on_streams open\n";
return 1;
}
std::cout << "drive_plan_on_streams_ok " << open0 << "\n";
}
return 0;
}

View file

@ -0,0 +1,86 @@
#include <cstdint>
#include <cstring>
#include <iostream>
#include <vector>
#include "dpf/launch.hpp"
#include "dpf/run_log.hpp"
// One party of a two- or three-party run, one process per party.
//
// c++ -std=c++17 -march=native -pthread -I include -I thirdparty \
// -DLIBDPF_GIT_REV="\"$(git describe --always --dirty)\"" \
// examples/protocol/party_node.cpp -lsctp -lssl -lcrypto -o party_node
// ./party_node --party=0 --peers=127.0.0.1:9100,127.0.0.1:9101 &
// ./party_node --party=1 --peers=127.0.0.1:9100,127.0.0.1:9101
//
// Any run_config key works as a flag (--transport=parallel --lanes=2 ...).
// Start order does not matter: connects retry until --connect_ms.
//
// Links are TLS 1.3. With no keys they are encrypted but unauthenticated (and
// the log says so). To authenticate, give each party a key and the others'
// public keys (dpf_keygen p0.key prints p0's), for example in p0.conf:
// identity = p0.key
// peer.1 = <p1's public key>
// and run ./party_node --config=p0.conf --party=0 --peers=...
// The run log goes to stderr at info; --log=file:/tmp/p0.log --log_level=debug
// or DPF_LOG / DPF_LOG_LEVEL redirect it (see dpf/run_log.hpp).
//
// With three peers the protocol adds a dealer-delivered mask: party 2 is the
// dealer, and parties 0 and 1 each open (input + mask share).
int main(int argc, char ** argv)
{
try
{
const auto args = dpf::app::parse_node_args(argc, argv);
dpf::app::start_logging(args.cfg);
const unsigned me = args.party;
const bool dealer = args.peers.size() == 3;
dpf::protocol::composer c(me);
auto x = c.input(dpf::protocol::domain::a, 8);
dpf::protocol::node opened{};
dpf::protocol::node mask{};
if (dealer)
{
auto pad = c.input(dpf::protocol::domain::a, 8);
mask = c.dealer_deliver(pad);
opened = c.exchange(x);
(void)pad;
}
else
opened = c.exchange(x);
auto plan = c.schedule();
dpf::app::party_values values(plan.nodes().size());
for (auto & v : values)
v.assign(8, 0);
const std::uint64_t mine = me == 0 ? 11 : 31;
std::memcpy(values[x.id].data(), &mine, 8);
if (dealer && me == 2)
{
const std::uint64_t pad = 7;
std::memcpy(values[plan.inputs_of(mask.id)[0]].data(), &pad, 8);
}
const auto wire = dpf::app::run_node(args, plan, values);
std::uint64_t got = 0;
std::memcpy(&got, values[opened.id].data(), 8);
std::cout << "party " << me << " open=" << got;
if (dealer && me < 2)
{
std::uint64_t m = 0;
std::memcpy(&m, values[mask.id].data(), 8);
std::cout << " mask=" << m;
}
std::cout << " wire_out=" << wire.bytes_out << " wire_in=" << wire.bytes_in
<< "\n";
return (me == 2 || got == 42) ? 0 : 1;
}
catch (const std::exception & e)
{
DPF_LOG(error, "node.failed").kv("what", e.what());
std::cerr << "party_node: " << e.what() << "\n";
return 1;
}
}

View file

@ -0,0 +1,99 @@
/// @file examples/protocol/share_runtime.cpp
/// @brief Smoke demo: stream_array dealer/peer gadgets and clear sanity checks.
/// @details Also drives a composed open on the stream framework
/// (`drive_both_on_streams`). The same protocol runs unchanged over the
/// truly asynchronous backends (`dpf::net::async_stream_array` +
/// `dpf::async::overlapped_byte_protocol`, or `async_round_sink`), and
/// on Linux over real SCTP (`async_sctp_stream_array`, index i -> SCTP
/// stream i). The performance harness (experiment_bench) selects any of
/// these with `DPF_TRANSPORT=memory|stream|async|mux|parallel|sctp`.
#include <cstdint>
#include <cstring>
#include <iostream>
#include <thread>
#include <vector>
#include "dpf/compose.hpp"
#include "dpf/factory_gadgets.hpp"
#include "dpf/net/stream_array.hpp"
#include "dpf/protocol_factory.hpp"
#include "dpf/rss_seed.hpp"
int main()
{
auto bundle = dpf::rss::sample_seed_bundle();
auto z0 = dpf::rss::zero_share<std::uint64_t>(
dpf::rss::party_seeds::from_bundle(bundle, 0), 0);
auto z1 = dpf::rss::zero_share<std::uint64_t>(
dpf::rss::party_seeds::from_bundle(bundle, 1), 0);
auto z2 = dpf::rss::zero_share<std::uint64_t>(
dpf::rss::party_seeds::from_bundle(bundle, 2), 0);
std::cout << "rss_zero_sum=" << (z0 + z1 + z2) << "\n";
// Dealer: one ring triple per party; peer: d and e opens.
auto peer = dpf::net::make_memory_stream_pair(2);
auto d0 = dpf::net::make_memory_stream_pair(1);
auto d1 = dpf::net::make_memory_stream_pair(1);
constexpr std::uint16_t limb = 8;
dpf::factory::make_dealer(d0.first, d1.first, 1, [limb] {
return dpf::factory::deal_ring_triple(limb);
});
std::uint64_t prod0 = 0, prod1 = 0;
std::thread t0([&] {
prod0 = dpf::factory::beaver_mul_online(0, 6, 7, peer.first, d0.second,
0, 0, 1, limb);
});
std::thread t1([&] {
prod1 = dpf::factory::beaver_mul_online(1, 0, 0, peer.second, d1.second,
0, 0, 1, limb);
});
t0.join();
t1.join();
std::cout << "beaver_mul_open=" << (prod0 + prod1) << "\n";
// GMW AND 1∧1 via make_protocol_factory.
auto and_peer = dpf::net::make_memory_stream_pair(1);
auto ad0 = dpf::net::make_memory_stream_pair(1);
auto ad1 = dpf::net::make_memory_stream_pair(1);
dpf::factory::make_dealer(ad0.first, ad1.first, 1,
dpf::factory::make_gmw_and_dealer_functor());
std::uint8_t z0b = 0, z1b = 0;
std::thread a0([&] {
z0b = dpf::factory::gmw_and_online(0, 1, 1, and_peer.first, ad0.second);
});
std::thread a1([&] {
z1b = dpf::factory::gmw_and_online(1, 0, 0, and_peer.second, ad1.second);
});
a0.join();
a1.join();
std::cout << "gmw_and_xor=" << static_cast<unsigned>(z0b ^ z1b) << "\n";
// Compose an open and drive both parties on the stream framework. This is
// the same schedule the async backends run; here it uses in-process memory
// stream arrays via drive_both_on_streams.
{
using dpf::protocol::domain;
dpf::protocol::composer comp0(0), comp1(1);
auto a = comp0.input(domain::a, 8);
auto b = comp1.input(domain::a, 8);
auto oa = comp0.exchange(a);
auto ob = comp1.exchange(b);
auto p0 = comp0.schedule();
auto p1 = comp1.schedule();
std::vector<std::vector<std::uint8_t>> va(p0.nodes().size()),
vb(p1.nodes().size());
const std::uint64_t xa = 17, xb = 25;
va[a.id].assign(8, 0);
vb[b.id].assign(8, 0);
std::memcpy(va[a.id].data(), &xa, 8);
std::memcpy(vb[b.id].data(), &xb, 8);
dpf::protocol::drive_both_on_streams(p0, p1, va, vb);
std::uint64_t open0 = 0;
std::memcpy(&open0, va[oa.id].data(), 8);
(void)ob;
std::cout << "compose_open_on_streams=" << open0 << "\n";
}
return 0;
}

View file

@ -0,0 +1,83 @@
/// @file examples/protocol/stream_app_smoke.cpp
/// @brief Thin app-runtime smoke: prep files or memory dealer + compose on streams.
/// @details Drives a composed open via `drive_both_on_streams` (the stream
/// framework). The same schedule runs unchanged over the event-driven
/// backends (`dpf::net::async_stream_array` + `async_round_sink` /
/// `dpf::async::overlapped_byte_protocol`) and, on Linux, over real
/// SCTP (`async_sctp_stream_array`). The `experiment_bench` harness
/// selects the transport with
/// `DPF_TRANSPORT=memory|stream|async|mux|parallel|sctp`.
#include <cstdint>
#include <cstring>
#include <iostream>
#include <vector>
#include "dpf/app_runtime.hpp"
#include "dpf/compose.hpp"
#include "dpf/prep_source.hpp"
#include "dpf/protocol_roles.hpp"
namespace
{
void put_raw(std::vector<std::vector<std::uint8_t>> & values,
dpf::protocol::node n, const void * src, std::size_t nbyte)
{
values[n.id].assign(nbyte, 0);
std::memcpy(values[n.id].data(), src, nbyte);
}
} // namespace
int main()
{
// Prep: file dealer round-trip (roles helper + app_runtime reader).
{
const std::string base = "/tmp/libdpf_stream_app_smoke_prep";
dpf::prep::demand d{};
d.ring_triples = 1;
dpf::roles::dealer_write_files(base, d);
auto c0 = dpf::app::open_file_prep_cursor(base, 0);
auto c1 = dpf::app::open_file_prep_cursor(base, 1);
if (c0.limb() != 8 || c1.limb() != 8)
{
std::cerr << "prep cursor\n";
return 1;
}
}
// Online: memory dealer + single-wave exchange plan on stream arrays.
dpf::protocol::composer c0(0);
dpf::protocol::composer c1(1);
auto x0 = c0.input(dpf::protocol::domain::a, 8);
auto x1 = c1.input(dpf::protocol::domain::a, 8);
auto e0 = c0.exchange(x0);
auto e1 = c1.exchange(x1);
auto p0 = c0.schedule();
auto p1 = c1.schedule();
dpf::prep::demand d{};
d.ring_triples = 1;
auto [d0, d1] = dpf::prep::deal_memory_pair(d);
(void)d0;
(void)d1;
std::vector<std::vector<std::uint8_t>> v0(p0.nodes().size());
std::vector<std::vector<std::uint8_t>> v1(p1.nodes().size());
const std::uint64_t a = 5, b = 9;
put_raw(v0, x0, &a, 8);
put_raw(v1, x1, &b, 8);
dpf::protocol::drive_both_on_streams(p0, p1, v0, v1);
std::uint64_t open0 = 0, open1 = 0;
std::memcpy(&open0, v0[e0.id].data(), 8);
std::memcpy(&open1, v1[e1.id].data(), 8);
if (open0 != 14u || open1 != 14u)
{
std::cerr << "exchange " << open0 << " " << open1 << "\n";
return 1;
}
std::cout << "ok\n";
return 0;
}

View file

@ -0,0 +1,145 @@
/// @file examples/protocol/stream_dpf3_smoke.cpp
/// @brief Local DPF3 eval plus trio-shaped stream edges (peer / rss_next / dealer).
#include <cstdint>
#include <cstring>
#include <iostream>
#include <thread>
#include <type_traits>
#include <vector>
#include "dpf.hpp"
#include "dpf/net/stream_array.hpp"
#include "dpf/net/stream_mesh.hpp"
#include "dpf/protocol_factory.hpp"
namespace
{
struct trio_edge_ping
{
std::uint32_t from = 0;
std::uint32_t to = 0;
std::uint64_t nonce = 0;
};
struct trio_party_streams
{
dpf::net::memory_stream_array & peer;
dpf::net::memory_stream_array & rss_next;
dpf::net::memory_stream_array & dealer;
};
template <typename T>
void exchange_pod(dpf::net::memory_stream_array & link, std::size_t stream,
const T & mine, T & theirs)
{
static_assert(std::is_trivially_copyable_v<T>, "pod");
link.write(stream, &mine, sizeof(T));
link.flush(stream);
link.read(stream, &theirs, sizeof(T));
}
void ping_edge(trio_party_streams s, unsigned me, unsigned peer_id,
std::uint64_t nonce, std::size_t stream_idx)
{
trio_edge_ping mine{me, peer_id, nonce};
trio_edge_ping theirs{};
exchange_pod(s.peer, stream_idx, mine, theirs);
if (theirs.from != peer_id || theirs.to != me)
throw std::runtime_error("stream_dpf3_smoke: peer edge");
}
} // namespace
int main()
{
using Input = std::uint8_t;
const Input alpha = 42;
const dpf::fp61 beta{7};
auto [k1, k2, k3] = dpf::make_dpf3(alpha, beta);
const dpf::fp61 y1 = dpf::eval_point(k1, alpha);
const dpf::fp61 y2 = dpf::eval_point(k2, alpha);
const dpf::fp61 y3 = dpf::eval_point(k3, alpha);
const dpf::fp61 opened = dpf::reconstruct(
dpf::as_share(k1, y1), dpf::as_share(k2, y2), dpf::as_share(k3, y3));
if (opened != beta)
{
std::cerr << "dpf3 local eval\n";
return 1;
}
// Dealer delivers a keyed marker on stream 0 to each evaluator.
auto dealer0 = dpf::net::make_memory_stream_pair(1);
auto dealer1 = dpf::net::make_memory_stream_pair(1);
struct key_delivery
{
std::uint32_t party = 0;
std::uint64_t beta_raw = 0;
};
key_delivery m0{0, beta.raw()}, m1{1, beta.raw()};
dealer0.first.write(0, &m0, sizeof(m0));
dealer0.first.flush(0);
dealer1.first.write(0, &m1, sizeof(m1));
dealer1.first.flush(0);
key_delivery got0{}, got1{};
dealer0.second.read(0, &got0, sizeof(got0));
dealer1.second.read(0, &got1, sizeof(got1));
if (got0.beta_raw != beta.raw() || got1.beta_raw != beta.raw())
{
std::cerr << "dealer stream delivery\n";
return 1;
}
// Trio-shaped edges on a 3-party clique (p2 = dealer): peer + rss_next + dealer.
constexpr std::size_t k_peer = 0;
constexpr std::size_t k_rss = 1;
auto clique = dpf::net::make_memory_stream_clique(3, 2);
std::exception_ptr err;
std::thread t0([&] {
try
{
trio_party_streams s{clique.end(0, 1), clique.end(0, 1), clique.end(0, 2)};
ping_edge(s, 0, 1, 11, k_peer);
const trio_edge_ping rss_out{0, 1, 99};
s.rss_next.write(k_rss, &rss_out, sizeof(rss_out));
s.rss_next.flush(k_rss);
}
catch (...)
{
err = std::current_exception();
}
});
std::thread t1([&] {
try
{
trio_party_streams s{clique.end(1, 0), clique.end(1, 0), clique.end(1, 2)};
ping_edge(s, 1, 0, 22, k_peer);
trio_edge_ping rss_in{};
s.rss_next.read(k_rss, &rss_in, sizeof(rss_in));
if (rss_in.from != 0u)
throw std::runtime_error("stream_dpf3_smoke: rss_next");
const trio_edge_ping dealer_out{1, 2, 88};
s.dealer.write(k_rss, &dealer_out, sizeof(dealer_out));
s.dealer.flush(k_rss);
}
catch (...)
{
err = std::current_exception();
}
});
t0.join();
t1.join();
if (err)
std::rethrow_exception(err);
trio_edge_ping dealer_in{};
clique.end(2, 1).read(k_rss, &dealer_in, sizeof(dealer_in));
if (dealer_in.from != 1u || dealer_in.to != 2u)
{
std::cerr << "dealer edge\n";
return 1;
}
std::cout << "stream_trio_ok\n";
return 0;
}

View file

@ -0,0 +1,64 @@
#include <cstdint>
#include <cstring>
#include <iostream>
#include <vector>
#include "dpf/launch.hpp"
#include "dpf/run_log.hpp"
// Two-party open. Every run choice is a flag:
//
// c++ -std=c++17 -march=native -pthread -I include -I thirdparty \
// examples/protocol/two_party_exchange.cpp -lsctp
// ./a.out # in-process async memory
// ./a.out --transport=mux --lanes=1 # TCP mux on localhost
// ./a.out --transport=parallel --window=65536
//
// For separate processes, see party_node.cpp.
int main(int argc, char ** argv)
{
try
{
auto cfg = dpf::app::run_config::from_env();
for (const auto & extra : cfg.apply_args(argc, argv))
throw std::invalid_argument("unknown argument " + extra);
dpf::app::start_logging(cfg);
dpf::protocol::composer c0(0);
dpf::protocol::composer c1(1);
auto x0 = c0.input(dpf::protocol::domain::a, 8);
auto x1 = c1.input(dpf::protocol::domain::a, 8);
auto o0 = c0.exchange(x0);
(void)c1.exchange(x1);
auto p0 = c0.schedule();
auto p1 = c1.schedule();
// One 8-byte input per instance; every instance opens to 42.
dpf::app::party_values v0(p0.nodes().size()), v1(p1.nodes().size());
v0[x0.id].assign(8 * cfg.instances, 0);
v1[x1.id].assign(8 * cfg.instances, 0);
for (std::size_t i = 0; i < cfg.instances; ++i)
{
const std::uint64_t a = 11 + i, b = 31 - i;
std::memcpy(v0[x0.id].data() + 8 * i, &a, 8);
std::memcpy(v1[x1.id].data() + 8 * i, &b, 8);
}
const auto r = dpf::run_two_party(p0, p1, v0, v1, {}, cfg);
bool all = true;
std::uint64_t open = 0;
for (std::size_t i = 0; i < cfg.instances; ++i)
{
std::memcpy(&open, v0[o0.id].data() + 8 * i, 8);
all = all && open == 42;
}
std::cout << "two_party_exchange " << (all ? 42 : open) << " "
<< cfg.summary() << " wire_out=" << r.wire[0].bytes_out << "\n";
return all ? 0 : 1;
}
catch (const std::exception & e)
{
std::cerr << "two_party_exchange: " << e.what() << "\n";
return 1;
}
}

View file

@ -0,0 +1,42 @@
#include <cstring>
#include <iostream>
#include <string>
#include "dpf/net/identity.hpp"
// Party identity keys for encrypted, authenticated party links.
//
// c++ -std=c++17 -I include -I thirdparty examples/tools/dpf_keygen.cpp \
// -lssl -lcrypto -o dpf_keygen
// ./dpf_keygen p0.key # new key file (mode 600); prints the public key
// ./dpf_keygen --public p0.key # public key of an existing key file
//
// Give each party its own key file (`identity = p0.key` in its config) and
// give the parties that should authenticate it the printed public key
// (`peer.0 = <public key>`).
int main(int argc, char ** argv)
{
try
{
if (argc == 3 && std::strcmp(argv[1], "--public") == 0)
{
std::cout << dpf::net::identity::load(argv[2]).key().base64() << "\n";
return 0;
}
if (argc == 2 && argv[1][0] != '-')
{
const auto id = dpf::net::identity::generate();
id.save(argv[1]);
std::cout << id.key().base64() << "\n";
return 0;
}
std::cerr << "usage: dpf_keygen KEYFILE | dpf_keygen --public KEYFILE\n";
return 2;
}
catch (const std::exception & e)
{
std::cerr << "dpf_keygen: " << e.what() << "\n";
return 1;
}
}

View file

@ -26,14 +26,18 @@
#include "dpf/bitstring.hpp"
#include "dpf/blob.hpp"
#include "dpf/pprf.hpp"
#include "dpf/dpf_key.hpp"
#include "dpf/doerner_shelat.hpp"
#include "dpf/geneval.hpp"
#include "dpf/incremental.hpp"
#include "dpf/geneval.hpp"
#include "dpf/eval_common.hpp"
#include "dpf/eval_interval.hpp"
@ -46,8 +50,28 @@
#include "dpf/eval_sequence.hpp"
#include "dpf/cohort.hpp"
#include "dpf/interleave_leaves.hpp"
#include "dpf/eval_unified.hpp"
#include "dpf/eval_walk.hpp"
#include "dpf/eval_until.hpp"
#include "dpf/idpf_agg.hpp"
#include "dpf/it_dpf3.hpp"
#include "dpf/caller_fold.hpp"
#include "dpf/leaf_later.hpp"
#include "dpf/grow.hpp"
#include "dpf/grow_ds.hpp"
#include "dpf/interval_memoizer.hpp"
#ifdef LIBDPF_HAS_NLOHMANN_JSON
@ -88,6 +112,34 @@
#include "dpf/beaver.hpp"
#include "dpf/compose.hpp"
#include "dpf/rss_seed.hpp"
#include "dpf/ot_pack.hpp"
#include "dpf/edabit.hpp"
#include "dpf/bit_inject.hpp"
#include "dpf/trunc.hpp"
#include "dpf/share_cmp.hpp"
#include "dpf/share_vec.hpp"
#include "dpf/fixed_share.hpp"
#include "dpf/gilboa.hpp"
#include "dpf/share_expr.hpp"
#include "dpf/matmul.hpp"
#include "dpf/shuffle.hpp"
#include "dpf/cost_pass.hpp"
#include "dpf/secret_share.hpp"
#include "dpf/rotation_iterable.hpp"
@ -102,6 +154,8 @@
#include "dpf/subinterval_iterable.hpp"
#include "dpf/deferred_rotated_subinterval.hpp"
#include "dpf/subsequence_iterable.hpp"
#include "dpf/twiddle.hpp"
@ -118,14 +172,42 @@
#include "dpf/fp61.hpp"
#include "dpf/gf2.hpp"
#include "dpf/field64.hpp"
#include "dpf/field128.hpp"
#include "dpf/p256.hpp"
#include "dpf/p256_scalar.hpp"
#include "dpf/shamir.hpp"
#include "dpf/shamir3.hpp"
#include "dpf/verifiable.hpp"
#include "dpf/path_sketch.hpp"
#include "dpf/sfss.hpp"
#include "dpf/multipoint.hpp"
#include "dpf/ppvc.hpp"
#include "dpf/dpf3.hpp"
#include "dpf/dpf3_ds.hpp"
#include "dpf/dpf3_multipoint.hpp"
#include "dpf/dpf3_cmp.hpp"
#include "dpf/vec.hpp"
#include "dpf/interval.hpp"
#include "dpf/eval_peel.hpp"
#endif // LIBDPF_INCLUDE_DPF_HPP__

View file

@ -47,6 +47,7 @@
#include "portable-snippets/exact-int/exact-int.h"
#include "dpf/utils.hpp"
#include "dpf/twiddle.hpp"
#include "dpf/bit_array.hpp"
namespace dpf
@ -55,16 +56,6 @@ namespace dpf
namespace detail
{
template <typename Iterator>
struct extract_bit_simde_node
{
bool operator()(Iterator it) const
{
auto buf = reinterpret_cast<const char *>(&*it);
return buf[0] & 1;
}
};
template <typename NodeT, typename Iterator>
struct extract_bit;
@ -72,11 +63,23 @@ HEDLEY_PRAGMA(GCC diagnostic push)
HEDLEY_PRAGMA(GCC diagnostic ignored "-Wignored-attributes")
template <typename Iterator>
struct extract_bit<simde__m128i, Iterator>
: public extract_bit_simde_node<Iterator> { };
{
bool operator()(Iterator it) const
{
// Same lo-bit extract used by tree walk / correction packing.
return static_cast<bool>(dpf::get_lo_bit(*it));
}
};
template <typename Iterator>
struct extract_bit<simde__m256i, Iterator>
: public extract_bit_simde_node<Iterator> { };
{
bool operator()(Iterator it) const
{
auto buf = reinterpret_cast<const char *>(&*it);
return buf[0] & 1;
}
};
HEDLEY_PRAGMA(GCC diagnostic pop)
} // namespace detail

161
include/dpf/aes_sbox_bp.hpp Normal file
View file

@ -0,0 +1,161 @@
/// @file dpf/aes_sbox_bp.hpp
/// @brief Boyar–Peralta AES S-box (Yale CMT SLP, 32 ANDs) over XOR bit shares.
/// @details Circuit: http://www.cs.yale.edu/homes/peralta/CircuitStuff/SLP_AES_113.txt
/// (Boyar and Peralta, ePrint 2011/332). Matches the AES S-box used
/// by `aes_ref` / hardware AES. `#` is XNOR.
#ifndef LIBDPF_INCLUDE_DPF_AES_SBOX_BP_HPP__
#define LIBDPF_INCLUDE_DPF_AES_SBOX_BP_HPP__
#include <cstddef>
#include <cstdint>
namespace aes_bp
{
inline constexpr std::size_t wire_count = 121;
inline constexpr std::size_t op_count = 113;
inline constexpr std::size_t and_count = 32;
/// @brief Multiplicative depth of the SLP (XOR/XNOR are free).
inline constexpr std::size_t and_layer_count = 6;
inline constexpr std::uint8_t out_wire[8] = {114,117,119,107,116,120,115,111};
/// @brief AND-op indices in `ops` grouped by multiplicative depth.
/// @details XOR/XNOR between layers stay local. Every AND in a layer has
/// both inputs ready, so one exchange covers the whole layer
/// (and every S-box byte sharing that schedule).
inline constexpr std::uint8_t and_layer_size[and_layer_count] = {9, 1, 2, 7, 5, 8};
inline constexpr std::uint8_t and_layer_ops[and_layer_count][9] = {
{23, 24, 26, 28, 29, 31, 33, 34, 36},
{47},
{49, 53},
{57, 69, 72, 73, 78, 81, 82},
{60, 67, 68, 76, 77},
{70, 71, 74, 75, 79, 80, 83, 84},
};
static_assert(
and_layer_size[0] + and_layer_size[1] + and_layer_size[2]
+ and_layer_size[3] + and_layer_size[4] + and_layer_size[5]
== and_count,
"Boyar–Peralta AND layer sizes must cover every AND");
// kind: 0=XOR, 1=AND, 2=XNOR. Each row is {kind, dst, a, b}.
inline constexpr std::uint8_t ops[op_count][4] = {
{0, 8, 3, 5},
{0, 9, 0, 6},
{0, 10, 0, 3},
{0, 11, 0, 5},
{0, 12, 1, 2},
{0, 13, 12, 7},
{0, 14, 13, 3},
{0, 15, 9, 8},
{0, 16, 13, 0},
{0, 17, 13, 6},
{0, 18, 17, 11},
{0, 19, 4, 15},
{0, 20, 19, 5},
{0, 21, 19, 1},
{0, 22, 20, 7},
{0, 23, 20, 12},
{0, 24, 21, 10},
{0, 25, 7, 24},
{0, 26, 23, 24},
{0, 27, 23, 11},
{0, 28, 12, 24},
{0, 29, 9, 28},
{0, 30, 0, 28},
{1, 31, 15, 20},
{1, 32, 18, 22},
{0, 33, 32, 31},
{1, 34, 14, 7},
{0, 35, 34, 31},
{1, 36, 9, 28},
{1, 37, 17, 13},
{0, 38, 37, 36},
{1, 39, 16, 25},
{0, 40, 39, 36},
{1, 41, 10, 24},
{1, 42, 8, 26},
{0, 43, 42, 41},
{1, 44, 11, 23},
{0, 45, 44, 41},
{0, 46, 33, 21},
{0, 47, 35, 45},
{0, 48, 38, 43},
{0, 49, 40, 45},
{0, 50, 46, 43},
{0, 51, 47, 27},
{0, 52, 48, 29},
{0, 53, 49, 30},
{0, 54, 50, 51},
{1, 55, 50, 52},
{0, 56, 53, 55},
{1, 57, 54, 56},
{0, 58, 57, 51},
{0, 59, 52, 53},
{0, 60, 51, 55},
{1, 61, 60, 59},
{0, 62, 61, 53},
{0, 63, 52, 62},
{0, 64, 56, 62},
{1, 65, 53, 64},
{0, 66, 65, 63},
{0, 67, 56, 65},
{1, 68, 58, 67},
{0, 69, 54, 68},
{0, 70, 69, 66},
{0, 71, 58, 62},
{0, 72, 58, 69},
{0, 73, 62, 66},
{0, 74, 71, 70},
{1, 75, 73, 20},
{1, 76, 66, 22},
{1, 77, 62, 7},
{1, 78, 72, 28},
{1, 79, 69, 13},
{1, 80, 58, 25},
{1, 81, 71, 24},
{1, 82, 74, 26},
{1, 83, 70, 23},
{1, 84, 73, 15},
{1, 85, 66, 18},
{1, 86, 62, 14},
{1, 87, 72, 9},
{1, 88, 69, 17},
{1, 89, 58, 16},
{1, 90, 71, 10},
{1, 91, 74, 8},
{1, 92, 70, 11},
{0, 93, 90, 91},
{0, 94, 85, 93},
{0, 95, 84, 94},
{0, 96, 75, 77},
{0, 97, 76, 75},
{0, 98, 78, 79},
{0, 99, 87, 96},
{0, 100, 82, 98},
{0, 101, 83, 99},
{0, 102, 100, 101},
{0, 103, 98, 97},
{0, 104, 78, 80},
{0, 105, 88, 93},
{0, 106, 96, 104},
{0, 107, 95, 103},
{0, 108, 81, 100},
{0, 109, 89, 102},
{0, 110, 105, 106},
{2, 111, 87, 110},
{0, 112, 90, 108},
{0, 113, 94, 86},
{0, 114, 95, 108},
{2, 115, 102, 110},
{0, 116, 106, 107},
{2, 117, 107, 108},
{0, 118, 109, 112},
{2, 119, 118, 92},
{0, 120, 113, 109},
};
} // namespace aes_bp
#endif // LIBDPF_INCLUDE_DPF_AES_SBOX_BP_HPP__

633
include/dpf/app_flow.hpp Normal file
View file

@ -0,0 +1,633 @@
/// @file dpf/app_flow.hpp
/// @brief Measure an application plan on an explicit `run_config`.
/// @details `exercise_plan(p, ex, cfg)` drives both parties of `p` over
/// `cfg.kind` (in-process memory, sync streams, async memory, unix
/// sockets, TCP mux, parallel TCP, or SCTP) with the configured lanes,
/// framing, instances, window, chunking, pipelining, compute threads,
/// warmup, and trials. Link setup is outside the timed region. The
/// overloads without a config read `run_config::from_env()` once, for
/// the example binaries.
#ifndef LIBDPF_INCLUDE_DPF_APP_FLOW_HPP__
#define LIBDPF_INCLUDE_DPF_APP_FLOW_HPP__
#include <algorithm>
#include <atomic>
#include <chrono>
#include <cmath>
#include <condition_variable>
#include <cstddef>
#include <cstdint>
#include <cstdlib>
#include <cstring>
#include <exception>
#include <iostream>
#include <map>
#include <mutex>
#include <optional>
#include <string>
#include <thread>
#include <utility>
#include <vector>
#include "dpf/net/asio_ns.hpp"
#include "dpf/app_runtime.hpp"
#include "dpf/compose.hpp"
#include "dpf/compose_async.hpp"
#include "dpf/experiment.hpp"
#include "dpf/net/async_round_sink.hpp"
#include "dpf/net/memory_sink.hpp"
#include "dpf/net/round_lane.hpp"
#include "dpf/net/stream_array.hpp"
#include "dpf/online_session.hpp"
#include "dpf/party_runner.hpp"
#include "dpf/run_config.hpp"
#include "dpf/run_log.hpp"
namespace dpf
{
namespace app
{
using transport_kind = net::transport;
inline const char * transport_name(transport_kind t) noexcept
{
return net::transport_name(t);
}
/// @brief `DPF_TRANSPORT` (default `async`). Unknown names throw.
inline transport_kind transport_from_env()
{
return run_config::from_env().kind;
}
/// @brief Drive a plan through `schedule_session` on an `async_round_sink`.
inline void drive_via_schedule_async(const protocol::plan & p,
net::async_round_sink & sink,
std::vector<std::vector<std::uint8_t>> & values,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels,
std::size_t party, const protocol::drive_options & opt = {})
{
protocol::drive_via_schedule(p, sink, values, kernels, party, opt);
}
/// @brief What one `exercise_plan` measured.
/// @details `bytes` is the plan's slot bytes (one instance). `wire_*` are
/// party 0's link counters including every header. `wall_ns` is the
/// median party-0 drive time over the timed trials.
struct cost
{
std::size_t rounds = 0;
std::size_t bytes = 0;
std::uint64_t wire_out = 0;
std::uint64_t wire_in = 0;
std::uint64_t frames_out = 0;
std::uint64_t frames_in = 0;
std::uint64_t write_calls = 0;
std::uint64_t wall_ns = 0;
};
inline cost plan_cost(const protocol::plan & p)
{
cost out;
out.rounds = p.rounds();
for (auto n : p.slot_bytes_all())
out.bytes += n;
return out;
}
namespace detail
{
struct once
{
std::uint64_t wall_ns = 0;
net::stream_stats wire;
/// Every party's drive time (`party_walls[0] == wall_ns`).
std::vector<std::uint64_t> party_walls;
};
/// @brief Parties 0 and 1 on two threads over a paired in-process sink. Party
/// `p` draws from `seeds->derive_party(p)` when `seeds` is set; `ex`
/// records party 0 and receives both parties' noted seeds.
template <typename Drive>
inline once two_threads(experiment * ex, const experiment * seeds, Drive drive)
{
once out;
out.party_walls.assign(2, 0);
std::exception_ptr err[2];
std::vector<experiment::noted_seed> noted[2];
start_gate gate(2);
auto side = [&](unsigned party) {
try
{
std::optional<experiment> stream;
if (seeds != nullptr)
stream.emplace(seeds->derive_party(party));
if (!gate.arrive_and_wait())
throw gate_broken();
experiment * mine = party == 0 ? ex : nullptr;
protocol::round_probe probe{};
if (mine != nullptr)
{
probe = mine->probe();
mine->begin_timing();
}
const auto a = std::chrono::steady_clock::now();
drive(party, mine != nullptr ? &probe : nullptr);
out.party_walls[party] = static_cast<std::uint64_t>(
std::chrono::duration_cast<std::chrono::nanoseconds>(
std::chrono::steady_clock::now() - a)
.count());
if (mine != nullptr)
mine->end_timing();
if (stream)
noted[party] = stream->seeds();
}
catch (...)
{
err[party] = std::current_exception();
gate.fail();
}
};
std::thread t0(side, 0u);
std::thread t1(side, 1u);
t0.join();
t1.join();
for (auto & e : err)
{
if (!e)
continue;
try
{
std::rethrow_exception(e);
}
catch (const gate_broken &)
{
continue;
}
catch (...)
{
}
std::rethrow_exception(e);
}
out.wall_ns = out.party_walls[0];
if (ex != nullptr && seeds != nullptr)
for (unsigned p = 0; p < 2; ++p)
ex->fold_seeds("p" + std::to_string(p), noted[p]);
return out;
}
/// @brief One run of `plans` on `cfg.kind` from fresh copies of `inputs`;
/// `ex` records party 0 when set, and parties draw from `seeds`'s
/// derived streams when it is set.
inline once run_once(const std::vector<protocol::plan> & plans,
const std::vector<party_values> & inputs,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels,
const run_config & cfg, experiment * ex, const experiment * seeds = nullptr)
{
std::vector<party_values> values = inputs;
values.resize(plans.size());
const auto slots = plans[0].slot_bytes_all();
auto opt = cfg.drive();
const bool in_memory = cfg.kind == net::transport::memory_sink
|| cfg.kind == net::transport::memory_stream;
if (in_memory && plans.size() != 2)
throw std::invalid_argument(std::string("transport ")
+ net::transport_name(cfg.kind) + " runs two parties");
switch (cfg.kind)
{
case net::transport::memory_sink:
{
auto sl = slots.empty() ? std::vector<std::size_t>{0} : slots;
auto sinks = net::make_memory_sink_pair(cfg.instances, sl);
return two_threads(ex, seeds, [&](unsigned party, const protocol::round_probe * pr) {
auto o = opt;
o.probe = pr;
protocol::drive_via_schedule(plans[party],
party == 0 ? sinks.first : sinks.second, values[party], kernels,
party, o);
});
}
case net::transport::memory_stream:
{
const std::size_t nstreams =
slots.empty() ? 1 : protocol::lanes_for_plan(slots.size(), opt);
auto ends = net::make_memory_stream_pair(nstreams);
return two_threads(ex, seeds, [&](unsigned party, const protocol::round_probe * pr) {
auto o = opt;
o.probe = pr;
protocol::drive_plan_on_streams(plans[party],
party == 0 ? ends.first : ends.second, values[party], kernels,
party, cfg.instances, o);
});
}
default:
{
const auto r = run_parties(plans, values, kernels, cfg, ex, nullptr, seeds);
once out;
out.wall_ns = r.party0_wall_ns;
out.wire = r.wire[0];
out.party_walls = r.party_wall_ns;
return out;
}
}
}
} // namespace detail
/// @brief Drive every party's plan over `cfg` and return party 0's cost.
/// @details `plans[i]` is party `i`'s plan (2 or 3 parties); each trial starts
/// from a fresh copy of `inputs` (missing entries start empty).
/// Warmup runs are not timed; `wall_ns` is the median timed trial.
inline cost exercise_parties(const std::vector<protocol::plan> & plans,
const std::vector<party_values> & inputs,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels, experiment * ex,
const run_config & cfg)
{
if (plans.size() < 2 || plans.size() > 3)
throw std::invalid_argument("exercise_parties needs 2 or 3 plans");
const protocol::plan & p = plans[0];
cost out = plan_cost(p);
if (ex != nullptr)
{
ex->ingest_plan(p);
ex->set_config(cfg.describe());
}
if (p.slot_bytes_all().empty() && cfg.kind != net::transport::memory_sink)
{
DPF_LOG(warning, "trials.skipped")
.kv("experiment", ex != nullptr ? ex->name() : std::string("none"))
.kv("detail", "the plan has no exchange rounds; nothing was timed and "
"wall_ns stays 0");
return out;
}
const std::size_t total = cfg.warmup + std::max<std::size_t>(1, cfg.trials);
std::vector<std::uint64_t> walls;
std::vector<std::uint64_t> slowest;
detail::once last;
for (std::size_t t = 0; t < total; ++t)
{
const bool timed = t >= cfg.warmup;
const bool record = t + 1 == total;
last = detail::run_once(plans, inputs, kernels, cfg, record ? ex : nullptr, ex);
DPF_LOG(debug, "trial").kv("index", t).kv("timed", timed)
.kv("instrumented", record && ex != nullptr).kv("p0_wall_ns", last.wall_ns)
.kv("wire_out", last.wire.bytes_out).kv("wire_in", last.wire.bytes_in);
if (timed)
{
walls.push_back(last.wall_ns);
std::uint64_t w = last.wall_ns;
for (auto p : last.party_walls)
w = std::max(w, p);
slowest.push_back(w);
if (ex != nullptr)
ex->add_trial(last.wall_ns, last.party_walls);
}
}
std::sort(walls.begin(), walls.end());
std::sort(slowest.begin(), slowest.end());
out.wall_ns = walls.empty() ? 0 : walls[walls.size() / 2];
if (!walls.empty() && log::enabled(log::level::info))
{
double mean = 0;
for (auto w : walls)
mean += static_cast<double>(w);
mean /= static_cast<double>(walls.size());
double var = 0;
for (auto w : walls)
var += (static_cast<double>(w) - mean) * (static_cast<double>(w) - mean);
const double sd = walls.size() > 1
? std::sqrt(var / static_cast<double>(walls.size() - 1)) : 0.0;
DPF_LOG(info, "trials")
.kv("experiment", ex != nullptr ? ex->name() : std::string("none"))
.kv("transport", net::transport_name(cfg.kind)).kv("n", walls.size())
.kv("warmup", cfg.warmup).kv("median_ns", out.wall_ns)
.kv("min_ns", walls.front()).kv("max_ns", walls.back())
.kv("mean_ns", mean).kv("sd_ns", sd)
.kv("slowest_median_ns", slowest[slowest.size() / 2])
.kv("instrumented_trial", "last")
.kv("links", "rebuilt per trial");
}
out.wire_out = last.wire.bytes_out;
out.wire_in = last.wire.bytes_in;
out.frames_out = last.wire.frames_out;
out.frames_in = last.wire.frames_in;
out.write_calls = last.wire.write_calls;
if (ex != nullptr)
{
experiment::wire_counts w;
w.bytes_out = last.wire.bytes_out;
w.bytes_in = last.wire.bytes_in;
w.payload_out = last.wire.payload_out;
w.payload_in = last.wire.payload_in;
w.frames_out = last.wire.frames_out;
w.frames_in = last.wire.frames_in;
w.write_calls = last.wire.write_calls;
ex->set_wire(w);
}
return out;
}
/// @brief Drive `p` as both parties' plan over `cfg` and return its cost.
inline cost exercise_plan(const protocol::plan & p, experiment * ex,
const run_config & cfg,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels = {})
{
return exercise_parties({p, p}, {}, kernels, ex, cfg);
}
/// @brief `exercise_plan` on `run_config::from_env()`.
inline cost exercise_plan(const protocol::plan & p, experiment * ex = nullptr)
{
return exercise_plan(p, ex, run_config::from_env());
}
inline experiment measure_plan(const char * name, const protocol::plan & p,
const run_config & cfg)
{
experiment ex(name, "p0");
(void)exercise_plan(p, &ex, cfg);
return ex;
}
inline experiment measure_plan(const char * name, const protocol::plan & p)
{
return measure_plan(name, p, run_config::from_env());
}
/// @brief Schedule `c`, then `exercise_plan`.
inline cost exercise(protocol::composer & c)
{
return exercise_plan(c.default_plan());
}
/// @brief Exercise `p` and require `expect_rounds`. Prints `name rounds= bytes=`.
inline int run_plan(const char * name, const protocol::plan & p,
std::size_t expect_rounds, const run_config & cfg = run_config::from_env())
{
try
{
start_logging(cfg);
const cost got = exercise_plan(p, nullptr, cfg);
std::cout << name << " rounds=" << got.rounds << " bytes=" << got.bytes
<< " wire_out=" << got.wire_out << " wire_in=" << got.wire_in
<< "\n";
if (got.rounds != expect_rounds)
{
std::cerr << name << " expected " << expect_rounds << " rounds\n";
return 1;
}
}
catch (const std::exception & ex)
{
DPF_LOG(error, "run.failed").kv("name", name).kv("what", ex.what());
std::cerr << name << " flow: " << ex.what() << "\n";
return 1;
}
return 0;
}
/// @brief Measure `p`, print cost, and write CSVs under `DPF_EXPERIMENT_DIR`.
/// @details When `prep` is set, the prep for that demand is dealt and shipped
/// over `cfg`'s transport and its size is printed.
inline int run_measured(const char * name, const protocol::plan & p,
std::size_t expect_rounds, const run_config & cfg = run_config::from_env(),
const prep::demand * prep = nullptr)
{
try
{
start_logging(cfg);
auto ex = measure_plan(name, p, cfg);
std::cout << name << " transport=" << net::transport_name(cfg.kind)
<< " rounds=" << ex.interactive_rounds()
<< " bytes=" << ex.plan_bytes_out()
<< " wire_out=" << ex.wire().bytes_out
<< " wire_in=" << ex.wire().bytes_in
<< " wall_ns=" << ex.wall_ns()
<< " median_ns=" << ex.median_trial_ns()
<< " cpu_ns=" << ex.cpu_ns()
<< " prg_evals=" << ex.prg_evals()
<< " random_bytes=" << ex.random_bytes()
<< " seed=" << ex.seed_hex();
if (prep != nullptr)
{
const auto shipped = session::ship_prep(*prep, cfg);
std::cout << " prep_bytes=" << shipped.bytes0;
}
std::cout << "\n";
if (ex.interactive_rounds() != expect_rounds)
{
std::cerr << name << " expected " << expect_rounds << " rounds\n";
return 1;
}
if (const char * dir = std::getenv("DPF_EXPERIMENT_DIR"))
{
if (dir[0] != '\0')
ex.write_csv(dir);
}
}
catch (const std::exception & ex)
{
DPF_LOG(error, "run.failed").kv("name", name).kv("what", ex.what());
std::cerr << name << " measure: " << ex.what() << "\n";
return 1;
}
return 0;
}
/// @brief Exercise `c` and require `expect_rounds`.
inline int run(const char * name, protocol::composer & c, std::size_t expect_rounds)
{
return run_plan(name, c.default_plan(), expect_rounds);
}
/// @brief `run_measured` on `c.default_plan()`.
inline int run_measured(const char * name, protocol::composer & c,
std::size_t expect_rounds)
{
return run_measured(name, c.default_plan(), expect_rounds);
}
/// @brief Many instances, both parties, delays that do not line up.
/// @details A worker never sits on a side whose peer is the one that still
/// has to submit. Parked receives yield the thread to whichever
/// side is behind, so a slow step does not stall the rest and a
/// round-robin that always resumes the waiter cannot deadlock.
inline void run_fleet(protocol::composer & c, std::size_t instances,
std::uint32_t chaos_seed)
{
if (instances == 0)
throw std::invalid_argument("run_fleet needs instances");
auto plan = c.default_plan();
auto slots = plan.slot_bytes_all();
if (slots.empty())
slots.push_back(0);
struct side
{
protocol::drive_cursor cursor{};
std::vector<std::vector<std::uint8_t>> values;
net::memory_sink * sink = nullptr;
std::chrono::steady_clock::time_point ready_at{};
bool busy = false;
};
struct inst
{
std::pair<net::memory_sink, net::memory_sink> sinks;
side party[2]{};
};
std::vector<inst> all;
all.reserve(instances);
auto chaos_us = [&](std::size_t id, int party, std::size_t step) {
std::uint32_t x = chaos_seed
^ static_cast<std::uint32_t>(id * 0x9E3779B9u)
^ static_cast<std::uint32_t>(party * 0x85EBCA6Bu)
^ static_cast<std::uint32_t>(step * 0xC2B2AE35u);
x ^= x << 13;
x ^= x >> 17;
// Mostly short, with occasional long stalls. Not sorted by instance.
const std::uint32_t bucket = x % 17u;
if (party == 0 && step == 0 && (id % 5u) == 0)
return 12000;
if (bucket == 0)
return 1500;
if (bucket < 4)
return static_cast<int>(x % 400u);
return static_cast<int>(x % 40u);
};
for (std::size_t i = 0; i < instances; ++i)
{
all.push_back(inst{net::make_memory_sink_pair(1, slots), {}});
auto & created = all.back();
created.party[0].sink = &created.sinks.first;
created.party[1].sink = &created.sinks.second;
const auto now = std::chrono::steady_clock::now();
created.party[0].ready_at = now + std::chrono::microseconds(
chaos_us(i, 0, 0));
created.party[1].ready_at = now + std::chrono::microseconds(
chaos_us(i, 1, 0));
}
std::mutex mu;
std::condition_variable cv;
std::map<std::uint32_t, protocol::kernel_fn> kernels;
std::size_t finished = 0;
const std::size_t workers = std::min<std::size_t>(
std::thread::hardware_concurrency() == 0
? 4
: std::thread::hardware_concurrency(),
instances);
auto pick = [&](std::unique_lock<std::mutex> & lock) -> side * {
for (;;)
{
const auto now = std::chrono::steady_clock::now();
side * best = nullptr;
int best_score = 0x7fffffff;
std::chrono::steady_clock::time_point soonest =
now + std::chrono::hours(1);
bool any_left = false;
for (auto & item : all)
{
for (int p = 0; p < 2; ++p)
{
side & s = item.party[p];
if (s.cursor.done)
continue;
if (s.busy)
{
any_left = true;
continue;
}
any_left = true;
if (s.ready_at > now)
{
if (s.ready_at < soonest)
soonest = s.ready_at;
continue;
}
// A side still submitting outranks one parked on a receive.
// Among those, the one further behind runs first.
const int score = static_cast<int>(s.cursor.exchange_i)
+ (s.cursor.awaiting_peer ? 100000 : 0);
if (score < best_score)
{
best_score = score;
best = &s;
}
}
}
if (best != nullptr)
{
best->busy = true;
return best;
}
if (!any_left)
return nullptr;
cv.wait_until(lock, soonest);
}
};
auto worker = [&] {
std::unique_lock<std::mutex> lock(mu);
for (;;)
{
side * s = pick(lock);
if (s == nullptr)
return;
std::size_t inst_i = 0;
int party = 0;
for (; inst_i < all.size(); ++inst_i)
{
if (&all[inst_i].party[0] == s)
{
party = 0;
break;
}
if (&all[inst_i].party[1] == s)
{
party = 1;
break;
}
}
protocol::drive_options opt;
opt.park_if_waiting = true;
opt.one_exchange = true;
opt.cursor = &s->cursor;
auto * sink = s->sink;
auto * values = &s->values;
const std::size_t step = s->cursor.exchange_i;
lock.unlock();
protocol::drive(plan, *sink, *values, kernels,
static_cast<std::size_t>(party), opt);
lock.lock();
s->busy = false;
if (s->cursor.done)
++finished;
else
{
s->ready_at = std::chrono::steady_clock::now()
+ std::chrono::microseconds(chaos_us(inst_i, party, step));
}
cv.notify_all();
}
};
std::vector<std::thread> pool;
pool.reserve(workers);
for (std::size_t i = 0; i < workers; ++i)
pool.emplace_back(worker);
for (auto & th : pool)
th.join();
if (finished != instances * 2)
throw std::runtime_error("run_fleet: not every side finished");
}
} // namespace app
} // namespace dpf
#endif

312
include/dpf/app_plans.hpp Normal file
View file

@ -0,0 +1,312 @@
/// @file dpf/app_plans.hpp
/// @brief Named compose plans and star drivers for the application sketches.
#ifndef LIBDPF_INCLUDE_DPF_APP_PLANS_HPP__
#define LIBDPF_INCLUDE_DPF_APP_PLANS_HPP__
#include <cstddef>
#include <cstdint>
#include <functional>
#include <memory>
#include <stdexcept>
#include <utility>
#include <vector>
#include "dpf/compose.hpp"
#include "dpf/mesh_apps.hpp"
#include "dpf/net/edge_mesh.hpp"
#include "dpf/pad_graphs.hpp"
#include "dpf/protocol.hpp"
#include "dpf/session_host.hpp"
namespace dpf
{
namespace protocol
{
/// @brief Drive every session until all instances are done (interleaved).
/// @details Callers must `submit` each live index first (see `submit_and_drive`).
inline void drive_sessions(std::vector<schedule_session *> sessions,
unsigned max_spins = 100000u)
{
if (sessions.empty())
return;
for (auto * s : sessions)
{
if (s == nullptr)
throw std::invalid_argument("drive_sessions null");
}
for (unsigned spins = 0;; ++spins)
{
if (spins > max_spins)
throw std::runtime_error("drive_sessions: peer not ready");
bool all_done = true;
for (auto * s : sessions)
{
s->drive();
for (std::size_t i = 0; i < s->count(); ++i)
all_done = all_done && s->done(i);
}
if (all_done)
return;
}
}
/// @brief Submit every unfinished index, then `drive_sessions`.
inline void submit_and_drive(std::vector<schedule_session *> sessions,
unsigned max_spins = 100000u)
{
for (auto * s : sessions)
{
if (s == nullptr)
throw std::invalid_argument("submit_and_drive null");
for (std::size_t i = 0; i < s->count(); ++i)
if (!s->done(i))
s->submit(i);
}
drive_sessions(std::move(sessions), max_spins);
}
/// @brief Client↔N-server star: build sessions from round lists and drive.
inline void drive_star(net::memory_star & star,
std::vector<schedule_round> client_rounds,
const std::function<std::vector<schedule_round>(std::size_t server_i)> &
server_rounds_for,
unsigned max_spins = 100000u)
{
schedule_session client(1, star.client_mesh(), std::move(client_rounds),
false);
std::vector<schedule_session> servers;
servers.reserve(star.servers);
for (std::size_t i = 0; i < star.servers; ++i)
servers.emplace_back(1, star.server_edge(i), server_rounds_for(i),
false);
std::vector<schedule_session *> ptrs;
ptrs.reserve(1 + servers.size());
ptrs.push_back(&client);
for (auto & s : servers)
ptrs.push_back(&s);
submit_and_drive(std::move(ptrs), max_spins);
}
/// @brief N-server PIR / keyword PIR compose plan (`client_servers` waves).
inline plan n_server_pir_plan(std::size_t party, std::size_t n_servers,
std::size_t query_bytes, std::size_t answer_bytes)
{
composer c(party);
auto q = c.client_servers(n_servers, query_bytes, answer_bytes);
(void)q;
return c.default_plan();
}
inline plan keyword_pir_compose_plan(std::size_t party, std::size_t depth,
std::size_t answer_bytes = sizeof(int))
{
const std::size_t query_bytes = 16 + depth * 16;
return n_server_pir_plan(party, 2, query_bytes, answer_bytes);
}
/// @brief Express / Sabre online walk: fused CW + trailer (sketch/proof).
inline plan mailbox_write_fused_plan(std::size_t party, std::size_t depth = 8,
std::size_t cw_bytes = 16, std::size_t trailer_bytes = 8)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
auto trailer = c.input(domain::a, trailer_bytes);
auto walk = c.fss_point_fused(seed, depth, cw_bytes, trailer);
(void)walk;
return c.default_plan();
}
/// @brief BitMore: L parallel bit-point walks (CSE → one wave per depth).
inline plan bitmore_fan_plan(std::size_t party, std::size_t L = 4,
std::size_t depth = 6)
{
composer c(party);
auto seeds = c.fan(L, [&](std::size_t) {
return c.input(domain::fss, 16);
});
(void)c.fan(L, [&](std::size_t i) {
return c.fss_point(seeds[i], depth, 16);
});
return c.default_plan();
}
/// @brief BitMore / PIRsona star key-ship + answer schedule rounds.
inline std::vector<schedule_round> bitmore_star_fetch(std::size_t L,
std::size_t seed_bytes, std::size_t answer_bytes,
const std::shared_ptr<std::vector<std::vector<std::uint8_t>>> & seeds,
const std::shared_ptr<std::vector<std::vector<std::uint8_t>>> & answers)
{
return pirsona_bitmore_fetch(L, seed_bytes, answer_bytes, seeds, answers);
}
/// @brief SUBLEQ one instruction: prepaid expand + cmp branch skeleton.
inline plan subleq_instruction_plan(std::size_t party, std::size_t addr_depth = 8,
std::size_t slot_bytes = 16)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
auto prepaid = c.defer_expand(seed, addr_depth, slot_bytes);
auto branch = c.fss_cmp(seed, addr_depth, slot_bytes);
(void)prepaid;
(void)branch;
return c.default_plan();
}
/// @brief Queue `n_insns` SUBLEQ instruction plans on a peer `session_host`.
inline void subleq_run_instructions(session_host & host, std::size_t n_insns,
std::size_t addr_depth = 8, std::size_t slot_bytes = 16)
{
for (std::size_t i = 0; i < n_insns; ++i)
host.push(subleq_instruction_plan(host.party(), addr_depth, slot_bytes));
host.drive_until_idle();
}
/// @brief Pika early-stop unit walk (`early_stop` bits pack into the leaf).
inline plan pika_lookup_plan(std::size_t party, std::size_t full_depth = 8,
std::size_t early_stop = 3)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
auto tip = c.fss_point_early_stop(seed, full_depth, early_stop, 16);
(void)tip;
return c.default_plan();
}
/// @brief Pika with dealer pad delivery of the unit key before the walk.
inline plan pika_dealer_lookup_plan(std::size_t party, std::size_t full_depth = 8,
std::size_t early_stop = 3)
{
composer c(party);
auto key = c.input(domain::fss, 16);
auto delivered = c.dealer_deliver(key);
auto tip = c.fss_point_early_stop(delivered, full_depth, early_stop, 16);
(void)tip;
return c.default_plan();
}
/// @brief Duoram write: path CWs with leaf applied later.
inline plan duoram_update_plan(std::size_t party, std::size_t depth = 8,
std::size_t slot_bytes = 16)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
auto walk = c.leaf_later_walk(seed, depth, slot_bytes);
auto leaf = c.input(domain::a, slot_bytes);
auto applied = c.apply_leaf_correction(walk.values, walk.control, leaf);
(void)applied;
return c.default_plan();
}
/// @brief Poplar / Prio / Mastic: prefix checkpoints each level.
inline plan poplar_prefix_plan(std::size_t party, std::size_t depth = 8,
std::size_t slot_bytes = 16, std::size_t prefix_bytes = 8)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
auto prefixes = c.level_walk_prefixes(seed, depth, slot_bytes, prefix_bytes);
(void)prefixes;
return c.default_plan();
}
/// @brief Weighted / multi-seed prefix fan (Prio multi-client shape).
inline plan poplar_prefix_fan_plan(std::size_t party, std::size_t n_keys,
std::size_t depth = 8, std::size_t slot_bytes = 16,
std::size_t prefix_bytes = 8)
{
composer c(party);
(void)c.fan(n_keys, [&](std::size_t) {
auto seed = c.input(domain::fss, 16);
return c.level_walk_prefixes(seed, depth, slot_bytes, prefix_bytes)
.at_level.back();
});
return c.default_plan();
}
/// @brief Ledger (2,3): verifiable DPF3 append — upload + proof exchange.
inline plan ledger23_append_plan(std::size_t party, std::size_t depth = 8,
std::size_t proof_bytes = 32)
{
composer c(party);
const std::size_t query_bytes = 16 + depth * 16;
auto up = c.client_servers(3, query_bytes, proof_bytes);
(void)up;
return c.default_plan();
}
/// @brief Floram DS keygen walk (read/write expand stay local).
inline plan floram_ds_plan(std::size_t party, std::size_t depth = 8,
std::size_t slot_bytes = 16, bool oh = false)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
auto tip = c.level_walk_ds(seed, depth, slot_bytes, oh);
(void)tip;
return c.default_plan();
}
/// @brief Splinter-style point walk (server expand cost).
inline plan fss_point_plan(std::size_t party, std::size_t depth = 8,
std::size_t slot_bytes = 16)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
auto tip = c.fss_point(seed, depth, slot_bytes);
(void)tip;
return c.default_plan();
}
/// @brief Waldo-style comparison walk.
inline plan fss_cmp_plan(std::size_t party, std::size_t depth = 8,
std::size_t slot_bytes = 16)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
auto tip = c.fss_cmp(seed, depth, slot_bytes);
(void)tip;
return c.default_plan();
}
/// @brief Range-count: two parallel cmp walks.
inline plan range_count_plan(std::size_t party, std::size_t depth = 8,
std::size_t slot_bytes = 16)
{
composer c(party);
auto seed = c.input(domain::fss, 16);
(void)c.fan(2, [&](std::size_t) {
return c.fss_cmp(seed, depth, slot_bytes);
});
return c.default_plan();
}
/// @brief PSI cuckoo probes → multipoint fan (occupied bucket ids).
inline plan psi_cuckoo_plan(std::size_t party,
const std::vector<std::size_t> & probes, std::size_t depth = 8,
std::size_t slot_bytes = 16, std::size_t answer_bytes = 8)
{
composer c(party);
auto mr = c.schedule_cuckoo_probes(probes,
[&](std::size_t) { return c.input(domain::fss, 16); }, depth,
slot_bytes, answer_bytes);
(void)mr;
return c.default_plan();
}
/// @brief idpf_agg: adaptive prefix retain loop (`n_bits` online rounds).
inline plan idpf_agg_plan(std::size_t party, std::size_t n_bits = 16,
std::size_t slot_bytes = 16, std::size_t prefix_bytes = 8)
{
composer c(party);
auto f = c.begin_adaptive_prefix(c.input(domain::fss, 16));
for (std::size_t bit = 0; bit < n_bits; ++bit)
{
f = c.step_adaptive_prefix(f, slot_bytes, prefix_bytes);
f = c.retain_adaptive_prefix(f, 0);
}
return c.default_plan();
}
} // namespace protocol
} // namespace dpf
#endif // LIBDPF_INCLUDE_DPF_APP_PLANS_HPP__

144
include/dpf/app_runtime.hpp Normal file
View file

@ -0,0 +1,144 @@
/// @file dpf/app_runtime.hpp
/// @brief File prep and two-party synchronous mux helpers.
/// @details `stream_runtime` configures only this synchronous TCP mux path.
/// For any other transport, party count, or a dealer, use
/// `app::run_config` with `app::run_parties` (`party_runner.hpp`).
/// The synchronous mux speaks the same frames as the async mux.
#ifndef LIBDPF_INCLUDE_DPF_APP_RUNTIME_HPP__
#define LIBDPF_INCLUDE_DPF_APP_RUNTIME_HPP__
#include <atomic>
#include <cstddef>
#include <cstdint>
#include <map>
#include <mutex>
#include <stdexcept>
#include <string>
#include <thread>
#include <utility>
#include <vector>
#include "hedley/hedley.h"
#include "dpf/compose.hpp"
#include "dpf/net/stream_array.hpp"
#include "dpf/net/sync_stream_array.hpp"
#include "dpf/party_run.hpp"
#include "dpf/prep_source.hpp"
#include "dpf/protocol_roles.hpp"
namespace dpf
{
namespace app
{
/// @brief File prep basename and synchronous mux settings (two parties).
struct stream_runtime
{
std::string prep_basename;
std::string mux_host = "127.0.0.1";
unsigned short mux_port = 0;
std::size_t mux_nstreams = 1;
};
/// @brief Read one party's prep blob from `basename-p0` / `basename-p1`.
HEDLEY_WARN_UNUSED_RESULT
inline std::vector<std::uint8_t> open_file_prep(const std::string & basename,
unsigned party)
{
return prep::read_file(basename + "-p" + std::to_string(party));
}
/// @brief Same as `open_file_prep`, parsed as a prep cursor.
HEDLEY_WARN_UNUSED_RESULT
inline prep::cursor open_file_prep_cursor(const std::string & basename,
unsigned party)
{
return prep::cursor(open_file_prep(basename, party));
}
/// @brief Run `fn(party, mux_stream_array &)` over one TCP connection with mux.
template <typename Fn>
void with_tcp_mux_peer(unsigned party, std::string host,
std::atomic<unsigned short> & port, std::size_t nstreams, Fn && fn)
{
dpf::run::tcp_pair_mux(party, std::move(host), port, nstreams,
std::forward<Fn>(fn));
}
/// @brief Convenience: mux settings from `stream_runtime`.
template <typename Fn>
void with_tcp_mux_peer(unsigned party, const stream_runtime & rt,
std::size_t nstreams, Fn && fn)
{
std::atomic<unsigned short> port{rt.mux_port};
with_tcp_mux_peer(party, rt.mux_host, port, nstreams, std::forward<Fn>(fn));
}
/// @brief One party: TCP mux + `drive_plan_on_streams`.
inline void drive_plan_mux(const protocol::plan & plan,
net::stream_array & mux, std::vector<std::vector<std::uint8_t>> & values,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels, std::size_t party,
std::size_t lanes = 1, const protocol::drive_options & opt = {})
{
protocol::drive_plan_on_streams(plan, mux, values, kernels, party, lanes, opt);
}
/// @brief Two localhost threads on one mux TCP link, both plans driven on streams.
inline void drive_plan_mux_both(const protocol::plan & p0,
const protocol::plan & p1, std::vector<std::vector<std::uint8_t>> & v0,
std::vector<std::vector<std::uint8_t>> & v1,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels = {},
std::size_t lanes = 1, const protocol::drive_options & opt = {},
const stream_runtime & rt = {})
{
const auto slots0 = p0.slot_bytes_all();
const auto slots1 = p1.slot_bytes_all();
if (slots0 != slots1)
throw std::invalid_argument("drive_plan_mux_both: slot shapes differ");
const std::size_t nstreams =
slots0.empty() ? 1 : protocol::lanes_for_plan(slots0.size(), opt);
std::atomic<unsigned short> port{rt.mux_port};
std::exception_ptr err;
std::mutex err_mu;
auto note = [&](std::exception_ptr e) {
std::lock_guard<std::mutex> lock(err_mu);
if (!err)
err = std::move(e);
};
std::thread t0([&] {
try
{
with_tcp_mux_peer(0, rt.mux_host, port, nstreams,
[&](unsigned, net::mux_stream_array & mux) {
drive_plan_mux(p0, mux, v0, kernels, 0, lanes, opt);
});
}
catch (...)
{
note(std::current_exception());
}
});
std::thread t1([&] {
try
{
with_tcp_mux_peer(1, rt.mux_host, port, nstreams,
[&](unsigned, net::mux_stream_array & mux) {
drive_plan_mux(p1, mux, v1, kernels, 1, lanes, opt);
});
}
catch (...)
{
note(std::current_exception());
}
});
t0.join();
t1.join();
if (err)
std::rethrow_exception(err);
}
} // namespace app
} // namespace dpf
#endif

View file

@ -0,0 +1,777 @@
/// @file dpf/arith_garble.hpp
/// @brief Constant-round arithmetic garbling gadgets.
/// @details Mixed-modulus circuits: free addition, free multiplication by a
/// public constant coprime to the modulus, and a unary projection.
/// A projection of modulus `m` sends `m - 1` ciphertexts (row
/// reduction). A symmetric boolean gate, including a high fan-in AND
/// or a threshold, is a projection of a free sum. Multiplication in a
/// small prime field is the discrete-log reduction: project to the
/// exponent, add, project back, and suppress the zero cases.
///
/// Labels are vectors in `(Z_m)^k`. Digit 0 is the point-and-permute
/// color. The global offset `Δ_m` has color digit 1, so the color of
/// semantic value `s` is `τ + s`. Addition and public scaling are
/// componentwise in that group, which is what makes them free.
///
/// This is not an ABY2.0 session. A session opens one masked wire and
/// then multiplies interactively. These gadgets never open an
/// intermediate wire: the evaluator finishes from the garbled rows.
/// @note Marshall Ball, Tal Malkin, and Mike Rosulek, "Garbling Gadgets for
/// Boolean and Arithmetic Circuits," CCS 2016 (ePrint 2016/969). The
/// free-addition offset is the one they attribute to Malkin, Pastro, and
/// shelat. The interactive product in `beaver.hpp` remains Patra,
/// Schneider, Suresh, and Yalame, USENIX Security 2021.
/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors)
/// @license Released under a GNU General Public v2.0 (GPLv2) license;
/// see [LICENSE.md](@ref license) for details.
#ifndef LIBDPF_INCLUDE_DPF_ARITH_GARBLE_HPP__
#define LIBDPF_INCLUDE_DPF_ARITH_GARBLE_HPP__
#include <array>
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <stdexcept>
#include <utility>
#include <vector>
#include "hedley/hedley.h"
#include "simde/simde/x86/avx2.h"
#include "dpf/prg_aes.hpp"
#include "dpf/random.hpp"
namespace dpf
{
namespace arith_garble
{
/// @brief Digit width of one label, including the color digit.
/// @details Ball, Malkin, and Rosulek use `λ / log2(m)` payload digits so the
/// label is `λ` bits. This instantiation fixes the width. Digit 0 is
/// the color in every modulus.
inline constexpr std::size_t k_digits = 16;
inline constexpr std::uint16_t k_max_mod = 128;
struct lab
{
std::uint16_t mod = 0;
std::array<std::uint16_t, k_digits> d{};
};
/// @brief Open a masked color. Garbler holds `mask` (the color of semantic 0).
/// Evaluator holds `color`.
HEDLEY_WARN_UNUSED_RESULT
inline std::uint16_t open_shares(std::uint16_t mod, std::uint16_t mask,
std::uint16_t color)
{
if (mod == 0)
throw std::invalid_argument("arith_garble: modulus");
return static_cast<std::uint16_t>((color + mod - (mask % mod)) % mod);
}
namespace detail
{
inline void require_mod(std::uint16_t m)
{
if (m < 2 || m > k_max_mod)
throw std::invalid_argument("arith_garble: modulus");
}
HEDLEY_WARN_UNUSED_RESULT
inline std::uint16_t gcd_u(std::uint16_t a, std::uint16_t b)
{
while (b != 0)
{
const std::uint16_t t = static_cast<std::uint16_t>(a % b);
a = b;
b = t;
}
return a;
}
HEDLEY_WARN_UNUSED_RESULT
inline lab sample_lab(std::uint16_t mod, simde__m128i & seed, std::uint32_t & n)
{
lab out;
out.mod = mod;
for (std::size_t i = 0; i < k_digits; ++i)
{
const auto block = prg::aes128::eval(seed, n++);
std::uint64_t lo = 0;
std::memcpy(&lo, &block, sizeof(lo));
out.d[i] = static_cast<std::uint16_t>(lo % mod);
}
return out;
}
HEDLEY_WARN_UNUSED_RESULT
inline lab make_delta(std::uint16_t mod, simde__m128i & seed, std::uint32_t & n)
{
lab d = sample_lab(mod, seed, n);
d.d[0] = 1;
return d;
}
HEDLEY_WARN_UNUSED_RESULT
inline lab add_lab(const lab & a, const lab & b)
{
if (a.mod != b.mod)
throw std::invalid_argument("arith_garble: modulus");
lab out;
out.mod = a.mod;
for (std::size_t i = 0; i < k_digits; ++i)
out.d[i] = static_cast<std::uint16_t>(
(static_cast<unsigned>(a.d[i]) + b.d[i]) % a.mod);
return out;
}
HEDLEY_WARN_UNUSED_RESULT
inline lab sub_lab(const lab & a, const lab & b)
{
if (a.mod != b.mod)
throw std::invalid_argument("arith_garble: modulus");
lab out;
out.mod = a.mod;
for (std::size_t i = 0; i < k_digits; ++i)
out.d[i] = static_cast<std::uint16_t>(
(static_cast<unsigned>(a.d[i]) + a.mod - b.d[i]) % a.mod);
return out;
}
HEDLEY_WARN_UNUSED_RESULT
inline lab scale_lab(const lab & a, std::uint16_t c)
{
lab out;
out.mod = a.mod;
for (std::size_t i = 0; i < k_digits; ++i)
out.d[i] = static_cast<std::uint16_t>(
(static_cast<unsigned>(a.d[i]) * c) % a.mod);
return out;
}
HEDLEY_WARN_UNUSED_RESULT
inline lab neg_lab(const lab & a)
{
lab out;
out.mod = a.mod;
for (std::size_t i = 0; i < k_digits; ++i)
out.d[i] = static_cast<std::uint16_t>((a.mod - a.d[i]) % a.mod);
return out;
}
/// @brief Label of semantic `s` on a wire whose semantic-0 label is `zero`.
HEDLEY_WARN_UNUSED_RESULT
inline lab shift_lab(const lab & zero, const lab & delta, std::uint16_t s)
{
return add_lab(zero, scale_lab(delta, static_cast<std::uint16_t>(s % zero.mod)));
}
HEDLEY_WARN_UNUSED_RESULT
inline lab hash_lab(std::uint32_t gid, std::uint32_t which, const lab & in,
std::uint16_t out_mod)
{
const prg::purpose_scope counted(prg::purpose::hash);
simde__m128i acc = simde_mm_set_epi64x(
static_cast<std::int64_t>(gid), static_cast<std::int64_t>(which));
for (std::size_t i = 0; i < k_digits; i += 8)
{
simde__m128i chunk;
std::memcpy(&chunk, in.d.data() + i, sizeof(chunk));
acc = prg::aes128::eval(simde_mm_xor_si128(acc, chunk),
static_cast<psnip_uint32_t>(i + which));
}
lab out;
out.mod = out_mod;
for (std::size_t i = 0; i < k_digits; ++i)
{
const auto block = prg::aes128::eval(acc,
static_cast<psnip_uint32_t>(1000u + i + gid));
std::uint64_t lo = 0;
std::memcpy(&lo, &block, sizeof(lo));
out.d[i] = static_cast<std::uint16_t>(lo % out_mod);
}
return out;
}
HEDLEY_WARN_UNUSED_RESULT
inline std::uint16_t pow_mod(std::uint16_t base, std::uint16_t exp, std::uint16_t mod)
{
unsigned r = 1;
unsigned b = base % mod;
unsigned e = exp;
while (e != 0)
{
if ((e & 1u) != 0)
r = (r * b) % mod;
b = (b * b) % mod;
e >>= 1u;
}
return static_cast<std::uint16_t>(r);
}
HEDLEY_WARN_UNUSED_RESULT
inline bool is_prime(std::uint16_t p)
{
if (p < 2)
return false;
for (std::uint16_t i = 2; i * i <= p; ++i)
if (p % i == 0)
return false;
return true;
}
HEDLEY_WARN_UNUSED_RESULT
inline std::uint16_t primitive_root(std::uint16_t p)
{
std::vector<std::uint16_t> factors;
std::uint16_t n = static_cast<std::uint16_t>(p - 1);
for (std::uint16_t i = 2; i * i <= n; ++i)
{
if (n % i != 0)
continue;
factors.push_back(i);
while (n % i == 0)
n = static_cast<std::uint16_t>(n / i);
}
if (n > 1)
factors.push_back(n);
for (std::uint16_t g = 2; g < p; ++g)
{
bool ok = true;
for (std::uint16_t f : factors)
{
if (pow_mod(g, static_cast<std::uint16_t>((p - 1) / f), p) == 1)
{
ok = false;
break;
}
}
if (ok)
return g;
}
throw std::logic_error("arith_garble: primitive root");
}
struct proj_rows
{
std::vector<lab> row;
};
struct pass_rows
{
lab payload[2]{};
std::uint16_t flag_ct[2]{};
};
} // namespace detail
/// @brief One wire. Valid only for the circuit that minted it.
struct wire
{
std::uint32_t id = 0;
};
/// @brief Shares of one evaluation. `opened[i] = color[i] - mask[i]` mod the
/// output modulus. `mask` is the garbler's share. `color` is the
/// evaluator's share.
struct shares
{
std::vector<std::uint16_t> mask;
std::vector<std::uint16_t> color;
std::vector<std::uint16_t> opened;
std::vector<std::uint16_t> modulus;
/// @brief Projection rows on the wire (`m - 1` each) plus two per bit-scale.
std::size_t ciphertext_rows = 0;
};
/// @brief Straight-line mixed-modulus circuit.
/// @details Inputs are declared first. Every later wire names earlier wires.
class circuit
{
public:
enum class op : unsigned char
{
in = 0,
add = 1,
addk = 2,
scale = 3,
proj = 4,
pass = 5
};
struct node
{
op code = op::in;
std::uint16_t mod = 0;
std::uint32_t a = 0;
std::uint32_t b = 0;
std::uint16_t k = 0;
std::vector<std::uint16_t> phi;
};
HEDLEY_WARN_UNUSED_RESULT
wire input(std::uint16_t mod)
{
detail::require_mod(mod);
node n;
n.code = op::in;
n.mod = mod;
nodes_.push_back(std::move(n));
return wire{static_cast<std::uint32_t>(nodes_.size() - 1)};
}
HEDLEY_WARN_UNUSED_RESULT
wire add(wire x, wire y)
{
const node & a = at(x);
const node & b = at(y);
if (a.mod != b.mod)
throw std::invalid_argument("arith_garble: add modulus");
node n;
n.code = op::add;
n.mod = a.mod;
n.a = x.id;
n.b = y.id;
nodes_.push_back(std::move(n));
return wire{static_cast<std::uint32_t>(nodes_.size() - 1)};
}
/// @brief Add a public constant. No ciphertext and no evaluator label change.
HEDLEY_WARN_UNUSED_RESULT
wire add_const(wire x, std::uint16_t k)
{
const node & a = at(x);
node n;
n.code = op::addk;
n.mod = a.mod;
n.a = x.id;
n.k = static_cast<std::uint16_t>(k % a.mod);
nodes_.push_back(std::move(n));
return wire{static_cast<std::uint32_t>(nodes_.size() - 1)};
}
/// @brief Multiply by a public constant coprime to the modulus.
HEDLEY_WARN_UNUSED_RESULT
wire scale(wire x, std::uint16_t c)
{
const node & a = at(x);
c = static_cast<std::uint16_t>(c % a.mod);
if (detail::gcd_u(c, a.mod) != 1)
throw std::invalid_argument("arith_garble: scale not coprime");
node n;
n.code = op::scale;
n.mod = a.mod;
n.a = x.id;
n.k = c;
nodes_.push_back(std::move(n));
return wire{static_cast<std::uint32_t>(nodes_.size() - 1)};
}
/// @brief Unary map `phi : Z_mod(x) → Z_out`. `phi.size()` is the input modulus.
/// The garbled row count is `phi.size() - 1`.
HEDLEY_WARN_UNUSED_RESULT
wire project(wire x, std::uint16_t out_mod, std::vector<std::uint16_t> phi)
{
detail::require_mod(out_mod);
const node & a = at(x);
if (phi.size() != a.mod)
throw std::invalid_argument("arith_garble: projection table");
for (std::uint16_t v : phi)
if (v >= out_mod)
throw std::invalid_argument("arith_garble: projection image");
node n;
n.code = op::proj;
n.mod = out_mod;
n.a = x.id;
n.phi = std::move(phi);
nodes_.push_back(std::move(n));
return wire{static_cast<std::uint32_t>(nodes_.size() - 1)};
}
/// @brief `bit ? word : 0`. `bit` is mod 2. Two ciphertext rows.
HEDLEY_WARN_UNUSED_RESULT
wire bit_scale(wire word, wire bit)
{
const node & w = at(word);
const node & b = at(bit);
if (b.mod != 2)
throw std::invalid_argument("arith_garble: bit_scale bit");
node n;
n.code = op::pass;
n.mod = w.mod;
n.a = word.id;
n.b = bit.id;
nodes_.push_back(std::move(n));
return wire{static_cast<std::uint32_t>(nodes_.size() - 1)};
}
/// @brief AND (or threshold `t`) of 0/1 wires that already live in `Z_{b+1}`.
/// @details The sum is free. The only rows are the final projection, `b`
/// ciphertexts, as in Section 5 of Ball, Malkin, and Rosulek.
/// The wires must have modulus `bits.size() + 1` and semantic
/// values in `{0, 1}`.
HEDLEY_WARN_UNUSED_RESULT
wire threshold(const std::vector<wire> & bits, std::uint16_t t)
{
if (bits.empty() || bits.size() > k_max_mod - 1)
throw std::invalid_argument("arith_garble: threshold fan-in");
const auto mod = static_cast<std::uint16_t>(bits.size() + 1);
if (t > bits.size())
throw std::invalid_argument("arith_garble: threshold");
wire acc = bits[0];
if (at(acc).mod != mod)
throw std::invalid_argument("arith_garble: threshold modulus");
for (std::size_t i = 1; i < bits.size(); ++i)
{
if (at(bits[i]).mod != mod)
throw std::invalid_argument("arith_garble: threshold modulus");
acc = add(acc, bits[i]);
}
std::vector<std::uint16_t> phi(mod, 0);
phi[t] = 1;
return project(acc, 2, std::move(phi));
}
/// @brief Fan-in AND of mod-2 bits. Lifts into `Z_{b+1}`, then `threshold`.
HEDLEY_WARN_UNUSED_RESULT
wire fanin_and(const std::vector<wire> & bits)
{
if (bits.empty())
throw std::invalid_argument("arith_garble: and");
const auto mod = static_cast<std::uint16_t>(bits.size() + 1);
std::vector<wire> lifted;
lifted.reserve(bits.size());
for (wire b : bits)
{
if (at(b).mod != 2)
throw std::invalid_argument("arith_garble: and bit");
lifted.push_back(project(b, mod, {0, 1}));
}
return threshold(lifted, static_cast<std::uint16_t>(bits.size()));
}
/// @brief Product in a prime field, via discrete log. Mod-2 product is AND.
HEDLEY_WARN_UNUSED_RESULT
wire mul(wire x, wire y)
{
const node & a = at(x);
const node & b = at(y);
if (a.mod != b.mod)
throw std::invalid_argument("arith_garble: mul modulus");
const std::uint16_t p = a.mod;
if (p == 2)
{
auto lx = project(x, 3, {0, 1});
auto ly = project(y, 3, {0, 1});
auto s = add(lx, ly);
return project(s, 2, {0, 0, 1});
}
if (!detail::is_prime(p))
throw std::invalid_argument("arith_garble: mul prime");
const std::uint16_t g = detail::primitive_root(p);
std::vector<std::uint16_t> dlog(p, 0);
std::vector<std::uint16_t> exp(static_cast<std::size_t>(p - 1), 0);
unsigned acc = 1;
for (std::uint16_t e = 0; e < p - 1; ++e)
{
dlog[acc] = e;
exp[e] = static_cast<std::uint16_t>(acc);
acc = (acc * g) % p;
}
auto zx = project(x, 2, zero_flag(p));
auto zy = project(y, 2, zero_flag(p));
auto dx = project(x, static_cast<std::uint16_t>(p - 1), dlog);
auto dy = project(y, static_cast<std::uint16_t>(p - 1), std::move(dlog));
auto ds = add(dx, dy);
auto gpow = project(ds, p, std::move(exp));
auto z1 = project(zx, 3, {0, 1});
auto z2 = project(zy, 3, {0, 1});
auto zsum = add(z1, z2);
auto zor = project(zsum, 2, {0, 1, 1});
auto nz = project(zor, 2, {1, 0});
return bit_scale(gpow, nz);
}
void out(wire x)
{
at(x);
outs_.push_back(x.id);
}
const std::vector<node> & nodes() const noexcept { return nodes_; }
const std::vector<std::uint32_t> & outputs() const noexcept { return outs_; }
std::uint16_t modulus_at(wire x) const { return at(x).mod; }
private:
const node & at(wire x) const
{
if (x.id >= nodes_.size())
throw std::invalid_argument("arith_garble: wire");
return nodes_[x.id];
}
static std::vector<std::uint16_t> zero_flag(std::uint16_t p)
{
std::vector<std::uint16_t> z(p, 0);
z[0] = 1;
return z;
}
std::vector<node> nodes_;
std::vector<std::uint32_t> outs_;
};
namespace detail
{
struct garble_state
{
std::vector<lab> zero;
std::vector<lab> delta;
std::vector<char> have_delta;
std::vector<proj_rows> proj;
std::vector<pass_rows> pass;
simde__m128i seed{};
std::uint32_t n = 0;
lab & delta_of(std::uint16_t mod)
{
if (!have_delta[mod])
{
delta[mod] = make_delta(mod, seed, n);
have_delta[mod] = 1;
}
return delta[mod];
}
};
inline garble_state garble(const circuit & c)
{
garble_state st;
st.seed = dpf::uniform_sample<simde__m128i>();
st.zero.resize(c.nodes().size());
st.delta.assign(static_cast<std::size_t>(k_max_mod) + 1, lab{});
st.have_delta.assign(static_cast<std::size_t>(k_max_mod) + 1, 0);
st.proj.resize(c.nodes().size());
st.pass.resize(c.nodes().size());
const auto & nodes = c.nodes();
for (std::uint32_t i = 0; i < nodes.size(); ++i)
{
const auto & nd = nodes[i];
switch (nd.code)
{
case circuit::op::in:
st.zero[i] = sample_lab(nd.mod, st.seed, st.n);
(void)st.delta_of(nd.mod);
break;
case circuit::op::add:
st.zero[i] = add_lab(st.zero[nd.a], st.zero[nd.b]);
break;
case circuit::op::addk:
st.zero[i] = sub_lab(st.zero[nd.a],
scale_lab(st.delta_of(nd.mod), nd.k));
break;
case circuit::op::scale:
st.zero[i] = scale_lab(st.zero[nd.a], nd.k);
break;
case circuit::op::proj:
{
const std::uint16_t m = nodes[nd.a].mod;
const std::uint16_t nmod = nd.mod;
const lab & zin = st.zero[nd.a];
const lab & din = st.delta_of(m);
const lab & dout = st.delta_of(nmod);
const std::uint16_t tau = zin.d[0];
const std::uint16_t s0 =
static_cast<std::uint16_t>((m - (tau % m)) % m);
const lab label0 = shift_lab(zin, din, s0);
const lab h0 = hash_lab(i, 0, label0, nmod);
const lab decrypted = neg_lab(h0);
const std::uint16_t p0 = nd.phi[s0];
st.zero[i] = sub_lab(decrypted, scale_lab(dout, p0));
st.proj[i].row.resize(static_cast<std::size_t>(m - 1));
for (std::uint16_t color = 1; color < m; ++color)
{
const std::uint16_t s = static_cast<std::uint16_t>(
(static_cast<unsigned>(color) + m - tau) % m);
const lab label = shift_lab(zin, din, s);
const lab active = shift_lab(st.zero[i], dout, nd.phi[s]);
const lab h = hash_lab(i, color, label, nmod);
st.proj[i].row[static_cast<std::size_t>(color - 1)] = add_lab(active, h);
}
break;
}
case circuit::op::pass:
{
const std::uint16_t p = nd.mod;
const lab & zw = st.zero[nd.a];
const lab & zb = st.zero[nd.b];
(void)st.delta_of(p);
(void)st.delta_of(2);
st.zero[i] = sample_lab(p, st.seed, st.n);
const std::uint16_t tau_b = zb.d[0];
const lab addend = sub_lab(st.zero[i], zw);
for (std::uint16_t color = 0; color < 2; ++color)
{
const std::uint16_t sem = static_cast<std::uint16_t>(
(color + 2u - (tau_b % 2u)) % 2u);
const lab bit_label = shift_lab(zb, st.delta_of(2), sem);
const lab h = hash_lab(i, color, bit_label, p);
const lab payload = (sem == 0) ? st.zero[i] : addend;
st.pass[i].payload[color] = add_lab(payload, h);
const std::uint16_t pad = hash_lab(i, 8u + color, bit_label, 2).d[0];
st.pass[i].flag_ct[color] =
static_cast<std::uint16_t>(sem ^ (pad & 1u));
}
break;
}
}
}
return st;
}
inline std::vector<lab> evaluate(const circuit & c, const garble_state & st,
const std::uint16_t * semantic)
{
const auto & nodes = c.nodes();
std::vector<lab> active(nodes.size());
std::uint32_t in_i = 0;
for (std::uint32_t i = 0; i < nodes.size(); ++i)
{
const auto & nd = nodes[i];
switch (nd.code)
{
case circuit::op::in:
{
if (semantic == nullptr)
throw std::invalid_argument("arith_garble: inputs");
if (semantic[in_i] >= nd.mod)
throw std::invalid_argument("arith_garble: input range");
active[i] = shift_lab(st.zero[i], st.delta[nd.mod], semantic[in_i]);
++in_i;
break;
}
case circuit::op::add:
active[i] = add_lab(active[nd.a], active[nd.b]);
break;
case circuit::op::addk:
active[i] = active[nd.a];
break;
case circuit::op::scale:
active[i] = scale_lab(active[nd.a], nd.k);
break;
case circuit::op::proj:
{
const std::uint16_t color = active[nd.a].d[0];
const lab h = hash_lab(i, color, active[nd.a], nd.mod);
if (color == 0)
active[i] = neg_lab(h);
else
active[i] = sub_lab(
st.proj[i].row[static_cast<std::size_t>(color - 1)], h);
break;
}
case circuit::op::pass:
{
const std::uint16_t color = active[nd.b].d[0];
const lab h = hash_lab(i, color, active[nd.b], nd.mod);
const lab payload = sub_lab(st.pass[i].payload[color], h);
const std::uint16_t pad =
hash_lab(i, 8u + color, active[nd.b], 2).d[0];
const std::uint16_t flag = static_cast<std::uint16_t>(
st.pass[i].flag_ct[color] ^ (pad & 1u));
if (flag == 0)
active[i] = payload;
else
active[i] = add_lab(active[nd.a], payload);
break;
}
}
}
return active;
}
} // namespace detail
/// @brief Clear semantics, one value per wire, inputs in wire order.
HEDLEY_WARN_UNUSED_RESULT
inline std::vector<std::uint16_t> eval_plain(const circuit & c,
const std::uint16_t * semantic)
{
const auto & nodes = c.nodes();
std::vector<std::uint16_t> s(nodes.size(), 0);
std::uint32_t in_i = 0;
for (std::uint32_t i = 0; i < nodes.size(); ++i)
{
const auto & nd = nodes[i];
switch (nd.code)
{
case circuit::op::in:
if (semantic == nullptr || semantic[in_i] >= nd.mod)
throw std::invalid_argument("arith_garble: input");
s[i] = semantic[in_i++];
break;
case circuit::op::add:
s[i] = static_cast<std::uint16_t>(
(static_cast<unsigned>(s[nd.a]) + s[nd.b]) % nd.mod);
break;
case circuit::op::addk:
s[i] = static_cast<std::uint16_t>(
(static_cast<unsigned>(s[nd.a]) + nd.k) % nd.mod);
break;
case circuit::op::scale:
s[i] = static_cast<std::uint16_t>(
(static_cast<unsigned>(s[nd.a]) * nd.k) % nd.mod);
break;
case circuit::op::proj:
s[i] = nd.phi[s[nd.a]];
break;
case circuit::op::pass:
s[i] = (s[nd.b] != 0) ? s[nd.a] : static_cast<std::uint16_t>(0);
break;
}
}
return s;
}
/// @brief Garble and evaluate in one process.
/// @details `semantic` is one value per `input` call, in that order.
HEDLEY_WARN_UNUSED_RESULT
inline shares eval_pair(const circuit & c, const std::uint16_t * semantic)
{
if (c.outputs().empty())
throw std::invalid_argument("arith_garble: no outputs");
auto st = detail::garble(c);
auto active = detail::evaluate(c, st, semantic);
shares out;
std::size_t rows = 0;
for (std::uint32_t i = 0; i < c.nodes().size(); ++i)
{
if (c.nodes()[i].code == circuit::op::proj)
rows += st.proj[i].row.size();
else if (c.nodes()[i].code == circuit::op::pass)
rows += 2;
}
out.ciphertext_rows = rows;
for (std::uint32_t id : c.outputs())
{
const std::uint16_t mod = c.nodes()[id].mod;
const std::uint16_t mask = st.zero[id].d[0];
const std::uint16_t color = active[id].d[0];
out.modulus.push_back(mod);
out.mask.push_back(mask);
out.color.push_back(color);
out.opened.push_back(open_shares(mod, mask, color));
}
return out;
}
} // namespace arith_garble
} // namespace dpf
#endif // LIBDPF_INCLUDE_DPF_ARITH_GARBLE_HPP__

View file

@ -68,6 +68,10 @@ auto async_post(ExecutorT executor, Function && func, CompletionToken && token)
// make_dpf
//
/// \complexity Local `make_dpf` is O(n) per key, then one write of each key. n is `depth`.
/// \rounds 1 per key. Each party is a single `asio::write` of six buffers; there is no reply.
/// \communication Per key, two copies (one per peer) of the correction-word array (n nodes), the advice array (n bytes), one root, the leaf tuple, the beaver tuple, and the offset word.
/// \preprocessing none beyond the local keygen. The dealer holds the clear point.
template <typename InteriorPRG = dpf::prg::aes128,
typename ExteriorPRG = InteriorPRG,
typename PeerT,
@ -123,6 +127,10 @@ auto make_dpf(PeerT & peer0, PeerT & peer1, std::size_t count, dpfargs<InputT, O
return std::make_tuple(bytes_written0, bytes_written1, count);
}
/// \complexity Local `make_dpf` is O(n) per key, then one write of each key. n is `depth`.
/// \rounds 1 per key. Each party is a single `asio::write` of six buffers; there is no reply.
/// \communication Per key, two copies (one per peer) of the correction-word array (n nodes), the advice array (n bytes), one root, the leaf tuple, the beaver tuple, and the offset word.
/// \preprocessing none beyond the local keygen. The dealer holds the clear point.
template <typename InteriorPRG = dpf::prg::aes128,
typename ExteriorPRG = InteriorPRG,
typename PeerT,
@ -138,6 +146,10 @@ auto make_dpf(PeerT & peer0, PeerT & peer1, std::size_t count, dpfargs<InputT, O
return ret;
}
/// \complexity Local `make_dpf` is O(n) per key, then one write of each key. n is `depth`.
/// \rounds 1 per key. Each party is a single `asio::write` of six buffers; there is no reply.
/// \communication Per key, two copies (one per peer) of the correction-word array (n nodes), the advice array (n bytes), one root, the leaf tuple, the beaver tuple, and the offset word.
/// \preprocessing none beyond the local keygen. The dealer holds the clear point.
template <typename InteriorPRG = dpf::prg::aes128,
typename ExteriorPRG = InteriorPRG,
typename PeerT,
@ -152,6 +164,10 @@ auto make_dpf(PeerT & peer0, PeerT & peer1, dpfargs<InputT, OutputT, OutputTs...
return std::make_tuple(bytes_written0, bytes_written1);
}
/// \complexity Local `make_dpf` is O(n) per key, then one write of each key. n is `depth`.
/// \rounds 1 per key. Each party is a single `asio::write` of six buffers; there is no reply.
/// \communication Per key, two copies (one per peer) of the correction-word array (n nodes), the advice array (n bytes), one root, the leaf tuple, the beaver tuple, and the offset word.
/// \preprocessing none beyond the local keygen. The dealer holds the clear point.
template <typename InteriorPRG = dpf::prg::aes128,
typename ExteriorPRG = InteriorPRG,
typename PeerT,
@ -646,6 +662,10 @@ auto async_read_dpf(DealerT & dealer, Emplaceable & output, CompletionToken && t
// assign_wildcard_input
//
/// \complexity O(1) arithmetic besides the socket transfer.
/// \rounds 1. One write of the local share, one read of the peer share (`async_assign_wildcard_input`).
/// \communication `sizeof(input_type)` bytes each way.
/// \preprocessing The mask in `offset_x` was sampled at `make_dpf`. This exchange opens mask − alpha.
template <typename PeerT,
typename DpfKey,
typename InputType>
@ -675,6 +695,10 @@ auto assign_wildcard_input(PeerT & peer_in, PeerT & peer_out, DpfKey & dpf,
return std::make_tuple(offset_share, bytes_written, bytes_read);
}
/// \complexity O(1) arithmetic besides the socket transfer.
/// \rounds 1. One write of the local share, one read of the peer share (`async_assign_wildcard_input`).
/// \communication `sizeof(input_type)` bytes each way.
/// \preprocessing The mask in `offset_x` was sampled at `make_dpf`. This exchange opens mask − alpha.
template <typename PeerT,
typename DpfKey,
typename InputType>
@ -686,6 +710,10 @@ auto assign_wildcard_input(PeerT & peer, DpfKey & dpf, InputType && input_share,
std::forward<InputType>(input_share), error);
}
/// \complexity O(1) arithmetic besides the socket transfer.
/// \rounds 1. One write of the local share, one read of the peer share (`async_assign_wildcard_input`).
/// \communication `sizeof(input_type)` bytes each way.
/// \preprocessing The mask in `offset_x` was sampled at `make_dpf`. This exchange opens mask − alpha.
template <typename PeerT,
typename DpfKey,
typename InputType>
@ -700,6 +728,10 @@ auto assign_wildcard_input(PeerT & peer_in, PeerT & peer_out, DpfKey & dpf,
return ret;
}
/// \complexity O(1) arithmetic besides the socket transfer.
/// \rounds 1. One write of the local share, one read of the peer share (`async_assign_wildcard_input`).
/// \communication `sizeof(input_type)` bytes each way.
/// \preprocessing The mask in `offset_x` was sampled at `make_dpf`. This exchange opens mask − alpha.
template <typename PeerT,
typename DpfKey,
typename InputType>
@ -717,6 +749,10 @@ auto assign_wildcard_input(PeerT & peer, DpfKey & dpf, InputType && input_share)
// async_assign_wildcard_input
//
/// \complexity O(1) arithmetic besides the socket transfer.
/// \rounds 1. One write of the local share, one read of the peer share (`async_assign_wildcard_input`).
/// \communication `sizeof(input_type)` bytes each way.
/// \preprocessing The mask in `offset_x` was sampled at `make_dpf`. This exchange opens mask − alpha.
template <typename PeerT,
typename ExecutorT,
typename DpfKey,
@ -796,6 +832,10 @@ auto async_assign_wildcard_input(PeerT & peer_in, PeerT & peer_out,
#include <asio/unyield.hpp>
}
/// \complexity O(1) arithmetic besides the socket transfer.
/// \rounds 1. One write of the local share, one read of the peer share (`async_assign_wildcard_input`).
/// \communication `sizeof(input_type)` bytes each way.
/// \preprocessing The mask in `offset_x` was sampled at `make_dpf`. This exchange opens mask − alpha.
template <typename PeerT,
typename ExecutorT,
typename DpfKey,
@ -811,6 +851,10 @@ auto async_assign_wildcard_input(PeerT & peer, ExecutorT work_executor,
std::forward<CompletionToken>(token));
}
/// \complexity O(1) arithmetic besides the socket transfer.
/// \rounds 1. One write of the local share, one read of the peer share (`async_assign_wildcard_input`).
/// \communication `sizeof(input_type)` bytes each way.
/// \preprocessing The mask in `offset_x` was sampled at `make_dpf`. This exchange opens mask − alpha.
template <typename PeerT,
typename DpfKey,
typename InputType,
@ -826,6 +870,10 @@ auto async_assign_wildcard_input(PeerT & peer_in, PeerT & peer_out,
std::forward<CompletionToken>(token));
}
/// \complexity O(1) arithmetic besides the socket transfer.
/// \rounds 1. One write of the local share, one read of the peer share (`async_assign_wildcard_input`).
/// \communication `sizeof(input_type)` bytes each way.
/// \preprocessing The mask in `offset_x` was sampled at `make_dpf`. This exchange opens mask − alpha.
template <typename PeerT,
typename DpfKey,
typename InputType,
@ -844,6 +892,10 @@ auto async_assign_wildcard_input(PeerT & peer, DpfKey & dpf, InputType && input_
// assign_wildcard_output
//
/// \complexity O(leaf bytes) for the Beaver leaf arithmetic, plus the transfers.
/// \rounds 2. Write/read the blinded output share, then write/read the leaf share (`async_assign_wildcard_output`).
/// \communication `sizeof(output_type)` plus `sizeof(leaf_type)` each way.
/// \preprocessing The leaf Beaver triple (`vector_blind`, `output_blind`, `blinded_vector`) was stored at keygen.
template <std::size_t I = 0,
typename PeerT,
typename DpfKey,
@ -861,7 +913,11 @@ auto assign_wildcard_output(PeerT & peer_in, PeerT & peer_out, DpfKey & dpf,
constexpr bool is_packed = true;
auto & leaf_wrapper = utils::get<I>(dpf.leaf_nodes);
// Second (and later) assigns install β'−β on top of the ready leaf.
if (leaf_wrapper.is_ready())
leaf_wrapper.begin_update();
auto blinded_output = leaf_wrapper.compute_and_get_blinded_output_share(output_share);
bytes_written += ::asio::write(peer_out, ::asio::buffer(&blinded_output, sizeof(output_type)), error);
if (error)
@ -898,6 +954,10 @@ auto assign_wildcard_output(PeerT & peer_in, PeerT & peer_out, DpfKey & dpf,
return std::make_tuple(leaf_share, bytes_written, bytes_read);
}
/// \complexity O(leaf bytes) for the Beaver leaf arithmetic, plus the transfers.
/// \rounds 2. Write/read the blinded output share, then write/read the leaf share (`async_assign_wildcard_output`).
/// \communication `sizeof(output_type)` plus `sizeof(leaf_type)` each way.
/// \preprocessing The leaf Beaver triple (`vector_blind`, `output_blind`, `blinded_vector`) was stored at keygen.
template <std::size_t I = 0,
typename PeerT,
typename DpfKey,
@ -910,6 +970,10 @@ auto assign_wildcard_output(PeerT & peer, DpfKey & dpf,
std::forward<OutputType>(output_share), error);
}
/// \complexity O(leaf bytes) for the Beaver leaf arithmetic, plus the transfers.
/// \rounds 2. Write/read the blinded output share, then write/read the leaf share (`async_assign_wildcard_output`).
/// \communication `sizeof(output_type)` plus `sizeof(leaf_type)` each way.
/// \preprocessing The leaf Beaver triple (`vector_blind`, `output_blind`, `blinded_vector`) was stored at keygen.
template <std::size_t I = 0,
typename PeerT,
typename DpfKey,
@ -925,6 +989,10 @@ auto assign_wildcard_output(PeerT & peer_in, PeerT & peer_out, DpfKey & dpf,
return ret;
}
/// \complexity O(leaf bytes) for the Beaver leaf arithmetic, plus the transfers.
/// \rounds 2. Write/read the blinded output share, then write/read the leaf share (`async_assign_wildcard_output`).
/// \communication `sizeof(output_type)` plus `sizeof(leaf_type)` each way.
/// \preprocessing The leaf Beaver triple (`vector_blind`, `output_blind`, `blinded_vector`) was stored at keygen.
template <std::size_t I = 0,
typename PeerT,
typename DpfKey,
@ -943,6 +1011,10 @@ auto assign_wildcard_output(PeerT & peer, DpfKey & dpf, OutputType && output_sha
// async_assign_wildcard_output
//
/// \complexity O(leaf bytes) for the Beaver leaf arithmetic, plus the transfers.
/// \rounds 2. Write/read the blinded output share, then write/read the leaf share (`async_assign_wildcard_output`).
/// \communication `sizeof(output_type)` plus `sizeof(leaf_type)` each way.
/// \preprocessing The leaf Beaver triple (`vector_blind`, `output_blind`, `blinded_vector`) was stored at keygen.
template <std::size_t I = 0,
typename PeerT,
typename ExecutorT,
@ -987,6 +1059,8 @@ auto async_assign_wildcard_output(PeerT & peer_in, PeerT & peer_out,
{
yield async_post(work_executor, [&leaf, output_share]() mutable
{
if (leaf.is_ready())
leaf.begin_update();
*output_share = leaf.compute_and_get_blinded_output_share(*output_share);
}, std::move(self));
@ -1059,6 +1133,10 @@ auto async_assign_wildcard_output(PeerT & peer_in, PeerT & peer_out,
#include <asio/unyield.hpp>
}
/// \complexity O(leaf bytes) for the Beaver leaf arithmetic, plus the transfers.
/// \rounds 2. Write/read the blinded output share, then write/read the leaf share (`async_assign_wildcard_output`).
/// \communication `sizeof(output_type)` plus `sizeof(leaf_type)` each way.
/// \preprocessing The leaf Beaver triple (`vector_blind`, `output_blind`, `blinded_vector`) was stored at keygen.
template <std::size_t I = 0,
typename PeerT,
typename ExecutorT,
@ -1075,6 +1153,10 @@ auto async_assign_wildcard_output(PeerT & peer, ExecutorT work_executor,
std::forward<CompletionToken>(token));
}
/// \complexity O(leaf bytes) for the Beaver leaf arithmetic, plus the transfers.
/// \rounds 2. Write/read the blinded output share, then write/read the leaf share (`async_assign_wildcard_output`).
/// \communication `sizeof(output_type)` plus `sizeof(leaf_type)` each way.
/// \preprocessing The leaf Beaver triple (`vector_blind`, `output_blind`, `blinded_vector`) was stored at keygen.
template <std::size_t I = 0,
typename PeerT,
typename DpfKey,
@ -1091,6 +1173,10 @@ auto async_assign_wildcard_output(PeerT & peer_in, PeerT & peer_out,
std::forward<CompletionToken>(token));
}
/// \complexity O(leaf bytes) for the Beaver leaf arithmetic, plus the transfers.
/// \rounds 2. Write/read the blinded output share, then write/read the leaf share (`async_assign_wildcard_output`).
/// \communication `sizeof(output_type)` plus `sizeof(leaf_type)` each way.
/// \preprocessing The leaf Beaver triple (`vector_blind`, `output_blind`, `blinded_vector`) was stored at keygen.
template <std::size_t I = 0,
typename PeerT,
typename DpfKey,
@ -1101,7 +1187,6 @@ HEDLEY_ALWAYS_INLINE
auto async_assign_wildcard_output(PeerT & peer, DpfKey & dpf,
OutputType && output_share, CompletionToken && token)
{
auto work_executor = ::asio::system_executor();
return async_assign_wildcard_output<I>(peer, peer, dpf,
std::forward<OutputType>(output_share),
std::forward<CompletionToken>(token));

View file

@ -0,0 +1,234 @@
/// @file dpf/async_protocol.hpp
/// @brief Real overlapped, event-driven byte-round protocol runner.
/// @details `overlapped_byte_protocol` replaces the blocking loop of
/// `factory::async_byte_protocol` with true asynchronous I/O over an
/// `async_stream_array`. Each round: read the blind from the dealer
/// (async, skipped when `blind_bytes == 0`), `produce` the outbound
/// message, then overlap the peer write and peer read as one
/// `async_exchange`, and on completion `finish`, fire the round
/// callback, and continue to the next round or the done handler.
/// Nothing spins on a `peer_ready` flag and nothing blocks the calling
/// thread — every continuation runs on an `io_context` thread.
#ifndef LIBDPF_INCLUDE_DPF_ASYNC_PROTOCOL_HPP__
#define LIBDPF_INCLUDE_DPF_ASYNC_PROTOCOL_HPP__
#include <cstddef>
#include <cstdint>
#include <functional>
#include <memory>
#include <mutex>
#include <stdexcept>
#include <system_error>
#include <utility>
#include <vector>
#include "dpf/net/asio_ns.hpp"
#include "dpf/net/async_stream_array.hpp"
#include "dpf/protocol_factory.hpp" // factory::async_byte_round
namespace dpf
{
namespace async
{
/// @brief Overlap a write and a read on stream `i`; fire `done` once both end.
/// @details The two operations are issued concurrently (full duplex): the
/// handler runs after both complete, carrying the first error seen.
/// `out` and `in` must stay valid until `done` fires.
inline void async_exchange(net::async_stream_array & s, std::size_t i,
const void * out, std::size_t out_n, void * in, std::size_t in_n,
net::async_handler done)
{
struct state
{
net::async_handler done;
int remaining = 2;
std::error_code ec;
std::mutex mu;
};
auto st = std::make_shared<state>();
st->done = std::move(done);
auto complete = [st](const std::error_code & ec) {
bool fire = false;
std::error_code final_ec;
{
std::lock_guard<std::mutex> lock(st->mu);
if (ec && !st->ec)
st->ec = ec;
if (--st->remaining == 0)
fire = true;
final_ec = st->ec;
}
if (fire && st->done)
st->done(final_ec);
};
s.async_write(i, out, out_n, complete);
s.async_read(i, in, in_n, complete);
}
/// @brief Fully overlapped runner for a list of `async_byte_round`s.
/// @details Constructed per party per run and driven through a `shared_ptr`
/// (kept alive by its own continuations). `start` kicks round 0.
class overlapped_byte_protocol
: public std::enable_shared_from_this<overlapped_byte_protocol>
{
public:
/// @param ec error (empty on success); @param state final protocol state.
using done_handler =
std::function<void(const std::error_code & ec,
std::vector<std::uint8_t> state)>;
using on_round_complete_fn = std::function<void(std::size_t round)>;
/// @param peer overlapped peer array; lane = `r % peer.size()` so a small
/// pool can carry many sequential rounds (known-size exchange).
/// @param dealer optional blind source (lane `r % dealer.size()`); may be
/// null when every round has `blind_bytes == 0`.
overlapped_byte_protocol(net::async_stream_array & peer,
net::async_stream_array * dealer,
std::vector<factory::async_byte_round> rounds,
on_round_complete_fn on_round_complete = {})
: peer_(&peer),
dealer_(dealer),
rounds_(std::move(rounds)),
on_round_complete_(std::move(on_round_complete))
{
if (peer_->size() == 0)
throw std::invalid_argument("overlapped_byte_protocol: empty peer");
for (const auto & r : rounds_)
{
if (!r.produce || !r.finish)
throw std::invalid_argument("overlapped_byte_protocol: missing callbacks");
if (r.blind_bytes != 0 && dealer_ == nullptr)
throw std::invalid_argument("overlapped_byte_protocol: blind needs a dealer");
}
if (dealer_ != nullptr && dealer_->size() == 0)
throw std::invalid_argument("overlapped_byte_protocol: empty dealer");
}
std::size_t rounds() const noexcept { return rounds_.size(); }
/// @brief Begin the protocol. `done` fires once, after the last round.
void start(std::size_t index, std::vector<std::uint8_t> state,
done_handler done)
{
(void)index;
state_ = std::move(state);
done_ = std::move(done);
run_round(0);
}
private:
void finish_with(const std::error_code & ec)
{
if (done_)
{
auto d = std::move(done_);
done_ = nullptr;
d(ec, std::move(state_));
}
}
void run_round(std::size_t r)
{
if (r >= rounds_.size())
{
finish_with(std::error_code{});
return;
}
const auto & spec = rounds_[r];
blind_.assign(spec.blind_bytes, 0);
auto self = shared_from_this();
if (spec.blind_bytes != 0)
{
const std::size_t dlane = r % dealer_->size();
dealer_->async_read(dlane, blind_.data(), blind_.size(),
[self, r](const std::error_code & ec) {
self->after_blind(r, ec);
});
}
else
{
// No blind: hop through the io_context so we never recurse deeply.
asio::post(peer_->context(),
[self, r]() { self->after_blind(r, std::error_code{}); });
}
}
void after_blind(std::size_t r, const std::error_code & ec)
{
if (ec)
{
finish_with(ec);
return;
}
const auto & spec = rounds_[r];
outbound_ = spec.produce(state_, blind_.data(), blind_.size());
if (outbound_.size() != spec.msg_bytes)
{
finish_with(std::make_error_code(std::errc::message_size));
return;
}
inbound_.assign(spec.msg_bytes, 0);
auto self = shared_from_this();
const std::size_t lane = r % peer_->size();
async_exchange(*peer_, lane, outbound_.data(), outbound_.size(),
inbound_.data(), inbound_.size(),
[self, r](const std::error_code & xec) {
self->after_exchange(r, xec);
});
}
void after_exchange(std::size_t r, const std::error_code & ec)
{
if (ec)
{
finish_with(ec);
return;
}
const auto & spec = rounds_[r];
spec.finish(state_, inbound_.data(), inbound_.size(), blind_.data(),
blind_.size());
if (on_round_complete_)
on_round_complete_(r);
run_round(r + 1);
}
net::async_stream_array * peer_ = nullptr;
net::async_stream_array * dealer_ = nullptr;
std::vector<factory::async_byte_round> rounds_;
on_round_complete_fn on_round_complete_;
std::vector<std::uint8_t> state_;
done_handler done_;
std::vector<std::uint8_t> blind_;
std::vector<std::uint8_t> outbound_;
std::vector<std::uint8_t> inbound_;
};
/// @brief Make an `overlapped_byte_protocol` as a `shared_ptr` (required for
/// the self-owning continuation chain).
HEDLEY_WARN_UNUSED_RESULT
inline std::shared_ptr<overlapped_byte_protocol> make_overlapped_byte_protocol(
net::async_stream_array & peer, net::async_stream_array * dealer,
std::vector<factory::async_byte_round> rounds,
overlapped_byte_protocol::on_round_complete_fn on_round_complete = {})
{
return std::make_shared<overlapped_byte_protocol>(peer, dealer,
std::move(rounds), std::move(on_round_complete));
}
/// @brief Post `start_all`, then run the io_context until all work drains.
/// @details The single entry point for driving overlapped parties: everything
/// the parties initiate is chased to completion by `io.run()`.
template <typename Fn>
void run_overlapped(asio::io_context & io, Fn start_all)
{
asio::post(io, [start_all = std::move(start_all)]() mutable { start_all(); });
io.run();
}
} // namespace async
} // namespace dpf
#endif // LIBDPF_INCLUDE_DPF_ASYNC_PROTOCOL_HPP__

File diff suppressed because it is too large Load diff

578
include/dpf/bench_cells.hpp Normal file
View file

@ -0,0 +1,578 @@
/// @file dpf/bench_cells.hpp
/// @brief Bench cells for work that is not a DPF walk.
/// @details A secret index stays a key. These cells time what happens after
/// the parties already hold shares: a constant-round word gadget, a
/// stacked branch of a leaf netlist, a short public table, and a
/// hidden reorder of an RSS column. Each cell is a compose plan. The
/// harness drives it with `run_parties`, so the payload crosses the
/// same transport as a DPF plan (`DPF_TRANSPORT`).
///
/// Two-party cells put the garble or the table on party 0 and the
/// second evaluation on party 1, with one peer payload between them.
/// The shuffle is three ring passes. The left-out party's slot is
/// empty; the other two carry that pass's array.
#ifndef LIBDPF_INCLUDE_DPF_BENCH_CELLS_HPP__
#define LIBDPF_INCLUDE_DPF_BENCH_CELLS_HPP__
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <numeric>
#include <stdexcept>
#include <string>
#include <utility>
#include <vector>
#include "dpf/arith_garble.hpp"
#include "dpf/compose.hpp"
#include "dpf/flute.hpp"
#include "dpf/shuffle.hpp"
#include "dpf/yao.hpp"
#include "dpf/yao_stack.hpp"
namespace dpf
{
namespace bench
{
/// @brief One measured configuration. The low 16 bits ride in the node aux.
enum class kind : std::uint32_t
{
proj5 = 1,
proj17,
proj64,
mul5,
mul7,
mul11,
thresh8,
thresh16,
chain7,
yao_if4,
yao_if16,
yao_hot4,
yao_hot8,
flute2,
flute4,
flute8,
flute4x8,
shuf16,
shuf64,
shuf256
};
inline constexpr std::uint32_t aux_pass(kind k, unsigned pass) noexcept
{
return static_cast<std::uint32_t>(k) | (static_cast<std::uint32_t>(pass) << 16);
}
inline kind kind_of(std::uint32_t aux) noexcept
{
return static_cast<kind>(aux & 0xffffu);
}
inline unsigned pass_of(std::uint32_t aux) noexcept
{
return aux >> 16;
}
inline bool is_shuffle(kind k) noexcept
{
return k == kind::shuf16 || k == kind::shuf64 || k == kind::shuf256;
}
inline std::size_t shuffle_n(kind k)
{
if (k == kind::shuf16)
return 16;
if (k == kind::shuf64)
return 64;
if (k == kind::shuf256)
return 256;
throw std::invalid_argument("bench: shuffle cell");
}
namespace detail
{
inline dpf::yao::netlist and_n(unsigned n)
{
dpf::yao::netlist nl;
std::vector<dpf::yao::bit> in;
for (unsigned i = 0; i < n; ++i)
in.push_back(nl.shared_in());
auto acc = in[0];
for (unsigned i = 1; i < n; ++i)
acc = nl.and_(acc, in[i]);
nl.out(acc);
return nl;
}
inline dpf::yao::netlist xor2()
{
dpf::yao::netlist nl;
auto a = nl.shared_in();
auto b = nl.shared_in();
nl.out(nl.xor_(a, b));
return nl;
}
inline std::uint64_t mix_bytes(const std::uint8_t * p, std::size_t n)
{
std::uint64_t h = 14695981039346656037ull;
for (std::size_t i = 0; i < n; ++i)
{
h ^= p[i];
h *= 1099511628211ull;
}
return h;
}
inline void paint(std::uint8_t * out, std::size_t n, std::uint64_t mix)
{
for (std::size_t i = 0; i < n; i += 8)
{
const std::uint64_t w = mix + static_cast<std::uint64_t>(i);
const std::size_t k = std::min<std::size_t>(8, n - i);
std::memcpy(out + i, &w, k);
}
}
inline std::size_t proj_bytes(std::uint16_t mod)
{
arith_garble::circuit c;
auto x = c.input(mod);
std::vector<std::uint16_t> id(mod);
for (std::uint16_t i = 0; i < mod; ++i)
id[i] = i;
c.out(c.project(x, mod, std::move(id)));
const std::uint16_t in = 1;
return std::max<std::size_t>(
32, dpf::arith_garble::eval_pair(c, &in).ciphertext_rows * 32u);
}
inline std::size_t mul_bytes(std::uint16_t p)
{
arith_garble::circuit c;
auto x = c.input(p);
auto y = c.input(p);
c.out(c.mul(x, y));
const std::uint16_t in[2] = {1, 1};
return std::max<std::size_t>(
32, dpf::arith_garble::eval_pair(c, in).ciphertext_rows * 32u);
}
inline std::size_t thresh_bytes(unsigned b)
{
arith_garble::circuit c;
std::vector<arith_garble::wire> bits;
const auto mod = static_cast<std::uint16_t>(b + 1);
for (unsigned i = 0; i < b; ++i)
bits.push_back(c.input(mod));
c.out(c.threshold(bits, static_cast<std::uint16_t>(b)));
std::vector<std::uint16_t> in(b, 1);
return std::max<std::size_t>(
32, dpf::arith_garble::eval_pair(c, in.data()).ciphertext_rows * 32u);
}
inline std::size_t chain_bytes()
{
arith_garble::circuit c;
auto x = c.input(7);
auto y = c.input(7);
auto z = c.mul(x, y);
for (int i = 0; i < 3; ++i)
z = c.mul(z, x);
c.out(z);
const std::uint16_t in[2] = {2, 3};
return std::max<std::size_t>(
32, dpf::arith_garble::eval_pair(c, in).ciphertext_rows * 32u);
}
inline std::uint64_t run_proj(std::uint16_t mod)
{
arith_garble::circuit c;
auto x = c.input(mod);
std::vector<std::uint16_t> id(mod);
for (std::uint16_t i = 0; i < mod; ++i)
id[i] = static_cast<std::uint16_t>((i * 3) % mod);
c.out(c.project(x, mod, std::move(id)));
const std::uint16_t in = static_cast<std::uint16_t>(mod / 2);
auto got = dpf::arith_garble::eval_pair(c, &in);
return got.opened.empty() ? 0 : got.opened[0];
}
inline std::uint64_t run_mul(std::uint16_t p)
{
arith_garble::circuit c;
auto x = c.input(p);
auto y = c.input(p);
c.out(c.mul(x, y));
const std::uint16_t in[2] = {static_cast<std::uint16_t>(p - 1), 2};
auto got = dpf::arith_garble::eval_pair(c, in);
return got.opened.empty() ? 0 : got.opened[0];
}
inline std::uint64_t run_thresh(unsigned b)
{
arith_garble::circuit c;
std::vector<arith_garble::wire> bits;
const auto mod = static_cast<std::uint16_t>(b + 1);
for (unsigned i = 0; i < b; ++i)
bits.push_back(c.input(mod));
c.out(c.threshold(bits, static_cast<std::uint16_t>(b / 2)));
std::vector<std::uint16_t> in(b, 0);
for (unsigned i = 0; i < b; i += 2)
in[i] = 1;
auto got = dpf::arith_garble::eval_pair(c, in.data());
return got.opened.empty() ? 0 : got.opened[0];
}
inline std::uint64_t run_chain()
{
arith_garble::circuit c;
auto x = c.input(7);
auto y = c.input(7);
auto z = c.mul(x, y);
for (int i = 0; i < 3; ++i)
z = c.mul(z, x);
c.out(z);
const std::uint16_t in[2] = {2, 3};
auto got = dpf::arith_garble::eval_pair(c, in);
return got.opened.empty() ? 0 : got.opened[0];
}
inline std::size_t yao_if_bytes(unsigned heavy, unsigned light)
{
std::vector<std::uint8_t> h(heavy, 1), l(light, 1);
auto got = dpf::yao::eval_if(and_n(heavy), and_n(light), 0, 0, h.data(),
h.data(), l.data(), l.data());
return std::max<std::size_t>(
16, (got.stack_blocks + got.extra_blocks) * 16u);
}
inline std::uint64_t run_if(unsigned heavy, unsigned light)
{
std::vector<std::uint8_t> h0(heavy, 1), h1(heavy, 0), l0(light, 1), l1(light, 1);
auto got = dpf::yao::eval_if(and_n(heavy), and_n(light), 1, 0, h0.data(),
h1.data(), l0.data(), l1.data());
return got.share0.empty()
? 0
: static_cast<std::uint64_t>(got.share0[0] ^ got.share1[0]);
}
inline std::size_t yao_hot_bytes(unsigned k)
{
std::vector<dpf::yao::netlist> br;
std::vector<std::vector<std::uint8_t>> p0, p1;
for (unsigned i = 0; i < k; ++i)
{
br.push_back(i % 2 == 0 ? and_n(2) : xor2());
p0.push_back({1, 1});
p1.push_back({0, 0});
}
auto got = dpf::yao::eval_one_hot(br, 1, 0, p0, p1);
return std::max<std::size_t>(
16, (got.stack_blocks + got.extra_blocks) * 16u);
}
inline std::uint64_t run_hot(unsigned k)
{
std::vector<dpf::yao::netlist> br;
std::vector<std::vector<std::uint8_t>> p0, p1;
for (unsigned i = 0; i < k; ++i)
{
br.push_back(i % 2 == 0 ? and_n(2) : xor2());
p0.push_back({1, static_cast<std::uint8_t>(i & 1u)});
p1.push_back({0, 1});
}
auto got = dpf::yao::eval_one_hot(br, 1, 2, p0, p1);
return got.share0.empty()
? 0
: static_cast<std::uint64_t>(got.share0[0] ^ got.share1[0]);
}
inline std::uint64_t run_flute(unsigned delta, unsigned n_out)
{
const unsigned rows = 1u << delta;
std::vector<std::uint8_t> columns(static_cast<std::size_t>(n_out) * rows);
for (unsigned w = 0; w < n_out; ++w)
for (unsigned j = 0; j < rows; ++j)
columns[static_cast<std::size_t>(w) * rows + j] =
static_cast<std::uint8_t>(((j >> (w % delta)) ^ w) & 1u);
std::vector<std::uint8_t> bits(delta, 0);
bits[0] = 1;
if (delta > 2)
bits[2] = 1;
auto got = dpf::flute::eval_pair(delta, n_out, columns.data(), bits.data());
std::uint64_t mix = 0;
for (auto b : got.opened)
mix = (mix << 1) | b;
return mix;
}
inline rss::seed_bundle fixed_bundle()
{
rss::seed_bundle b{};
auto fill = [](rss::seed_block & s, std::uint8_t tag) {
auto * p = reinterpret_cast<std::uint8_t *>(&s);
for (std::size_t i = 0; i < sizeof(s); ++i)
p[i] = static_cast<std::uint8_t>(tag + i * 17u);
};
fill(b.k01, 1);
fill(b.k12, 2);
fill(b.k20, 3);
return b;
}
inline void shuffle_outbound(unsigned me, unsigned pass, std::size_t n,
std::uint8_t * out, std::size_t out_n)
{
if (out_n != n * sizeof(std::uint64_t))
throw std::logic_error("bench shuffle slot");
const unsigned order[3] = {2u, 0u, 1u};
const unsigned left = order[pass];
if (me == left)
{
std::memset(out, 0, out_n);
return;
}
const auto bundle = fixed_bundle();
std::vector<std::uint64_t> column(n);
std::iota(column.begin(), column.end(), 0);
shuffle::shuffle_party_view<std::uint64_t> held[3];
for (unsigned p = 0; p < 3; ++p)
{
held[p].own.assign(n, 0);
held[p].next.assign(n, 0);
}
std::uint64_t rng = 0xA5A5A5A5A5A5A5A5ull;
auto draw = [&] {
rng = rng * 6364136223846793005ull + 1u;
return rng;
};
for (std::size_t i = 0; i < n; ++i)
{
const auto a = draw();
const auto b = draw();
const auto c = column[i] - a - b;
held[0].own[i] = a;
held[0].next[i] = b;
held[1].own[i] = b;
held[1].next[i] = c;
held[2].own[i] = c;
held[2].next[i] = a;
}
for (unsigned step = 0; step <= pass; ++step)
{
const unsigned L = order[step];
const unsigned u = shuffle::hidden_u_party(L);
const unsigned side = shuffle::hidden_side_party(L);
auto u_step = shuffle::shuffle_hidden_pass<std::uint64_t>(
u, rss::party_seeds::from_bundle(bundle, u), held[u], 0, L, nullptr);
auto s_step = shuffle::shuffle_hidden_pass<std::uint64_t>(side,
rss::party_seeds::from_bundle(bundle, side), held[side], 0, L,
&u_step.out.data);
std::vector<std::uint64_t> side_msg = s_step.out.data;
auto l_step = shuffle::shuffle_hidden_pass<std::uint64_t>(L,
rss::party_seeds::from_bundle(bundle, L), held[L], 0, L, &side_msg);
if (step == pass)
{
const auto & msg = (me == u) ? u_step.out.data : s_step.out.data;
std::memcpy(out, msg.data(), out_n);
}
held[u] = std::move(u_step.view);
held[side] = std::move(s_step.view);
held[L] = std::move(l_step.view);
}
}
inline std::uint64_t heavy(kind k)
{
switch (k)
{
case kind::proj5:
return run_proj(5);
case kind::proj17:
return run_proj(17);
case kind::proj64:
return run_proj(64);
case kind::mul5:
return run_mul(5);
case kind::mul7:
return run_mul(7);
case kind::mul11:
return run_mul(11);
case kind::thresh8:
return run_thresh(8);
case kind::thresh16:
return run_thresh(16);
case kind::chain7:
return run_chain();
case kind::yao_if4:
return run_if(4, 2);
case kind::yao_if16:
return run_if(16, 8);
case kind::yao_hot4:
return run_hot(4);
case kind::yao_hot8:
return run_hot(8);
case kind::flute2:
return run_flute(2, 1);
case kind::flute4:
return run_flute(4, 1);
case kind::flute8:
return run_flute(8, 1);
case kind::flute4x8:
return run_flute(4, 8);
default:
return 0;
}
}
inline std::size_t payload_of(kind k)
{
switch (k)
{
case kind::proj5:
return proj_bytes(5);
case kind::proj17:
return proj_bytes(17);
case kind::proj64:
return proj_bytes(64);
case kind::mul5:
return mul_bytes(5);
case kind::mul7:
return mul_bytes(7);
case kind::mul11:
return mul_bytes(11);
case kind::thresh8:
return thresh_bytes(8);
case kind::thresh16:
return thresh_bytes(16);
case kind::chain7:
return chain_bytes();
case kind::yao_if4:
return yao_if_bytes(4, 2);
case kind::yao_if16:
return yao_if_bytes(16, 8);
case kind::yao_hot4:
return yao_hot_bytes(4);
case kind::yao_hot8:
return yao_hot_bytes(8);
case kind::flute2:
case kind::flute4:
case kind::flute8:
return 8;
case kind::flute4x8:
return 16;
case kind::shuf16:
return 16 * sizeof(std::uint64_t);
case kind::shuf64:
return 64 * sizeof(std::uint64_t);
case kind::shuf256:
return 256 * sizeof(std::uint64_t);
}
throw std::invalid_argument("bench: cell");
}
inline protocol::plan peer_plan(std::size_t, kind k)
{
protocol::composer c(0);
auto done = c.bench_peer(static_cast<std::uint32_t>(k), payload_of(k));
(void)done;
return c.default_plan();
}
inline protocol::plan ring_plan(std::size_t, kind k)
{
protocol::composer c(0);
auto done = c.bench_ring(static_cast<std::uint32_t>(k), payload_of(k));
(void)done;
return c.default_plan();
}
} // namespace detail
struct cell
{
const char * name;
int parties;
kind id;
};
/// @brief The battery. Secret indexes stay on the DPF plans beside these.
inline std::vector<cell> battery()
{
return {
{"arith_proj_m5", 2, kind::proj5},
{"arith_proj_m17", 2, kind::proj17},
{"arith_proj_m64", 2, kind::proj64},
{"arith_mul_p5", 2, kind::mul5},
{"arith_mul_p7", 2, kind::mul7},
{"arith_mul_p11", 2, kind::mul11},
{"arith_thresh_b8", 2, kind::thresh8},
{"arith_thresh_b16", 2, kind::thresh16},
{"arith_chain_mul4", 2, kind::chain7},
{"yao_if_4_2", 2, kind::yao_if4},
{"yao_if_16_8", 2, kind::yao_if16},
{"yao_onehot_k4", 2, kind::yao_hot4},
{"yao_onehot_k8", 2, kind::yao_hot8},
{"flute_d2", 2, kind::flute2},
{"flute_d4", 2, kind::flute4},
{"flute_d8", 2, kind::flute8},
{"flute_d4_o8", 2, kind::flute4x8},
{"shuffle_n16", 3, kind::shuf16},
{"shuffle_n64", 3, kind::shuf64},
{"shuffle_n256", 3, kind::shuf256},
};
}
inline protocol::plan plan_for(std::size_t party, kind id)
{
if (is_shuffle(id))
return detail::ring_plan(party, id);
return detail::peer_plan(party, id);
}
/// @brief Local half of a cell. Party 0 emits the payload. Party 1 applies it.
/// Every shuffle party emits its own ring slot.
inline void run_cell(std::uint32_t opcode, std::size_t party, std::uint32_t aux,
std::uint8_t * out, std::size_t out_n)
{
if (out == nullptr && out_n != 0)
throw std::invalid_argument("bench cell buffer");
const auto k = kind_of(aux);
if (is_shuffle(k))
{
if (opcode != protocol::opcodes::bench_emit)
return;
detail::shuffle_outbound(static_cast<unsigned>(party), pass_of(aux),
shuffle_n(k), out, out_n);
return;
}
if (opcode == protocol::opcodes::bench_apply)
{
if (party != 1)
{
if (out_n != 0)
out[0] = 0;
return;
}
const auto mix = detail::heavy(k);
if (out_n >= sizeof(mix))
std::memcpy(out, &mix, sizeof(mix));
return;
}
if (party != 0)
{
if (out_n != 0)
std::memset(out, 0, out_n);
return;
}
detail::paint(out, out_n, detail::heavy(k));
}
} // namespace bench
} // namespace dpf
#endif

View file

@ -15,7 +15,7 @@
/// @author Ryan Henry <ryan.henry@ucalgary.ca>
/// @copyright Copyright (c) 2019-2024 Ryan Henry and [others](@ref authors)
/// @license Released under a GNU General Public v2.0 (GPLv2) license;
/// see [LICENSE.md](@ref license) for details.
/// see LICENSE.md for details.
#ifndef LIBDPF_INCLUDE_DPF_BIT_HPP__
#define LIBDPF_INCLUDE_DPF_BIT_HPP__
@ -38,6 +38,14 @@ namespace dpf
{
/// @brief binary type whose representation can be packed into one bit
/// @note Not a DPF domain. `msb_of<dpf::bit>` does not compile. A 1-bit
/// index is `dpf::modint<1>` or `dpf::xint<1>`. Leaf `+` and `-` are
/// XOR. Lanes pack low-bit first.
/// @see dpf::twobit
/// @see dpf::nyble
/// @see dpf::modint
/// @see dpf::xint
/// @see output_types
enum bit : bool
{
zero = false, ///< `0`, `false`, "unset", "off"
@ -225,6 +233,10 @@ struct bitlength_of_output<dpf::bit, NodeT>
template <>
struct is_packed_subbyte<dpf::bit> : std::true_type {};
/// @brief Packed bits add by XOR (`bit::one + bit::one == bit::zero`).
template <>
struct has_characteristic_two<dpf::bit> : std::true_type {};
template <>
struct packed_lane_bits<dpf::bit>
: public std::integral_constant<std::size_t, 1> {};

View file

@ -43,6 +43,7 @@ template <typename Word>
HEDLEY_NO_THROW
constexpr void check_one_bit(Word mask) noexcept
{
(void)mask;
#if defined(__GNUC__) || defined(__clang__)
if (__builtin_is_constant_evaluated()) return;
#endif

View file

@ -0,0 +1,77 @@
/// @file dpf/bit_inject.hpp
/// @brief Boolean bit × arithmetic value (2PC bit_mul and 3PC RSS injection).
#ifndef LIBDPF_INCLUDE_DPF_BIT_INJECT_HPP__
#define LIBDPF_INCLUDE_DPF_BIT_INJECT_HPP__
#include <cstddef>
#include <cstdint>
#include <utility>
#include "hedley/hedley.h"
#include "dpf/beaver.hpp"
#include "dpf/rss_seed.hpp"
#include "dpf/secret_share.hpp"
namespace dpf
{
namespace bit_inject
{
/// @brief 2PC: `b * x` via a beaver bit_mul on an existing session.
template <typename Ring>
HEDLEY_WARN_UNUSED_RESULT
beavers::wire<Ring> inject2(beavers::session<Ring> & s,
beavers::wire<Ring> bit_wire, beavers::wire<Ring> arith_wire)
{
return s.bit_mul(bit_wire, arith_wire);
}
/// @brief Local y-factor of RSS bit injection before ring send.
/// @details Party holds RSS bit `(b_own, b_next)` and arithmetic `(x_own, x_next)`.
/// Local contribution mirrors RSS mul with the bit as a 0/1 factor.
template <typename T>
HEDLEY_WARN_UNUSED_RESULT
T rss_inject_local(const rss::party_seeds & seeds, T b_own, T b_next,
T x_own, T x_next, std::uint64_t index)
{
// Treat bit components as ring elements in {0,1}.
return rss::rss_mul_local(seeds, b_own, b_next, x_own, x_next, index);
}
/// @brief After ring refresh: party receives `y_prev` and forms RSS `(y_own, y_prev)`
/// wait — standard RSS refresh: send y_own to next, receive from prev,
/// store `(y_own, y_from_prev)`? Actually ABY3: party i holds y_i after
/// local mul; sends y_i to party i-1; ends with (y_i, y_{i+1}).
template <typename T>
HEDLEY_WARN_UNUSED_RESULT
std::pair<T, T> rss_refresh(T y_own, T y_from_next)
{
return {y_own, y_from_next};
}
/// @brief Cleartext identity: `(b0⊕b1) * (x0+x1)`.
template <typename Ring>
HEDLEY_WARN_UNUSED_RESULT
Ring inject_clear(std::uint8_t b0, std::uint8_t b1, Ring x0, Ring x1)
{
const Ring b = static_cast<Ring>((b0 ^ b1) & 1u);
return static_cast<Ring>(b * (x0 + x1));
}
/// @brief Boolean RSS AND local factor (GF(2)).
inline std::uint8_t rss_and_local(const rss::party_seeds & seeds,
std::uint8_t a_own, std::uint8_t a_next, std::uint8_t b_own,
std::uint8_t b_next, std::uint64_t index)
{
const std::uint8_t cross = static_cast<std::uint8_t>(
(a_own & b_own) ^ (a_own & b_next) ^ (a_next & b_own));
const std::uint8_t mask = static_cast<std::uint8_t>(
rss::zero_share<std::uint8_t>(seeds, index) & 1u);
return static_cast<std::uint8_t>(cross ^ mask);
}
} // namespace bit_inject
} // namespace dpf
#endif // LIBDPF_INCLUDE_DPF_BIT_INJECT_HPP__

781
include/dpf/bitmore_mod.hpp Normal file
View file

@ -0,0 +1,781 @@
/// @file dpf/bitmore_mod.hpp
/// @brief Byte-slot reduction for BitMore moduli through 255.
/// @details Hafiz and Henry (PoPETs 2019 §5.3) fold each server's DPF bits
/// into an integer and reduce modulo the server count. When that
/// count is not a power of two the integer does not fit in a byte
/// for long, so each byte is only partially reduced until the end.
///
/// A partial step reads the high nibble `h`. Every byte with that
/// nibble is at least `16*h`, so `floor(16*h / M) * M` is the largest
/// multiple of `M` that is safe to subtract. `pshufb` selects it and
/// `sub_epi8` removes it. The byte stays congruent modulo `M` and is
/// at most `partial_bound`. For moduli 121..127 that ceiling is 128
/// or more, so a second partial step is what leaves `stable_bound`
/// (at most 127). `resume_bound` is the ceiling the accumulator
/// actually resumes from: one step through modulus 120 and 128, two
/// steps for 121..127.
///
/// `add_budget = 255 - resume_bound` is how much can still be added
/// before a byte might reach 256. `shift_budget` is how many
/// `2*acc + bit` insertions fit in that slack. Both stay positive
/// through modulus 128, which is as far as a byte can hold two
/// resumed slots or one more bit. `partial_reduce` and `full_reduce`
/// themselves stay correct through 255: the two nibble residues sum
/// to at most 255 and to less than `2*M`, and one unsigned compare
/// subtracts `M`. Above 128 there is no slack left to defer that
/// correction across another add.
///
/// `bitmore_mod<M, 16>` is the same idea on 16-bit lanes, still on
/// AVX2. The top nibble (bits 12..15) selects `floor(4096*h / M)*M`
/// through two `pshufb`s, one per byte of that multiple, and
/// `sub_epi16` removes it. The accumulator has slack through modulus
/// 32768 (`resume_bound` at most 32767, so one more bit fits in the
/// lane). A full reduction sums the four nibble residues, which fit
/// in the lane for every modulus through 65535, then subtracts `M`
/// up to three times.
/// @copyright Copyright (c) 2019-2026 Ryan Henry and [others](@ref authors)
/// @license Released under a GNU General Public v2.0 (GPLv2) license.
#ifndef LIBDPF_INCLUDE_DPF_BITMORE_MOD_HPP__
#define LIBDPF_INCLUDE_DPF_BITMORE_MOD_HPP__
#include <array>
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <type_traits>
#include "hedley/hedley.h"
#include "simde/simde/x86/avx2.h"
namespace dpf
{
namespace bitmore_detail
{
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg zero() noexcept
{
if constexpr (std::is_same_v<Reg, simde__m256i>)
return simde_mm256_setzero_si256();
else
return simde_mm_setzero_si128();
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg load_lut(const std::array<unsigned char, 16> & lut) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
{
simde__m128i table;
std::memcpy(&table, lut.data(), 16);
return table;
}
else
{
alignas(32) unsigned char both[32];
std::memcpy(both, lut.data(), 16);
std::memcpy(both + 16, lut.data(), 16);
simde__m256i table;
std::memcpy(&table, both, 32);
return table;
}
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg high_nibble(Reg x) noexcept
{
// `srli_epi16` moves the neighbouring byte's low nibble into bits 4..7.
// Masking with `0x0f` leaves this byte's own high nibble.
if constexpr (std::is_same_v<Reg, simde__m128i>)
{
const auto m = simde_mm_set1_epi8(0x0f);
return simde_mm_and_si128(simde_mm_srli_epi16(x, 4), m);
}
else
{
const auto m = simde_mm256_set1_epi8(0x0f);
return simde_mm256_and_si256(simde_mm256_srli_epi16(x, 4), m);
}
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg low_nibble(Reg x) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_and_si128(x, simde_mm_set1_epi8(0x0f));
else
return simde_mm256_and_si256(x, simde_mm256_set1_epi8(0x0f));
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg shuffle(Reg table, Reg index) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_shuffle_epi8(table, index);
else
return simde_mm256_shuffle_epi8(table, index);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg add_bytes(Reg a, Reg b) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_add_epi8(a, b);
else
return simde_mm256_add_epi8(a, b);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg sub_bytes(Reg a, Reg b) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_sub_epi8(a, b);
else
return simde_mm256_sub_epi8(a, b);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg and_bytes(Reg a, Reg b) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_and_si128(a, b);
else
return simde_mm256_and_si256(a, b);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg xor_bytes(Reg a, Reg b) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_xor_si128(a, b);
else
return simde_mm256_xor_si256(a, b);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg splat_epi8(unsigned char v) noexcept
{
const auto s = static_cast<int8_t>(v);
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_set1_epi8(s);
else
return simde_mm256_set1_epi8(s);
}
/// @brief Unsigned `a > b` per byte. `cmpgt_epi8` is signed; XOR `0x80` maps
/// unsigned order onto that signed order.
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg cmpgt_epu8(Reg a, Reg b) noexcept
{
const auto bias = splat_epi8<Reg>(0x80);
const auto aa = xor_bytes<Reg>(a, bias);
const auto bb = xor_bytes<Reg>(b, bias);
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_cmpgt_epi8(aa, bb);
else
return simde_mm256_cmpgt_epi8(aa, bb);
}
/// @brief Shift each byte left by 1 and set bit 0 from `bit`.
/// @details `slli_epi16` spills bit 7 into the next byte of the 16-bit lane.
/// Clearing bit 0 afterwards drops that spill. Bit 7 of the odd byte
/// shifts out of the lane, which is the byte-local shift.
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg shift_in_bit(Reg acc, Reg bit) noexcept
{
const auto one = splat_epi8<Reg>(1);
const auto keep = splat_epi8<Reg>(0xfe);
bit = and_bytes<Reg>(bit, one);
Reg shifted;
if constexpr (std::is_same_v<Reg, simde__m128i>)
shifted = simde_mm_and_si128(simde_mm_slli_epi16(acc, 1), keep);
else
shifted = simde_mm256_and_si256(simde_mm256_slli_epi16(acc, 1), keep);
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_or_si128(shifted, bit);
else
return simde_mm256_or_si256(shifted, bit);
}
template <unsigned Modulus, unsigned LaneBits = 8>
HEDLEY_CONST
constexpr unsigned partial_bound() noexcept
{
static_assert(LaneBits == 8u || LaneBits == 16u, "bitmore lane width is 8 or 16");
constexpr unsigned block = LaneBits == 8u ? 16u : 4096u;
constexpr unsigned tail = block - 1u;
unsigned u = 0;
for (unsigned h = 0; h < 16u; ++h)
{
const unsigned q = (block * h / Modulus) * Modulus;
const unsigned top = block * h + tail - q;
if (top > u)
u = top;
}
return u;
}
template <unsigned Modulus, unsigned LaneBits = 8>
HEDLEY_CONST
constexpr unsigned stable_bound() noexcept
{
constexpr unsigned block = LaneBits == 8u ? 16u : 4096u;
const unsigned lim = partial_bound<Modulus, LaneBits>();
unsigned u = 0;
for (unsigned h = 0; h < 16u; ++h)
{
const unsigned lo = block * h;
if (lo > lim)
break;
unsigned hi = lo + block - 1u;
if (hi > lim)
hi = lim;
const unsigned q = (lo / Modulus) * Modulus;
const unsigned top = hi - q;
if (top > u)
u = top;
}
return u;
}
template <unsigned Modulus>
HEDLEY_CONST
constexpr unsigned shift_budget(unsigned bound, unsigned capacity = 256u) noexcept
{
unsigned k = 0;
unsigned span = bound + 1u;
const unsigned half = capacity >> 1;
while (span <= half)
{
span *= 2u;
++k;
}
return k;
}
template <unsigned Modulus>
HEDLEY_CONST
constexpr std::array<unsigned char, 16> partial_lut() noexcept
{
std::array<unsigned char, 16> lut{};
for (unsigned h = 0; h < 16u; ++h)
lut[h] = static_cast<unsigned char>((16u * h / Modulus) * Modulus);
return lut;
}
template <unsigned Modulus>
HEDLEY_CONST
constexpr std::array<unsigned char, 16> low_residue_lut() noexcept
{
std::array<unsigned char, 16> lut{};
for (unsigned n = 0; n < 16u; ++n)
lut[n] = static_cast<unsigned char>(n % Modulus);
return lut;
}
template <unsigned Modulus>
HEDLEY_CONST
constexpr std::array<unsigned char, 16> high_residue_lut() noexcept
{
std::array<unsigned char, 16> lut{};
for (unsigned h = 0; h < 16u; ++h)
lut[h] = static_cast<unsigned char>((16u * h) % Modulus);
return lut;
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg splat_epi16(unsigned v) noexcept
{
const auto s = static_cast<std::int16_t>(static_cast<std::uint16_t>(v));
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_set1_epi16(s);
else
return simde_mm256_set1_epi16(s);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg add_epi16(Reg a, Reg b) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_add_epi16(a, b);
else
return simde_mm256_add_epi16(a, b);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg sub_epi16(Reg a, Reg b) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_sub_epi16(a, b);
else
return simde_mm256_sub_epi16(a, b);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg srli_epi16(Reg a, int imm) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_srli_epi16(a, imm);
else
return simde_mm256_srli_epi16(a, imm);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg slli_epi16(Reg a, int imm) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_slli_epi16(a, imm);
else
return simde_mm256_slli_epi16(a, imm);
}
/// @brief Unsigned `a > b` per 16-bit lane.
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg cmpgt_epu16(Reg a, Reg b) noexcept
{
const auto bias = splat_epi16<Reg>(0x8000u);
const auto aa = xor_bytes<Reg>(a, bias);
const auto bb = xor_bytes<Reg>(b, bias);
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_cmpgt_epi16(aa, bb);
else
return simde_mm256_cmpgt_epi16(aa, bb);
}
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg or_bytes(Reg a, Reg b) noexcept
{
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_or_si128(a, b);
else
return simde_mm256_or_si256(a, b);
}
/// @brief Look up a 16-bit table entry. `idx` holds the nibble in the low byte
/// of each lane and zero in the high byte, so `pshufb` writes `lut[h]`
/// into the low byte and `lut[0]` into the high byte. `lut[0]` is 0.
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg lookup_u16(Reg idx, const std::array<unsigned char, 16> & lo,
const std::array<unsigned char, 16> & hi) noexcept
{
const auto lo_v = shuffle<Reg>(load_lut<Reg>(lo), idx);
const auto hi_v = slli_epi16<Reg>(shuffle<Reg>(load_lut<Reg>(hi), idx), 8);
return or_bytes<Reg>(lo_v, hi_v);
}
constexpr std::array<unsigned char, 16> u16_lo(const std::array<std::uint16_t, 16> & v) noexcept
{
std::array<unsigned char, 16> out{};
for (unsigned i = 0; i < 16u; ++i)
out[i] = static_cast<unsigned char>(v[i] & 0xffu);
return out;
}
constexpr std::array<unsigned char, 16> u16_hi(const std::array<std::uint16_t, 16> & v) noexcept
{
std::array<unsigned char, 16> out{};
for (unsigned i = 0; i < 16u; ++i)
out[i] = static_cast<unsigned char>(v[i] >> 8);
return out;
}
template <unsigned Modulus>
HEDLEY_CONST
constexpr std::array<std::uint16_t, 16> wide_partial_lut() noexcept
{
std::array<std::uint16_t, 16> lut{};
for (unsigned h = 0; h < 16u; ++h)
lut[h] = static_cast<std::uint16_t>((4096u * h / Modulus) * Modulus);
return lut;
}
template <unsigned Modulus, unsigned Place>
HEDLEY_CONST
constexpr std::array<std::uint16_t, 16> wide_residue_lut() noexcept
{
std::array<std::uint16_t, 16> lut{};
for (unsigned n = 0; n < 16u; ++n)
lut[n] = static_cast<std::uint16_t>(
(static_cast<std::uint32_t>(Place) * n) % Modulus);
return lut;
}
/// @brief `acc = 2*acc + bit0` inside each 16-bit lane. `slli_epi16` does not
/// cross lanes.
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
Reg shift_in_bit16(Reg acc, Reg bit) noexcept
{
const auto one = splat_epi16<Reg>(1u);
bit = and_bytes<Reg>(bit, one);
if constexpr (std::is_same_v<Reg, simde__m128i>)
return simde_mm_or_si128(simde_mm_slli_epi16(acc, 1), bit);
else
return simde_mm256_or_si256(simde_mm256_slli_epi16(acc, 1), bit);
}
} // namespace bitmore_detail
/// @brief Partial and full reduction of `LaneBits`-wide slots modulo `Modulus`.
/// @tparam Modulus server count. `2` through `255` for bytes, `2` through `65535` for 16-bit lanes.
/// @tparam LaneBits `8` (one byte per slot) or `16` (one AVX2 `epi16` lane per slot)
template <unsigned Modulus, unsigned LaneBits = 8>
struct bitmore_mod
{
static_assert(LaneBits == 8u || LaneBits == 16u,
"bitmore lane width is 8 or 16");
static_assert(Modulus >= 2u, "bitmore modulus must be at least 2");
static_assert(LaneBits == 16u || Modulus <= 255u,
"bitmore byte reduction: modulus must be in 2..255");
static_assert(LaneBits == 8u || Modulus <= 65535u,
"bitmore 16-bit reduction: modulus must be in 2..65535");
static constexpr unsigned modulus = Modulus;
static constexpr unsigned lane_bits = LaneBits;
static constexpr unsigned slot_max = LaneBits == 8u ? 255u : 65535u;
static constexpr unsigned nibble_block = LaneBits == 8u ? 16u : 4096u;
static constexpr unsigned nibble_shift = LaneBits == 8u ? 4u : 12u;
/// @brief Largest slot one top-nibble partial reduction can leave.
static constexpr unsigned partial_bound = bitmore_detail::partial_bound<Modulus, LaneBits>();
/// @brief Largest slot a second partial step can leave, starting from `partial_bound`.
static constexpr unsigned stable_bound = bitmore_detail::stable_bound<Modulus, LaneBits>();
/// @brief Ceiling the accumulator resumes from. One step when that already
/// fits in the lower half of the lane; otherwise the second step.
static constexpr unsigned resume_bound =
partial_bound <= (slot_max >> 1) ? partial_bound : stable_bound;
/// @brief How much can be added to a resumed slot before it might reach `slot_max + 1`.
static constexpr unsigned add_budget = slot_max - resume_bound;
/// @brief Bit insertions that fit in `add_budget` after resuming.
static constexpr unsigned shift_budget =
bitmore_detail::shift_budget<Modulus>(resume_bound, slot_max + 1u);
/// @brief `x - floor(block*(x>>shift) / M) * M` inside one slot.
/// \complexity One division of a nibble index.
HEDLEY_CONST
HEDLEY_ALWAYS_INLINE
static constexpr unsigned partial_reduce_slot(unsigned x) noexcept
{
x &= slot_max;
const unsigned h = x >> nibble_shift;
const unsigned q = (nibble_block * h / Modulus) * Modulus;
return x - q;
}
/// @brief Byte-slot name for `partial_reduce_slot`.
HEDLEY_CONST
HEDLEY_ALWAYS_INLINE
static constexpr unsigned partial_reduce_byte(unsigned x) noexcept
{
static_assert(LaneBits == 8u, "partial_reduce_byte is the 8-bit slot");
return partial_reduce_slot(x);
}
/// @brief `x mod Modulus` for one slot.
/// \complexity One remainder of a slot.
HEDLEY_CONST
HEDLEY_ALWAYS_INLINE
static constexpr unsigned full_reduce_slot(unsigned x) noexcept
{
return (x & slot_max) % Modulus;
}
/// @brief Byte-slot name for `full_reduce_slot`.
HEDLEY_CONST
HEDLEY_ALWAYS_INLINE
static constexpr unsigned full_reduce_byte(unsigned x) noexcept
{
static_assert(LaneBits == 8u, "full_reduce_byte is the 8-bit slot");
return full_reduce_slot(x);
}
/// @brief Partial-reduce every byte of `x`.
/// @details The high nibble selects `floor(16*h / M) * M`. Subtracting it
/// preserves the residue and leaves a byte of at most `partial_bound`.
/// \complexity One `pshufb` and one `sub_epi8` per register.
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
HEDLEY_PURE
static Reg partial_reduce(Reg x) noexcept
{
static_assert(std::is_same_v<Reg, simde__m128i> || std::is_same_v<Reg, simde__m256i>,
"bitmore partial_reduce: register must be simde__m128i or simde__m256i");
if constexpr (LaneBits == 8u)
{
constexpr auto lut = bitmore_detail::partial_lut<Modulus>();
const auto q = bitmore_detail::shuffle<Reg>(
bitmore_detail::load_lut<Reg>(lut),
bitmore_detail::high_nibble<Reg>(x));
return bitmore_detail::sub_bytes<Reg>(x, q);
}
else
{
constexpr auto lut = bitmore_detail::wide_partial_lut<Modulus>();
constexpr auto lo = bitmore_detail::u16_lo(lut);
constexpr auto hi = bitmore_detail::u16_hi(lut);
const auto idx = bitmore_detail::srli_epi16<Reg>(x, 12);
const auto q = bitmore_detail::lookup_u16<Reg>(idx, lo, hi);
return bitmore_detail::sub_epi16<Reg>(x, q);
}
}
/// @brief Fully reduce every byte of `x` into `0 .. Modulus-1`.
/// @details `pshufb` maps the low nibble to itself modulo `M` and the high
/// nibble to `(16*h) mod M`. The sum is less than `2*M`, so one
/// compare subtracts `M` where the sum is still too big.
/// \complexity Two `pshufb`s, one byte add, one compare, one byte subtract.
template <typename Reg>
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
HEDLEY_PURE
static Reg full_reduce(Reg x) noexcept
{
static_assert(std::is_same_v<Reg, simde__m128i> || std::is_same_v<Reg, simde__m256i>,
"bitmore full_reduce: register must be simde__m128i or simde__m256i");
if constexpr (LaneBits == 8u)
{
constexpr auto lo_lut = bitmore_detail::low_residue_lut<Modulus>();
constexpr auto hi_lut = bitmore_detail::high_residue_lut<Modulus>();
const auto lo = bitmore_detail::shuffle<Reg>(
bitmore_detail::load_lut<Reg>(lo_lut),
bitmore_detail::low_nibble<Reg>(x));
const auto hi = bitmore_detail::shuffle<Reg>(
bitmore_detail::load_lut<Reg>(hi_lut),
bitmore_detail::high_nibble<Reg>(x));
const auto sum = bitmore_detail::add_bytes<Reg>(lo, hi);
// Sum of the two residues is at most 255 and less than `2*M`, so one
// subtraction finishes the byte. The compare is unsigned: for M > 120
// the sum can exceed 127.
const auto limit = bitmore_detail::splat_epi8<Reg>(
static_cast<unsigned char>(Modulus - 1u));
const auto ge = bitmore_detail::cmpgt_epu8<Reg>(sum, limit);
const auto corr = bitmore_detail::and_bytes<Reg>(
ge, bitmore_detail::splat_epi8<Reg>(static_cast<unsigned char>(Modulus)));
return bitmore_detail::sub_bytes<Reg>(sum, corr);
}
else
{
// Four nibble residues, each `< M`. Their sum fits in a 16-bit lane
// for every modulus through 65535 and is less than `4*M`, so three
// conditional subtractions finish the lane.
constexpr auto r0 = bitmore_detail::wide_residue_lut<Modulus, 1u>();
constexpr auto r1 = bitmore_detail::wide_residue_lut<Modulus, 16u>();
constexpr auto r2 = bitmore_detail::wide_residue_lut<Modulus, 256u>();
constexpr auto r3 = bitmore_detail::wide_residue_lut<Modulus, 4096u>();
constexpr auto r0_lo = bitmore_detail::u16_lo(r0);
constexpr auto r0_hi = bitmore_detail::u16_hi(r0);
constexpr auto r1_lo = bitmore_detail::u16_lo(r1);
constexpr auto r1_hi = bitmore_detail::u16_hi(r1);
constexpr auto r2_lo = bitmore_detail::u16_lo(r2);
constexpr auto r2_hi = bitmore_detail::u16_hi(r2);
constexpr auto r3_lo = bitmore_detail::u16_lo(r3);
constexpr auto r3_hi = bitmore_detail::u16_hi(r3);
const auto nib = bitmore_detail::splat_epi16<Reg>(0x000fu);
const auto n0 = bitmore_detail::and_bytes<Reg>(x, nib);
const auto n1 = bitmore_detail::and_bytes<Reg>(bitmore_detail::srli_epi16<Reg>(x, 4), nib);
const auto n2 = bitmore_detail::and_bytes<Reg>(bitmore_detail::srli_epi16<Reg>(x, 8), nib);
const auto n3 = bitmore_detail::srli_epi16<Reg>(x, 12);
auto sum = bitmore_detail::lookup_u16<Reg>(n0, r0_lo, r0_hi);
sum = bitmore_detail::add_epi16<Reg>(sum, bitmore_detail::lookup_u16<Reg>(n1, r1_lo, r1_hi));
sum = bitmore_detail::add_epi16<Reg>(sum, bitmore_detail::lookup_u16<Reg>(n2, r2_lo, r2_hi));
sum = bitmore_detail::add_epi16<Reg>(sum, bitmore_detail::lookup_u16<Reg>(n3, r3_lo, r3_hi));
const auto limit = bitmore_detail::splat_epi16<Reg>(Modulus - 1u);
const auto modv = bitmore_detail::splat_epi16<Reg>(Modulus);
for (int step = 0; step < 3; ++step)
{
const auto ge = bitmore_detail::cmpgt_epu16<Reg>(sum, limit);
const auto corr = bitmore_detail::and_bytes<Reg>(ge, modv);
sum = bitmore_detail::sub_epi16<Reg>(sum, corr);
}
return sum;
}
}
};
/// @brief Running slot sum modulo `Modulus`, partial-reduced on a budget.
/// @details `bound()` is a proven upper bound on every slot. An add whose
/// worst-case total would pass the lane maximum partial-reduces the
/// accumulator, then the addend. A modulus whose first partial step
/// still sets the high bit takes a second step before two slots fit
/// again. `insert_bit` is the MSB-first fold `acc = 2*acc + bit`.
/// `reduced()` is the full per-slot residue. The byte accumulator
/// stops at modulus 128 and the 16-bit accumulator at 32768; past
/// that a resumed slot no longer fits next to another or under a
/// shift. `partial_reduce` and `full_reduce` cover the wider ranges.
/// @tparam Modulus server count, `2` through `128` for bytes and `32768` for 16-bit lanes
/// @tparam Reg `simde__m128i` or `simde__m256i`
/// @tparam LaneBits `8` or `16`
template <unsigned Modulus, typename Reg, unsigned LaneBits = 8>
class bitmore_accumulator
{
using mod = bitmore_mod<Modulus, LaneBits>;
static_assert(std::is_same_v<Reg, simde__m128i> || std::is_same_v<Reg, simde__m256i>,
"bitmore_accumulator: register must be simde__m128i or simde__m256i");
static_assert((LaneBits == 8u && Modulus <= 128u) || (LaneBits == 16u && Modulus <= 32768u),
"bitmore_accumulator: no slack past modulus 128 in a byte or 32768 in a 16-bit slot");
static_assert(mod::stable_bound <= (mod::slot_max >> 1),
"bitmore_accumulator: second partial step must leave room for one bit");
static_assert(mod::stable_bound * 2u <= mod::slot_max,
"bitmore_accumulator: two resumed slots must fit in one lane");
static_assert(mod::shift_budget >= 1u,
"bitmore_accumulator: at least one bit insertion after resuming");
public:
/// \complexity One zeroed register.
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
bitmore_accumulator() noexcept
: acc_(bitmore_detail::zero<Reg>()), bound_(0)
{ }
/// @brief Proven upper bound on every slot in `value()`.
HEDLEY_PURE
HEDLEY_ALWAYS_INLINE
unsigned bound() const noexcept
{
return bound_;
}
/// @brief Unreduced slots. Each is `≤ bound()` and congruent to the sum.
HEDLEY_ALWAYS_INLINE
Reg value() const noexcept
{
return acc_;
}
/// @brief Add one slot.
/// @param x addends. Each slot must be `≤ addend_max`.
/// @param addend_max worst-case slot in `x`, clamped to `slot_max`. Pass
/// `slot_max` when the addend is an arbitrary lane; the accumulator
/// partial-reduces it if the slack cannot absorb that.
/// \complexity At most four partial reductions and one lane add.
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
void add(Reg x, unsigned addend_max) noexcept
{
if (addend_max > mod::slot_max)
addend_max = mod::slot_max;
for (;;)
{
if (bound_ + addend_max <= mod::slot_max)
break;
if (bound_ > mod::partial_bound)
{
acc_ = mod::partial_reduce(acc_);
bound_ = mod::partial_bound;
continue;
}
if (addend_max > mod::partial_bound)
{
x = mod::partial_reduce(x);
addend_max = mod::partial_bound;
continue;
}
if (bound_ > mod::stable_bound)
{
acc_ = mod::partial_reduce(acc_);
bound_ = mod::stable_bound;
continue;
}
if (addend_max > mod::stable_bound)
{
x = mod::partial_reduce(x);
addend_max = mod::stable_bound;
continue;
}
break;
}
if constexpr (LaneBits == 8u)
acc_ = bitmore_detail::add_bytes<Reg>(acc_, x);
else
acc_ = bitmore_detail::add_epi16<Reg>(acc_, x);
bound_ += addend_max;
}
/// @brief Fold one BitMore bit: `acc = 2*acc + bit0` inside each slot.
/// @param bit bit 0 of each slot is the new low bit. Higher bits are ignored.
/// \complexity Up to two partial reductions when the proven bound sets the
/// lane's high bit, then a shift and an OR.
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
void insert_bit(Reg bit) noexcept
{
while (bound_ > (mod::slot_max >> 1))
{
acc_ = mod::partial_reduce(acc_);
bound_ = bound_ > mod::partial_bound ? mod::partial_bound
: mod::stable_bound;
}
if constexpr (LaneBits == 8u)
acc_ = bitmore_detail::shift_in_bit<Reg>(acc_, bit);
else
acc_ = bitmore_detail::shift_in_bit16<Reg>(acc_, bit);
bound_ = bound_ * 2u + 1u;
}
/// @brief Full residue of every slot, in `0 .. Modulus-1`.
/// \complexity One `full_reduce` of the register.
HEDLEY_ALWAYS_INLINE
HEDLEY_NO_THROW
HEDLEY_PURE
Reg reduced() const noexcept
{
return mod::full_reduce(acc_);
}
private:
Reg acc_;
unsigned bound_;
};
} // namespace dpf
#endif // LIBDPF_INCLUDE_DPF_BITMORE_MOD_HPP__

Some files were not shown because too many files have changed in this diff Show more