Checkpoint the party/runtime stack before share-program and malicious-mode work.

Ship the TLS mesh, composer, Beaver/Yao/leaf MPC, prep/online paths, apps, and docs so the tree is pushable before elevating share_expr, security_mode, and prep resume.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Ryan Henry 2026-09-28 05:59:19 -06:00
parent 695f8e84f7
commit 0d22946a0e
1835 changed files with 170291 additions and 2849 deletions

Binary file not shown.

505
test/profile/cost_matrix.py Executable file
View file

@ -0,0 +1,505 @@
#!/usr/bin/env python3
"""Cost matrix for libdpf profile harnesses.
Each cell is one named case. Cells are grouped into slices so an optimization
pass can rerun only the code it touched. A label is a directory of results.
Rerunning a slice under the same label replaces those rows and leaves the rest.
Examples:
test/profile/cost_matrix.py --list
test/profile/cost_matrix.py --label baseline
test/profile/cost_matrix.py --label baseline --slice interval
test/profile/cost_matrix.py --label baseline --slice memoizer --slice bits
test/profile/cost_matrix.py --label baseline --family grotto --slice horner
test/profile/cost_matrix.py --label baseline --case interval_u32_L4096
test/profile/cost_matrix.py --label after-horner --slice horner
test/profile/cost_matrix.py --compare baseline after-horner
Tiers:
std default. Benchable party flows, plus every in-process cell.
heavy wildcard_single_leaf, long beaver streams, full-domain DCFs,
and party flows whose names end in _domain.
all std and heavy.
smoke party flows that are negative tests, not cost points.
Default repeats are per slice (party 3, tables 20, walks 12). --repeat
overrides every selected cell.
Columns (blank when not instrumented):
prg_evals, bytes_sent, bytes_recv_from_p2, bytes_recv_from_peer,
rounds (channel exchange barriers; not frames), preprocess_bytes,
alloc_bytes, logical_bytes, layout_waste (alloc-logical),
payload_sent/recv, wire_overhead (framed bytes minus payload).
For p2, bytes_*_from_peer / bytes_*_to_peer sum the p0 and p1 links.
"""
from __future__ import annotations
import argparse
import csv
import os
import subprocess
import sys
from collections import defaultdict
from pathlib import Path
LONG_FIELDS = [
"family",
"slice",
"tier",
"case",
"role",
"items",
"repeat",
"warmup",
"avg_ns",
"min_ns",
"max_ns",
"avg_cycles",
"per_item_ns",
"out_bytes",
"prg_evals",
"preprocess_bytes",
"alloc_bytes",
"logical_bytes",
"layout_waste",
"bytes_sent",
"bytes_recv",
"bytes_recv_from_p2",
"bytes_recv_from_peer",
"bytes_sent_to_p2",
"bytes_sent_to_peer",
"payload_sent",
"payload_recv",
"wire_overhead",
"rounds",
"frames_sent",
"frames_recv",
"avg_bytes_sent",
"avg_frames_sent",
"wall_ms",
"rc",
"sink",
]
MATRIX_FIELDS = [
"family",
"slice",
"tier",
"case",
"role",
"avg_ns",
"per_item_ns",
"avg_cycles",
"out_bytes",
"prg_evals",
"preprocess_bytes",
"alloc_bytes",
"logical_bytes",
"layout_waste",
"bytes_sent",
"bytes_recv",
"bytes_recv_from_p2",
"bytes_recv_from_peer",
"rounds",
"wire_overhead",
"frames_sent",
"frames_recv",
"wall_ms",
"repeat",
]
COMPARE_NUM = [
"avg_ns",
"prg_evals",
"bytes_sent",
"bytes_recv_from_p2",
"bytes_recv_from_peer",
"rounds",
"preprocess_bytes",
"alloc_bytes",
"wire_overhead",
]
def default_bin_dir() -> Path:
here = Path(__file__).resolve().parent
candidate = here.parents[1] / "build" / "test" / "bin"
return candidate
def timing_for(family: str, slice_name: str, tier: str) -> tuple[int, int]:
if tier == "heavy":
return 1, 0
if family == "party":
return 3, 1
if slice_name in {"window", "principal", "closed", "reduced", "exact"}:
return 20, 2
if slice_name == "keygen":
return 8, 1
if slice_name in {"memoizer", "bits", "inner-product", "helpers"}:
return 12, 2
return 12, 2
def run_capture(cmd: list[str]) -> tuple[int, str, str]:
proc = subprocess.run(cmd, text=True, capture_output=True)
return proc.returncode, proc.stdout, proc.stderr
def parse_list(text: str) -> list[dict[str, str]]:
rows = []
reader = csv.DictReader(text.splitlines(), delimiter="\t")
for row in reader:
if not row.get("case"):
continue
rows.append(row)
return rows
def discover(bin_dir: Path) -> list[dict[str, str]]:
cells: list[dict[str, str]] = []
for binary in ("profile_eval", "profile_grotto", "profile_party"):
path = bin_dir / binary
if not path.is_file():
raise SystemExit(f"missing {path}; build the profile targets first")
cmd = [str(path), "--list"]
if binary == "profile_party":
cmd.extend(["--tier", "all"])
rc, out, err = run_capture(cmd)
if rc != 0:
raise SystemExit(err or out or f"{binary} --list failed")
if err.strip():
print(err, file=sys.stderr, end="" if err.endswith("\n") else "\n")
for row in parse_list(out):
row["driver"] = binary
row.setdefault("tier", "std")
cells.append(row)
heavy_party = [
"wildcard_single_leaf",
"beaver_stream_n512",
"beaver_stream_n2048",
]
for row in cells:
if row["family"] == "party" and (
row["case"] in heavy_party or row["case"].startswith("dcf_full_")
):
row["tier"] = "heavy"
return cells
def select_cells(cells, args) -> list[dict[str, str]]:
chosen = []
for row in cells:
if args.family and row["family"] not in args.family:
continue
if args.slice and row["slice"] not in args.slice:
continue
if args.case and row["case"] not in args.case:
continue
tier = row.get("tier") or "std"
if args.tier == "std" and tier != "std":
continue
if args.tier == "heavy" and tier != "heavy":
continue
if args.tier == "smoke" and tier != "smoke":
continue
if args.tier == "all" and tier == "smoke":
continue
chosen.append(row)
return chosen
def blank_cost_fields() -> dict[str, str]:
return {
"prg_evals": "",
"preprocess_bytes": "",
"alloc_bytes": "",
"logical_bytes": "",
"layout_waste": "",
"bytes_sent": "",
"bytes_recv": "",
"bytes_recv_from_p2": "",
"bytes_recv_from_peer": "",
"bytes_sent_to_p2": "",
"bytes_sent_to_peer": "",
"payload_sent": "",
"payload_recv": "",
"wire_overhead": "",
"rounds": "",
"frames_sent": "",
"frames_recv": "",
"avg_bytes_sent": "",
"avg_frames_sent": "",
"wall_ms": "",
}
def parse_inprocess(text: str, tier_of: dict[tuple[str, str], str]) -> list[dict[str, str]]:
rows = []
reader = csv.DictReader(text.splitlines(), delimiter="\t")
for row in reader:
if not row or row.get("family") == "sink" or not row.get("case"):
continue
# Skip the trailing sink summary line if DictReader mis-parses it.
if row.get("family") == "sink":
continue
out = blank_cost_fields()
out.update({
"family": row.get("family") or "",
"slice": row.get("slice") or "",
"tier": tier_of.get((row.get("family") or "", row.get("case") or ""), "std"),
"case": row.get("case") or "",
"role": "-",
"items": row.get("items") or "",
"repeat": row.get("repeat") or "",
"warmup": row.get("warmup") or "",
"avg_ns": row.get("avg_ns") or "",
"min_ns": row.get("min_ns") or "",
"max_ns": row.get("max_ns") or "",
"avg_cycles": row.get("avg_cycles") or "",
"per_item_ns": row.get("per_item_ns") or "",
"out_bytes": row.get("out_bytes") or "",
"prg_evals": row.get("prg_evals") or "",
"preprocess_bytes": row.get("preprocess_bytes") or "",
"alloc_bytes": row.get("alloc_bytes") or "",
"logical_bytes": row.get("logical_bytes") or "",
"layout_waste": row.get("layout_waste") or "",
"rc": "0",
"sink": row.get("sink") or "",
})
rows.append(out)
return rows
def parse_party(text: str) -> list[dict[str, str]]:
rows = []
reader = csv.DictReader(text.splitlines(), delimiter="\t")
for row in reader:
if not row.get("flow"):
continue
out = blank_cost_fields()
out.update({
"family": row.get("family") or "party",
"slice": row.get("slice") or "",
"tier": row.get("tier") or "std",
"case": row["flow"],
"role": row.get("role") or "-",
"items": "1",
"repeat": "",
"warmup": "",
"avg_ns": row.get("avg_ns") or "",
"min_ns": row.get("min_ns") or "",
"max_ns": row.get("max_ns") or "",
"avg_cycles": "",
"per_item_ns": row.get("avg_ns") or "",
"out_bytes": "",
"prg_evals": row.get("prg_evals") or "",
"bytes_sent": row.get("bytes_sent") or "",
"bytes_recv": row.get("bytes_recv") or "",
"bytes_recv_from_p2": row.get("bytes_recv_from_p2") or "",
"bytes_recv_from_peer": row.get("bytes_recv_from_peer") or "",
"bytes_sent_to_p2": row.get("bytes_sent_to_p2") or "",
"bytes_sent_to_peer": row.get("bytes_sent_to_peer") or "",
"payload_sent": row.get("payload_sent") or "",
"payload_recv": row.get("payload_recv") or "",
"wire_overhead": row.get("wire_overhead") or "",
"rounds": row.get("rounds") or "",
"frames_sent": row.get("frames_sent") or "",
"frames_recv": row.get("frames_recv") or "",
"avg_bytes_sent": row.get("avg_bytes_sent") or "",
"avg_frames_sent": row.get("avg_frames_sent") or "",
"wall_ms": row.get("wall_ms") or "",
"rc": row.get("rc") or "",
"sink": "",
})
rows.append(out)
return rows
def fill_timing(rows: list[dict[str, str]], repeat: int, warmup: int) -> None:
for row in rows:
if not row["repeat"]:
row["repeat"] = str(repeat)
if not row["warmup"]:
row["warmup"] = str(warmup)
def invoke_group(bin_dir: Path, driver: str, cases: list[dict[str, str]],
repeat: int, warmup: int) -> tuple[int, list[dict[str, str]], str]:
cmd = [str(bin_dir / driver), "--repeat", str(repeat), "--warmup", str(warmup)]
for case in cases:
cmd.extend(["--case", case["case"]])
rc, out, err = run_capture(cmd)
tier_of = {(c["family"], c["case"]): c.get("tier") or "std" for c in cases}
if driver == "profile_party":
parsed = parse_party(out)
else:
parsed = parse_inprocess(out, tier_of)
fill_timing(parsed, repeat, warmup)
return rc, parsed, err
def write_tsv(path: Path, fields: list[str], rows: list[dict[str, str]]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", newline="") as fh:
writer = csv.DictWriter(fh, fieldnames=fields, delimiter="\t", extrasaction="ignore")
writer.writeheader()
for row in rows:
writer.writerow(row)
def read_tsv(path: Path) -> list[dict[str, str]]:
if not path.is_file():
return []
with path.open(newline="") as fh:
return list(csv.DictReader(fh, delimiter="\t"))
def merge_rows(old: list[dict[str, str]], new: list[dict[str, str]]) -> list[dict[str, str]]:
replaced = {(row["family"], row["slice"], row["case"], row["role"]) for row in new}
kept = [
row for row in old
if (row.get("family"), row.get("slice"), row.get("case"), row.get("role")) not in replaced
]
# Older long.tsv rows may lack new columns; normalize on merge.
merged = []
for row in kept + new:
full = blank_cost_fields()
full.update({k: "" for k in LONG_FIELDS})
full.update({k: v for k, v in row.items() if v is not None})
merged.append(full)
return merged
def matrix_view(rows: list[dict[str, str]]) -> list[dict[str, str]]:
view = []
for row in rows:
view.append({key: row.get(key, "") for key in MATRIX_FIELDS})
view.sort(key=lambda r: (r["family"], r["slice"], r["case"], r["role"]))
return view
def print_catalog(cells: list[dict[str, str]]) -> None:
groups: dict[tuple[str, str, str], int] = defaultdict(int)
for row in cells:
groups[(row["family"], row["slice"], row.get("tier") or "std")] += 1
print("family\tslice\ttier\tcases")
for (family, slice_name, tier), count in sorted(groups.items()):
print(f"{family}\t{slice_name}\t{tier}\t{count}")
print(f"# {len(cells)} cases in {len(groups)} slices")
def compare_labels(out_root: Path, left: str, right: str) -> int:
a_rows = read_tsv(out_root / left / "matrix.tsv")
b_rows = read_tsv(out_root / right / "matrix.tsv")
if not a_rows or not b_rows:
raise SystemExit(f"need matrix.tsv in both {left} and {right}")
b_index = {
(row["family"], row["slice"], row["case"], row["role"]): row
for row in b_rows
}
header = ["family", "slice", "case", "role"]
for col in COMPARE_NUM:
header.extend([f"{col}_a", f"{col}_b", f"delta_{col}"])
print("\t".join(header))
missing = 0
for row in a_rows:
key = (row["family"], row["slice"], row["case"], row["role"])
other = b_index.get(key)
if other is None:
missing += 1
continue
parts = [row["family"], row["slice"], row["case"], row["role"]]
for col in COMPARE_NUM:
av_s = row.get(col, "") or ""
bv_s = other.get(col, "") or ""
try:
av = float(av_s)
bv = float(bv_s)
delta = f"{bv - av:.0f}"
except ValueError:
delta = ""
parts.extend([av_s, bv_s, delta])
print("\t".join(parts))
if missing:
print(f"# {missing} rows in {left} have no match in {right}", file=sys.stderr)
return 0
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--bin-dir", type=Path, default=default_bin_dir())
parser.add_argument("--out", type=Path, default=None,
help="directory that holds one subdirectory per label")
parser.add_argument("--label", default="baseline")
parser.add_argument("--family", action="append", default=[])
parser.add_argument("--slice", action="append", default=[])
parser.add_argument("--case", action="append", default=[])
parser.add_argument("--tier", choices=("std", "heavy", "all", "smoke"), default="std")
parser.add_argument("--repeat", type=int, default=None)
parser.add_argument("--warmup", type=int, default=None)
parser.add_argument("--list", action="store_true")
parser.add_argument("--compare", nargs=2, metavar=("LABEL_A", "LABEL_B"))
args = parser.parse_args()
out_root = args.out
if out_root is None:
out_root = args.bin_dir.parent / "cost-matrix"
if args.compare:
return compare_labels(out_root, args.compare[0], args.compare[1])
cells = discover(args.bin_dir)
chosen = select_cells(cells, args)
if args.list:
print_catalog(chosen)
return 0
if not chosen:
print("no cells match", file=sys.stderr)
return 2
by_driver: dict[str, list[dict[str, str]]] = defaultdict(list)
for row in chosen:
by_driver[row["driver"]].append(row)
fresh: list[dict[str, str]] = []
failures = 0
for driver, rows in by_driver.items():
buckets: dict[tuple[str, str, str], list[dict[str, str]]] = defaultdict(list)
for row in rows:
buckets[(row["family"], row["slice"], row.get("tier") or "std")].append(row)
for (family, slice_name, tier), group in buckets.items():
repeat, warmup = timing_for(family, slice_name, tier)
if args.repeat is not None:
repeat = args.repeat
if args.warmup is not None:
warmup = args.warmup
print(f"# {driver} {family}/{slice_name} tier={tier} "
f"cases={len(group)} repeat={repeat} warmup={warmup}",
file=sys.stderr)
rc, parsed, err = invoke_group(args.bin_dir, driver, group, repeat, warmup)
if err.strip():
print(err, file=sys.stderr, end="" if err.endswith("\n") else "\n")
got = {(row["case"], row["role"]) for row in parsed}
expected = {row["case"] for row in group}
have_cases = {case for case, _role in got}
missing = sorted(expected - have_cases)
if missing:
print(f"# missing results: {', '.join(missing)}", file=sys.stderr)
failures += len(missing)
if rc != 0:
failures += 1
fresh.extend(parsed)
label_dir = out_root / args.label
merged = merge_rows(read_tsv(label_dir / "long.tsv"), fresh)
write_tsv(label_dir / "long.tsv", LONG_FIELDS, merged)
write_tsv(label_dir / "matrix.tsv", MATRIX_FIELDS, matrix_view(merged))
print(f"# wrote {label_dir / 'matrix.tsv'} ({len(merged)} rows)", file=sys.stderr)
return 1 if failures else 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,907 @@
/// @file test/profile/eval_profile.cpp
/// @brief Workload for sequence and interval evaluation.
///
/// Both party keys are evaluated inside each sample. Key generation is its
/// own case, not part of the eval samples. `out_bytes` is the output buffer
/// volume of one sample (both parties). `per_item_ns` divides by the point
/// count, not by the number of parties.
///
/// profile_eval --list
/// profile_eval --case interval_u32_L4096 --repeat 50 --warmup 3
///
/// Profile-guided build, separate from COVERAGE:
/// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release
/// cmake --build build-pgo --target profile_eval
/// build-pgo/bin/profile_eval --repeat 40 --warmup 2
/// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release
/// cmake --build build-pgo --target profile_eval
#include "harness.hpp"
#include "dpf.hpp"
#include <cstdint>
#include <initializer_list>
#include <memory>
#include <string>
#include <utility>
#include <vector>
#include <algorithm>
#include <cstring>
namespace
{
using profile::sample;
using profile::touch_buf;
using profile::touch_word;
using profile::work;
template <typename Input>
std::shared_ptr<std::vector<Input>> clustered(Input start, std::size_t n)
{
auto pts = std::make_shared<std::vector<Input>>(n);
for (std::size_t i = 0; i < n; ++i)
(*pts)[i] = static_cast<Input>(start + static_cast<Input>(i));
return pts;
}
template <typename Input>
std::shared_ptr<std::vector<Input>> strided(Input start, Input step, std::size_t n)
{
auto pts = std::make_shared<std::vector<Input>>(n);
for (std::size_t i = 0; i < n; ++i)
(*pts)[i] = static_cast<Input>(start + step * static_cast<Input>(i));
return pts;
}
template <typename Keys>
sample hash_roots(const Keys & keys)
{
std::uint64_t w0 = 0;
std::uint64_t w1 = 0;
const auto r0 = std::get<0>(keys).root();
const auto r1 = std::get<1>(keys).root();
const std::size_t n0 = sizeof(r0) < sizeof(w0) ? sizeof(r0) : sizeof(w0);
const std::size_t n1 = sizeof(r1) < sizeof(w1) ? sizeof(r1) : sizeof(w1);
std::memcpy(&w0, &r0, n0);
std::memcpy(&w1, &r1, n1);
return touch_word(w0 ^ w1, sizeof(r0) + sizeof(r1));
}
template <typename Keys, typename Input>
sample interval_packed(const Keys & keys, Input from, Input to, unsigned lane_bits)
{
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(keys), from, to);
auto b = dpf::eval_interval(std::get<1>(keys), from, to);
auto fold = [lane_bits](const auto & buf) {
std::uint64_t w = 0;
const std::size_t bytes = (buf.size() * lane_bits + 7u) / 8u;
if (bytes != 0 && buf.data() != nullptr)
{
const std::size_t n = bytes < sizeof(w) ? bytes : sizeof(w);
std::memcpy(&w, buf.data(), n);
}
return touch_word(w, bytes);
};
return fold(a.first) + fold(b.first);
});
}
template <typename Keys, typename Input>
sample interval_both(const Keys & keys, Input from, Input to)
{
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(keys), from, to);
auto b = dpf::eval_interval(std::get<1>(keys), from, to);
return touch_buf(a.first) + touch_buf(b.first);
});
}
template <typename Keys, typename Input>
sample interval_reuse(const Keys & keys, Input from, Input to,
decltype(dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to)) & buf0,
decltype(dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to)) & buf1)
{
return profile::with_prg([&] {
auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0);
auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1);
(void)i0;
(void)i1;
return touch_buf(buf0) + touch_buf(buf1);
});
}
template <typename Keys, typename Input>
sample interval_prove(const Keys & keys, Input from, Input to)
{
auto buf0 = dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to);
auto buf1 = dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to);
dpf::proof_token p0{};
dpf::proof_token p1{};
auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0, dpf::prove(p0));
auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1, dpf::prove(p1));
(void)i0;
(void)i1;
std::uint64_t extra = 0;
std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0));
return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0);
}
template <typename Keys, typename Pts>
sample sequence_both(const Keys & keys, const Pts & pts, bool output_only)
{
return profile::with_prg([&] {
sample s;
if (output_only)
{
auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(),
dpf::return_output_only_tag_{});
auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(),
dpf::return_output_only_tag_{});
s = touch_buf(a.first) + touch_buf(b.first);
}
else
{
auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end());
auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end());
s = touch_buf(a.first) + touch_buf(b.first);
}
return s;
});
}
template <typename Keys, typename Pts>
sample sequence_breadth(const Keys & keys, const Pts & pts)
{
return profile::with_prg([&] {
auto a = dpf::eval_sequence_breadth_first(std::get<0>(keys), pts.begin(), pts.end());
auto b = dpf::eval_sequence_breadth_first(std::get<1>(keys), pts.begin(), pts.end());
return touch_buf(a.first) + touch_buf(b.first);
});
}
template <typename Keys, typename Recipe0, typename Recipe1>
sample sequence_recipe(const Keys & keys, const Recipe0 & r0, const Recipe1 & r1)
{
return profile::with_prg([&] {
auto a = dpf::eval_sequence(std::get<0>(keys), r0);
auto b = dpf::eval_sequence(std::get<1>(keys), r1);
return touch_buf(a.first) + touch_buf(b.first);
});
}
template <typename Keys, typename Pts>
sample sequence_prove(const Keys & keys, const Pts & pts)
{
auto buf0 = dpf::make_output_buffer_for_subsequence(std::get<0>(keys),
pts.begin(), pts.end(), dpf::return_output_only_tag_{});
auto buf1 = dpf::make_output_buffer_for_subsequence(std::get<1>(keys),
pts.begin(), pts.end(), dpf::return_output_only_tag_{});
dpf::proof_token p0{};
dpf::proof_token p1{};
auto i0 = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(), buf0,
dpf::prove(p0), dpf::return_output_only_tag_{});
auto i1 = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(), buf1,
dpf::prove(p1), dpf::return_output_only_tag_{});
(void)i0;
(void)i1;
std::uint64_t extra = 0;
std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0));
return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0);
}
template <typename Input, typename... Extra>
auto make_keys(Input alpha, Extra && ...extra)
{
auto keys = dpf::make_dpf(alpha, std::uint64_t{0x9e3779b97f4a7c15ull},
std::forward<Extra>(extra)...);
using keys_t = std::decay_t<decltype(keys)>;
return std::make_shared<keys_t>(std::move(keys));
}
template <typename Input, typename Output, typename... Extra>
auto make_out_keys(Input alpha, Output beta, Extra && ...extra)
{
auto keys = dpf::make_dpf(alpha, beta, std::forward<Extra>(extra)...);
using keys_t = std::decay_t<decltype(keys)>;
return std::make_shared<keys_t>(std::move(keys));
}
const char kUsage[] =
"profile_eval [--list] [--family F] [--slice S] [--case NAME] [--tier std]\n"
" [--repeat N=20] [--warmup W=2]\n"
"Slices: keygen, interval, interval-shape, sequence, sequence-algo,\n"
" memoizer, bits, gf, inner-product, helpers, defer.\n"
"Times sequence and interval evaluation on both party keys.\n";
template <typename Input, typename KeysPtr>
void add_interval_at(std::vector<work> & works, const char * slice,
const std::string & name, Input from, Input to, const KeysPtr & keys)
{
const auto items = static_cast<std::uint64_t>(to) - static_cast<std::uint64_t>(from) + 1;
works.push_back(profile::make_work("eval", slice, name, items, [keys, from, to] {
return interval_both(*keys, from, to);
}));
}
} // namespace
int main(int argc, char ** argv)
{
const auto opt = profile::parse_args(argc, argv, 20, 2, kUsage);
using u8 = std::uint8_t;
using u16 = std::uint16_t;
using u32 = std::uint32_t;
const auto k8 = make_keys(u8{40});
const auto k16 = make_keys(u16{1512});
const auto k16v = make_keys(u16{1512}, dpf::verifiable{});
const auto k32 = make_keys(u32{1002048u});
const auto k32v = make_keys(u32{1002048u}, dpf::verifiable{});
std::vector<work> works;
works.push_back(profile::make_work("eval", "keygen", "keygen_u8", 1, [] {
return hash_roots(*make_keys(u8{40}));
}));
works.push_back(profile::make_work("eval", "keygen", "keygen_u16", 1, [] {
return hash_roots(*make_keys(u16{1512}));
}));
works.push_back(profile::make_work("eval", "keygen", "keygen_u32", 1, [] {
return hash_roots(*make_keys(u32{0x01000000u}));
}));
works.push_back(profile::make_work("eval", "keygen", "keygen_u32_verifiable", 1, [] {
return hash_roots(*make_keys(u32{0x01000000u}, dpf::verifiable{}));
}));
const auto add_gf = [&](const char * name, auto beta) {
using out = std::decay_t<decltype(beta)>;
works.push_back(profile::make_work("eval", "gf",
std::string("keygen_") + name, 1, [beta] {
return hash_roots(*make_out_keys(u8{40}, beta));
}));
works.push_back(profile::make_work("eval", "gf",
std::string("keygen_") + name + "_verifiable", 1, [beta] {
return hash_roots(*make_out_keys(u8{40}, beta, dpf::verifiable{}));
}));
const auto keys = make_out_keys(u8{40}, beta);
if constexpr (dpf::utils::is_packed_subbyte_v<out>)
{
constexpr unsigned bits = dpf::utils::packed_lane_bits_v<out>;
works.push_back(profile::make_work("eval", "gf",
std::string("interval_") + name + "_u8_L256", 256,
[keys, bits] {
return interval_packed(*keys, u8{0}, u8{255}, bits);
}));
}
else
{
add_interval_at(works, "gf", std::string("interval_") + name + "_u8_L256",
u8{0}, u8{255}, keys);
}
};
add_gf("gf2", dpf::gf2{1});
add_gf("gf22", dpf::gf22{3});
add_gf("gf24", dpf::gf24{0xa});
add_gf("gf28", dpf::gf28{0x1b});
add_gf("gf216", dpf::gf216{0x2d});
add_gf("gf232", dpf::gf232{0x90200001u});
add_gf("gf264", dpf::gf264{0x11});
const auto add_lengths = [&](auto from0, auto keys, const char * width,
std::initializer_list<std::uint64_t> lengths) {
using input = decltype(from0);
for (const std::uint64_t n : lengths)
{
const input from = from0;
const input to = static_cast<input>(from + static_cast<input>(n - 1));
add_interval_at(works, "interval",
std::string("interval_") + width + "_L" + std::to_string(n),
from, to, keys);
}
};
add_lengths(u8{0}, k8, "u8", {1, 16, 64, 256});
add_lengths(u16{1000}, k16, "u16", {1, 16, 256, 1024, 4096});
add_lengths(u32{1000000}, k32, "u32", {1, 16, 64, 256, 1024, 4096, 16384});
add_interval_at(works, "interval-shape", "interval_u32_L256_unaligned",
u32{1000003}, u32{1000258}, k32);
add_interval_at(works, "interval-shape", "interval_u32_L4096_unaligned",
u32{1000003}, u32{1004098}, k32);
const auto add_reuse = [&](u32 from, u32 to, const char * name) {
using buf0_t = std::decay_t<decltype(dpf::make_output_buffer_for_interval(
std::get<0>(*k32), from, to))>;
using buf1_t = std::decay_t<decltype(dpf::make_output_buffer_for_interval(
std::get<1>(*k32), from, to))>;
auto buf0 = std::make_shared<buf0_t>(
dpf::make_output_buffer_for_interval(std::get<0>(*k32), from, to));
auto buf1 = std::make_shared<buf1_t>(
dpf::make_output_buffer_for_interval(std::get<1>(*k32), from, to));
const auto items = static_cast<std::uint64_t>(to - from + 1);
works.push_back(profile::make_work("eval", "interval-shape", name, items,
[k32, buf0, buf1, from, to] {
return interval_reuse(*k32, from, to, *buf0, *buf1);
}));
};
add_reuse(1000000, 1000255, "interval_u32_L256_reuse");
add_reuse(1000000, 1004095, "interval_u32_L4096_reuse");
works.push_back(profile::make_work("eval", "interval-shape",
"interval_u16_L1024_verifiable", 1024, [k16v] {
return interval_prove(*k16v, u16{1000}, u16{2023});
}));
works.push_back(profile::make_work("eval", "interval-shape",
"interval_u32_L256_verifiable", 256, [k32v] {
return interval_prove(*k32v, u32{1000000}, u32{1000255});
}));
const auto add_seq = [&](auto keys, const char * width, const char * shape,
auto pts) {
const auto n = static_cast<std::uint64_t>(pts->size());
const std::string name = std::string("sequence_") + width + "_" + shape
+ "_" + std::to_string(n);
works.push_back(profile::make_work("eval", "sequence", name, n,
[keys, pts] {
return sequence_both(*keys, *pts, false);
}));
};
const std::size_t seq32[] = {8, 32, 64, 128, 256, 512, 1024, 2048};
for (const std::size_t n : seq32)
{
add_seq(k32, "u32", "cluster", clustered<u32>(1u << 20, n));
const u32 step = n <= 64 ? u32{1u << 20} : u32{1u << 12};
add_seq(k32, "u32", "stride", strided<u32>(16u, step, n));
}
const std::size_t seq16[] = {8, 64, 256, 1024};
for (const std::size_t n : seq16)
add_seq(k16, "u16", "cluster", clustered<u16>(1000, n));
const std::size_t seq8[] = {8, 32, 64};
for (const std::size_t n : seq8)
{
add_seq(k8, "u8", "cluster", clustered<u8>(0, n));
add_seq(k8, "u8", "stride", strided<u8>(0, 3, n));
}
const auto c256 = clustered<u32>(1u << 20, 256);
const auto s64 = strided<u32>(16u, 1u << 20, 64);
const auto c16s = clustered<u16>(1000, 64);
const auto c32s = clustered<u32>(1u << 20, 64);
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_cluster_256_output_only", 256, [k32, c256] {
return sequence_both(*k32, *c256, true);
}));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_cluster_256_breadth", 256, [k32, c256] {
return sequence_breadth(*k32, *c256);
}));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_stride_64_breadth", 64, [k32, s64] {
return sequence_breadth(*k32, *s64);
}));
using recipe0_t = std::decay_t<decltype(dpf::make_sequence_recipe(
std::get<0>(*k32), c256->begin(), c256->end()))>;
using recipe1_t = std::decay_t<decltype(dpf::make_sequence_recipe(
std::get<1>(*k32), c256->begin(), c256->end()))>;
auto recipe0 = std::make_shared<recipe0_t>(dpf::make_sequence_recipe(
std::get<0>(*k32), c256->begin(), c256->end()));
auto recipe1 = std::make_shared<recipe1_t>(dpf::make_sequence_recipe(
std::get<1>(*k32), c256->begin(), c256->end()));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_cluster_256_recipe", 256, [k32, recipe0, recipe1] {
return sequence_recipe(*k32, *recipe0, *recipe1);
}));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u16_cluster_64_verifiable", 64, [k16v, c16s] {
return sequence_prove(*k16v, *c16s);
}));
works.push_back(profile::make_work("eval", "sequence-algo",
"sequence_u32_cluster_64_verifiable", 64, [k32v, c32s] {
return sequence_prove(*k32v, *c32s);
}));
// --- memoizer: built-in reuse vs hand-rolled pointwise / fresh memo ---
{
using u32 = std::uint32_t;
const u32 from = 1000000u;
const u32 to = 1000255u; // L256
const auto items = static_cast<std::uint64_t>(to - from + 1);
const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32));
using dpf_t = std::decay_t<decltype(std::get<0>(*k32))>;
using node_t = typename dpf_t::interior_node;
const auto out_elem = sizeof(std::uint64_t);
const auto logical = items * out_elem * 2;
// Match `basic_interval_memoizer` capacity (leaf nodes, not output slots).
const auto leaf_nodes =
dpf::utils::get_leafnodes_in_output_interval<dpf_t>(from, to);
const auto slots = leaf_nodes == 0 ? std::size_t{1} : leaf_nodes;
const auto pivot = std::max((slots >> 1) + (slots & 1) - 1,
(slots + 6) >> 2);
const auto memo_nodes = pivot + ((slots + 2) >> 1);
const auto memo_bytes = 2ull * memo_nodes * sizeof(node_t);
auto memo0 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to))>>(
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to));
auto memo1 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to))>>(
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to));
works.push_back(profile::make_work("eval", "memoizer",
"memo_interval_reuse", items, [k32, memo0, memo1, from, to, prep, memo_bytes, logical] {
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0);
auto b = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1);
auto s = touch_buf(a.first) + touch_buf(b.first);
// Warm reuse: second pass on the same memoizers.
auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0);
auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1);
s = s + touch_buf(a2.first) + touch_buf(b2.first);
return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2);
});
}));
works.push_back(profile::make_work("eval", "memoizer",
"memo_interval_fresh", items, [k32, from, to, prep, logical, memo_bytes] {
return profile::with_prg([&] {
auto m0 = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to);
auto m1 = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to);
auto a = dpf::eval_interval(std::get<0>(*k32), from, to, m0);
auto b = dpf::eval_interval(std::get<1>(*k32), from, to, m1);
auto s = touch_buf(a.first) + touch_buf(b.first);
auto m0b = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to);
auto m1b = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to);
auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, m0b);
auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, m1b);
s = s + touch_buf(a2.first) + touch_buf(b2.first);
return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2);
});
}));
const auto pts = clustered<u32>(1u << 20, 64);
auto path0 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_path_memoizer(std::get<0>(*k32)))>>(
dpf::make_basic_path_memoizer(std::get<0>(*k32)));
auto path1 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_path_memoizer(std::get<1>(*k32)))>>(
dpf::make_basic_path_memoizer(std::get<1>(*k32)));
works.push_back(profile::make_work("eval", "memoizer",
"memo_path_reuse", 64, [k32, pts, path0, path1, prep] {
return profile::with_prg([&] {
std::uint64_t h = 0;
for (auto x : *pts)
{
h ^= static_cast<std::uint64_t>(
(*dpf::eval_point(std::get<0>(*k32), x, *path0)).raw());
h ^= static_cast<std::uint64_t>(
(*dpf::eval_point(std::get<1>(*k32), x, *path1)).raw());
}
auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2);
return profile::with_costs(s, prep,
2 * sizeof(node_t) * 32, 64 * sizeof(std::uint64_t) * 2);
});
}));
works.push_back(profile::make_work("eval", "memoizer",
"memo_path_pointwise", 64, [k32, pts, prep] {
return profile::with_prg([&] {
std::uint64_t h = 0;
for (auto x : *pts)
{
h ^= static_cast<std::uint64_t>(
(*dpf::eval_point(std::get<0>(*k32), x)).raw());
h ^= static_cast<std::uint64_t>(
(*dpf::eval_point(std::get<1>(*k32), x)).raw());
}
auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2);
return profile::with_costs(s, prep, 0,
64 * sizeof(std::uint64_t) * 2);
});
}));
using recipe0_t = std::decay_t<decltype(dpf::make_sequence_recipe(
std::get<0>(*k32), pts->begin(), pts->end()))>;
using recipe1_t = std::decay_t<decltype(dpf::make_sequence_recipe(
std::get<1>(*k32), pts->begin(), pts->end()))>;
auto rec0 = std::make_shared<recipe0_t>(dpf::make_sequence_recipe(
std::get<0>(*k32), pts->begin(), pts->end()));
auto rec1 = std::make_shared<recipe1_t>(dpf::make_sequence_recipe(
std::get<1>(*k32), pts->begin(), pts->end()));
auto smemo0 = std::make_shared<std::decay_t<decltype(
dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0))>>(
dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0));
auto smemo1 = std::make_shared<std::decay_t<decltype(
dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1))>>(
dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1));
works.push_back(profile::make_work("eval", "memoizer",
"memo_recipe_reuse", 64, [k32, rec0, rec1, smemo0, smemo1, prep] {
return profile::with_prg([&] {
auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0);
auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1);
auto s = touch_buf(a.first) + touch_buf(b.first);
auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0);
auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1);
s = s + touch_buf(a2.first) + touch_buf(b2.first);
return profile::with_costs(s, prep, s.out_bytes,
64 * sizeof(std::uint64_t) * 4);
});
}));
works.push_back(profile::make_work("eval", "memoizer",
"memo_recipe_fresh", 64, [k32, rec0, rec1, prep] {
return profile::with_prg([&] {
auto m0 = dpf::make_inplace_reversing_sequence_memoizer(
std::get<0>(*k32), *rec0);
auto m1 = dpf::make_inplace_reversing_sequence_memoizer(
std::get<1>(*k32), *rec1);
auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0);
auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1);
auto s = touch_buf(a.first) + touch_buf(b.first);
auto m0b = dpf::make_inplace_reversing_sequence_memoizer(
std::get<0>(*k32), *rec0);
auto m1b = dpf::make_inplace_reversing_sequence_memoizer(
std::get<1>(*k32), *rec1);
auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0b);
auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1b);
s = s + touch_buf(a2.first) + touch_buf(b2.first);
return profile::with_costs(s, prep, s.out_bytes,
64 * sizeof(std::uint64_t) * 4);
});
}));
}
// --- bits: built-in iterators vs hand-rolled scans (u8 domain) ---
{
using input_type = std::uint8_t;
using output_type = dpf::bit;
const input_type alpha = 40;
auto bit_keys = std::make_shared<std::decay_t<decltype(
dpf::make_dpf(alpha, output_type::one))>>(
dpf::make_dpf(alpha, output_type::one));
using dpf_type = std::decay_t<decltype(std::get<0>(*bit_keys))>;
auto memo0 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_full_memoizer<dpf_type>())>>(
dpf::make_basic_full_memoizer<dpf_type>());
auto memo1 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_full_memoizer<dpf_type>())>>(
dpf::make_basic_full_memoizer<dpf_type>());
auto full0 = dpf::eval_full(std::get<0>(*bit_keys), *memo0);
auto full1 = dpf::eval_full(std::get<1>(*bit_keys), *memo1);
auto buf0 = std::make_shared<std::decay_t<decltype(full0.first)>>(std::move(full0.first));
auto buf1 = std::make_shared<std::decay_t<decltype(full1.first)>>(std::move(full1.first));
auto iter0 = std::make_shared<std::decay_t<decltype(full0.second)>>(std::move(full0.second));
auto iter1 = std::make_shared<std::decay_t<decltype(full1.second)>>(std::move(full1.second));
const auto leaf_nodes = static_cast<std::uint64_t>(1) << dpf_type::depth;
const auto bit_items = static_cast<std::uint64_t>(buf0->size());
const auto prep = sizeof(std::get<0>(*bit_keys)) + sizeof(std::get<1>(*bit_keys));
works.push_back(profile::make_work("eval", "bits",
"bits_advice_builtin", leaf_nodes, [memo0, memo1, prep, leaf_nodes] {
std::uint64_t h = 0;
std::uint64_t n = 0;
for (auto b : dpf::advice_bits_of(*memo0))
{
h = (h << 1) ^ (b ? 1u : 0u);
++n;
}
for (auto b : dpf::advice_bits_of(*memo1))
h ^= (b ? 1u : 0u);
auto s = touch_word(h ^ n, n);
return profile::with_costs(s, prep, 0, leaf_nodes);
}));
works.push_back(profile::make_work("eval", "bits",
"bits_advice_handroll", leaf_nodes, [memo0, memo1, prep, leaf_nodes] {
std::uint64_t h = 0;
std::uint64_t n = 0;
const auto * p0 = memo0->begin();
const auto * e0 = memo0->end();
for (; p0 != e0; ++p0)
{
const auto * bytes = reinterpret_cast<const char *>(p0);
h = (h << 1) ^ (bytes[0] & 1u);
++n;
}
const auto * p1 = memo1->begin();
const auto * e1 = memo1->end();
for (; p1 != e1; ++p1)
{
const auto * bytes = reinterpret_cast<const char *>(p1);
h ^= (bytes[0] & 1u);
}
auto s = touch_word(h ^ n, n);
return profile::with_costs(s, prep, 0, leaf_nodes);
}));
works.push_back(profile::make_work("eval", "bits",
"bits_setbit_builtin", bit_items, [iter0, iter1, prep] {
std::uint64_t h = 0;
std::uint64_t n = 0;
for (auto i : dpf::indices_set_in(*iter0))
{
h ^= static_cast<std::uint64_t>(i) + 0x9e3779b97f4a7c15ull;
++n;
}
for (auto i : dpf::indices_set_in(*iter1))
h ^= static_cast<std::uint64_t>(i);
auto s = touch_word(h ^ n, n * sizeof(std::size_t));
return profile::with_costs(s, prep, 0, n * sizeof(std::size_t));
}));
works.push_back(profile::make_work("eval", "bits",
"bits_setbit_handroll", bit_items, [buf0, buf1, prep] {
std::uint64_t h = 0;
std::uint64_t n = 0;
const auto scan = [&](const auto & buf) {
for (std::size_t i = 0; i < buf.size(); ++i)
{
if (static_cast<bool>(buf[i]))
{
h ^= i + 0x9e3779b97f4a7c15ull;
++n;
}
}
};
scan(*buf0);
scan(*buf1);
auto s = touch_word(h ^ n, n * sizeof(std::size_t));
return profile::with_costs(s, prep, 0, n * sizeof(std::size_t));
}));
works.push_back(profile::make_work("eval", "bits",
"bits_parallel_builtin", bit_items, [buf0, buf1, prep, bit_items] {
std::uint64_t h = 0;
std::uint64_t n = 0;
for (auto word : dpf::batch_of(*buf0, *buf1))
{
h ^= static_cast<std::uint64_t>(word[0])
^ static_cast<std::uint64_t>(word[1]);
++n;
}
auto s = touch_word(h ^ n, n * sizeof(std::uint64_t));
return profile::with_costs(s, prep, 0, bit_items);
}));
works.push_back(profile::make_work("eval", "bits",
"bits_parallel_handroll", bit_items, [buf0, buf1, prep, bit_items] {
std::uint64_t h = 0;
const std::size_t n = std::min(buf0->size(), buf1->size());
for (std::size_t i = 0; i < n; ++i)
h ^= (static_cast<bool>((*buf0)[i]) ? 1ull : 0ull)
^ (static_cast<bool>((*buf1)[i]) ? 2ull : 0ull);
auto s = touch_word(h ^ n, n);
return profile::with_costs(s, prep, 0, bit_items);
}));
}
// --- inner-product: built-in vs hand-rolled interval + dot ---
{
using u32 = std::uint32_t;
const u32 from = 1000000u;
const u32 to = 1000255u;
const auto items = static_cast<std::uint64_t>(to - from + 1);
auto weights = std::make_shared<std::vector<std::uint64_t>>(items);
for (std::size_t i = 0; i < items; ++i)
(*weights)[i] = 0x9e3779b97f4a7c15ull * (i + 1);
const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32));
auto ip_memo0 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to))>>(
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to));
auto ip_memo1 = std::make_shared<std::decay_t<decltype(
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to))>>(
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to));
works.push_back(profile::make_work("eval", "inner-product",
"ip_interval_builtin", items,
[k32, weights, from, to, prep, items, ip_memo0, ip_memo1] {
return profile::with_prg([&] {
auto a = dpf::eval_inner_product(std::get<0>(*k32), from, to,
*weights, *ip_memo0);
auto b = dpf::eval_inner_product(std::get<1>(*k32), from, to,
*weights, *ip_memo1);
std::uint64_t ha = 0;
std::uint64_t hb = 0;
std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a));
std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b));
auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b));
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
});
}));
works.push_back(profile::make_work("eval", "inner-product",
"ip_interval_handroll", items, [k32, weights, from, to, prep, items] {
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(*k32), from, to);
auto b = dpf::eval_interval(std::get<1>(*k32), from, to);
std::uint64_t dot0 = 0;
std::uint64_t dot1 = 0;
const std::size_t n = std::min(items, a.first.size());
for (std::size_t i = 0; i < n; ++i)
{
dot0 += static_cast<std::uint64_t>(a.first[i].raw()) * (*weights)[i];
dot1 += static_cast<std::uint64_t>(b.first[i].raw()) * (*weights)[i];
}
auto s = touch_word(dot0 ^ dot1, a.first.size() * sizeof(a.first[0])
+ b.first.size() * sizeof(b.first[0]));
return profile::with_costs(s, prep, s.out_bytes,
items * sizeof(std::uint64_t) * 2);
});
}));
works.push_back(profile::make_work("eval", "inner-product",
"ip_interval_paired", items,
[k32, weights, from, to, prep, items] {
return profile::with_prg([&] {
auto a = dpf::eval_inner_product(dpf::paired,
std::get<0>(*k32), from, to, *weights);
auto b = dpf::eval_inner_product(dpf::paired,
std::get<1>(*k32), from, to, *weights);
std::uint64_t ha = 0;
std::uint64_t hb = 0;
std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a));
std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b));
auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b));
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
});
}));
works.push_back(profile::make_work("eval", "inner-product",
"ip_interval_columns", items,
[k32, weights, from, to, prep, items] {
return profile::with_prg([&] {
auto a = dpf::eval_inner_product(dpf::columns,
std::get<0>(*k32), from, to, std::tie(*weights));
auto b = dpf::eval_inner_product(dpf::columns,
std::get<1>(*k32), from, to, std::tie(*weights));
std::uint64_t ha = 0;
std::uint64_t hb = 0;
const auto & a0 = std::get<0>(a);
const auto & b0 = std::get<0>(b);
std::memcpy(&ha, &a0,
sizeof(ha) < sizeof(a0) ? sizeof(ha) : sizeof(a0));
std::memcpy(&hb, &b0,
sizeof(hb) < sizeof(b0) ? sizeof(hb) : sizeof(b0));
auto s = touch_word(ha ^ hb, sizeof(a0) + sizeof(b0));
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
});
}));
}
// --- helpers: point vs interval vs full (representative widths) ---
{
using u16 = std::uint16_t;
const auto prep16 = sizeof(std::get<0>(*k16)) + sizeof(std::get<1>(*k16));
works.push_back(profile::make_work("eval", "helpers",
"helper_point_u16", 1, [k16, prep16] {
return profile::with_prg([&] {
auto a = *dpf::eval_point(std::get<0>(*k16), u16{1512});
auto b = *dpf::eval_point(std::get<1>(*k16), u16{1512});
auto s = touch_word(
static_cast<std::uint64_t>(a.raw()) ^ static_cast<std::uint64_t>(b.raw()),
sizeof(a) + sizeof(b));
return profile::with_costs(s, prep16, 0, sizeof(a) + sizeof(b));
});
}));
works.push_back(profile::make_work("eval", "helpers",
"helper_interval_u16_L256", 256, [k16, prep16] {
return profile::with_prg([&] {
auto a = dpf::eval_interval(std::get<0>(*k16), u16{1000}, u16{1255});
auto b = dpf::eval_interval(std::get<1>(*k16), u16{1000}, u16{1255});
auto s = touch_buf(a.first) + touch_buf(b.first);
return profile::with_costs(s, prep16, s.out_bytes,
256 * sizeof(std::uint64_t) * 2);
});
}));
works.push_back(profile::make_work("eval", "helpers",
"helper_full_u8", 256, [k8] {
const auto prep = sizeof(std::get<0>(*k8)) + sizeof(std::get<1>(*k8));
return profile::with_prg([&] {
auto a = dpf::eval_full(std::get<0>(*k8));
auto b = dpf::eval_full(std::get<1>(*k8));
auto s = touch_buf(a.first) + touch_buf(b.first);
return profile::with_costs(s, prep, s.out_bytes, s.out_bytes);
});
}));
}
// --- defer: pre-assign full-domain expand vs rotate-after-assign ---
{
using u8 = std::uint8_t;
using out_t = std::uint64_t;
const u8 alpha = 40;
const out_t beta = 0x9e3779b97f4a7c15ull;
works.push_back(profile::make_work("eval", "defer",
"defer_full_u8_expand", 256, [beta] {
return profile::with_prg([&] {
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys));
auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys));
auto d0 = dpf::defer_eval_full(std::get<0>(keys), b0);
auto d1 = dpf::defer_eval_full(std::get<1>(keys), b1);
(void)d0;
(void)d1;
return touch_buf(b0) + touch_buf(b1);
});
}));
works.push_back(profile::make_work("eval", "defer",
"defer_interval_u8_L64_expand", 64, [beta] {
return profile::with_prg([&] {
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys));
auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys));
auto d0 = dpf::defer_eval_interval(std::get<0>(keys),
u8{16}, u8{79}, b0);
auto d1 = dpf::defer_eval_interval(std::get<1>(keys),
u8{16}, u8{79}, b1);
(void)d0;
(void)d1;
return touch_buf(b0) + touch_buf(b1);
});
}));
// Expand once against stable key storage, assign, then time .get().
{
using keys_t = std::decay_t<decltype(
dpf::make_dpf(dpf::wildcard_value<u8>{}, beta))>;
auto keys = std::make_shared<keys_t>(
dpf::make_dpf(dpf::wildcard_value<u8>{}, beta));
using buf0_t = std::decay_t<decltype(
dpf::make_output_buffer_for_full(std::get<0>(*keys)))>;
using buf1_t = std::decay_t<decltype(
dpf::make_output_buffer_for_full(std::get<1>(*keys)))>;
auto b0 = std::make_shared<buf0_t>(
dpf::make_output_buffer_for_full(std::get<0>(*keys)));
auto b1 = std::make_shared<buf1_t>(
dpf::make_output_buffer_for_full(std::get<1>(*keys)));
using def0_t = std::decay_t<decltype(
dpf::defer_eval_full(std::get<0>(*keys), *b0))>;
using def1_t = std::decay_t<decltype(
dpf::defer_eval_full(std::get<1>(*keys), *b1))>;
auto def0 = std::make_shared<def0_t>(
dpf::defer_eval_full(std::get<0>(*keys), *b0));
auto def1 = std::make_shared<def1_t>(
dpf::defer_eval_full(std::get<1>(*keys), *b1));
{
auto & k0 = std::get<0>(*keys);
auto & k1 = std::get<1>(*keys);
const u8 a0 = 0x12;
const u8 a1 = static_cast<u8>(alpha - a0);
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
k0.offset_x.reconstruct(sh1);
k1.offset_x.reconstruct(sh0);
}
works.push_back(profile::make_work("eval", "defer",
"defer_full_u8_get", 256, [keys, def0, def1, b0, b1] {
(void)keys;
auto v0 = def0->get();
auto v1 = def1->get();
std::uint64_t sink = 0;
for (auto it = std::begin(v0); it != std::end(v0); ++it)
sink ^= static_cast<std::uint64_t>((*it).raw());
for (auto it = std::begin(v1); it != std::end(v1); ++it)
sink ^= static_cast<std::uint64_t>((*it).raw());
return touch_word(sink, b0->size() * sizeof((*b0)[0])
+ b1->size() * sizeof((*b1)[0]));
}));
}
works.push_back(profile::make_work("eval", "defer",
"eager_full_u8_after_assign", 256, [beta, alpha] {
return profile::with_prg([&] {
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
auto & k0 = std::get<0>(keys);
auto & k1 = std::get<1>(keys);
const u8 a0 = 0x12;
const u8 a1 = static_cast<u8>(alpha - a0);
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
k0.offset_x.reconstruct(sh1);
k1.offset_x.reconstruct(sh0);
auto a = dpf::eval_full(k0);
auto b = dpf::eval_full(k1);
return touch_buf(a.first) + touch_buf(b.first);
});
}));
}
return profile::run_works(opt, works);
}

View file

@ -0,0 +1,509 @@
/// @file test/profile/grotto_profile.cpp
/// @brief Workload for grotto evaluation: tables, offset walks, prefix parity.
///
/// Key generation is a separate case from the eval that consumes those keys.
/// Table cases probe the domain once, then time only inputs that evaluate.
/// `items` is the number of table evaluations inside one sample, or 1 for a
/// single offset / prefix walk (both parties).
///
/// profile_grotto --list
/// profile_grotto --case window_gelu_p16 --repeat 100
///
/// Profile-guided build, separate from COVERAGE:
/// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release
/// cmake --build build-pgo --target profile_grotto
/// build-pgo/bin/profile_grotto --repeat 40 --warmup 2
/// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release
/// cmake --build build-pgo --target profile_grotto
#include "harness.hpp"
#include "dpf.hpp"
#include "grotto/closed_form.hpp"
#include "grotto/exact_steps.hpp"
#include "grotto/offset_horner.hpp"
#include "grotto/offset_jet.hpp"
#include "grotto/offset_poly.hpp"
#include "grotto/offset_repr.hpp"
#include "grotto/offset_twist.hpp"
#include "grotto/prefix_parity.hpp"
#include "grotto/range_lut.hpp"
#include "grotto/window_lut.hpp"
#include <array>
#include <cstdint>
#include <functional>
#include <initializer_list>
#include <memory>
#include <string>
#include <vector>
namespace
{
using profile::sample;
using profile::touch_arr;
using profile::touch_vec;
using profile::touch_word;
using profile::work;
const char kUsage[] =
"profile_grotto [--list] [--family grotto] [--slice S] [--case NAME]\n"
" [--repeat N=30] [--warmup W=2]\n"
"Slices: window, principal, closed, reduced, exact, horner, poly, jet,\n"
" twist, repr, prefix, keygen.\n"
"Times grotto table evals and offset / prefix walks.\n";
inline work cell(const char * slice, std::string name, std::uint64_t items,
std::function<sample()> fn)
{
return profile::make_work("grotto", slice, std::move(name), items, std::move(fn));
}
template <typename Enum, typename Fn>
std::vector<std::int64_t> probe(Enum which, unsigned bits, Fn fn,
const std::vector<std::int64_t> & cand)
{
std::vector<std::int64_t> ok;
ok.reserve(cand.size());
for (const std::int64_t raw : cand)
{
try
{
(void)fn(which, bits, raw);
ok.push_back(raw);
}
catch (const std::exception &)
{
}
}
return ok;
}
std::vector<std::int64_t> around_one(unsigned bits)
{
const std::int64_t one = std::int64_t{1} << bits;
return {
-4 * one, -2 * one, -one, -one / 2, -1, 0, 1,
one / 20, one / 5, one / 3, one / 2, one, (3 * one) / 2, 2 * one, 4 * one
};
}
template <typename Fn>
void add_table(std::vector<work> & works, const char * slice, std::string name, Fn fn,
std::vector<std::int64_t> samples)
{
if (samples.empty())
{
std::cerr << name << " skipped: no in-domain sample\n";
return;
}
auto held = std::make_shared<std::vector<std::int64_t>>(std::move(samples));
const std::uint64_t items = static_cast<std::uint64_t>(held->size()) * 32u;
std::function<sample()> body = [held, fn]() -> sample {
std::uint64_t h = 0x84222325cbf29ce4ull;
for (int rep = 0; rep < 32; ++rep)
{
for (const std::int64_t raw : *held)
{
// Multiply-add, not a plain xor: an even number of identical
// xors cancels, and the compiler will delete the evals.
h = (h * 0x100000001b3ull) ^ static_cast<std::uint64_t>(fn(raw))
^ static_cast<std::uint64_t>(rep);
}
}
return touch_word(h, held->size() * sizeof(std::int64_t));
};
works.push_back(cell(slice, std::move(name), items, std::move(body)));
}
template <typename T>
std::vector<T> knots_of(std::size_t n, T gap)
{
std::vector<T> knots(n);
for (std::size_t i = 0; i < n; ++i)
knots[i] = static_cast<T>(1 + gap * static_cast<T>(i));
return knots;
}
template <std::size_t Degree>
std::vector<std::array<std::uint64_t, Degree + 1>> horner_rows(std::size_t n)
{
std::vector<std::array<std::uint64_t, Degree + 1>> rows(n);
for (std::size_t i = 0; i < n; ++i)
for (std::size_t m = 0; m <= Degree; ++m)
rows[i][m] = (i + 1) * (m + 3);
return rows;
}
std::vector<std::vector<std::uint64_t>> poly_rows(std::size_t n, std::size_t degree)
{
std::vector<std::vector<std::uint64_t>> rows(n, std::vector<std::uint64_t>(degree + 1));
for (std::size_t i = 0; i < n; ++i)
for (std::size_t m = 0; m <= degree; ++m)
rows[i][m] = (i + 1) * (m + 3);
return rows;
}
std::vector<std::uint64_t> coeff_col(std::size_t degree)
{
std::vector<std::uint64_t> c(degree + 1);
for (std::size_t m = 0; m <= degree; ++m)
c[m] = m + 5;
return c;
}
template <std::size_t N, typename T>
std::array<T, N> ends_of(T start, T step)
{
std::array<T, N> ends{};
for (std::size_t i = 0; i < N; ++i)
ends[i] = static_cast<T>(start + step * static_cast<T>(i));
return ends;
}
} // namespace
template <std::size_t Degree>
void add_horner_degree(std::vector<work> & works)
{
using in_t = std::uint32_t;
const in_t center = 50;
const in_t eta = 7;
works.push_back(cell("keygen", "horner_keygen_d" + std::to_string(Degree), 1, [] {
auto built = grotto::make_offset_horner_keys<in_t, Degree>(in_t{50});
return touch_word(built.wrap_share[0][0] ^ built.wrap_share[Degree][1],
sizeof(built.wrap_share));
}));
using mat_t = decltype(grotto::make_offset_horner_keys<in_t, Degree>(center));
auto mat = std::make_shared<mat_t>(grotto::make_offset_horner_keys<in_t, Degree>(center));
for (const std::size_t pieces : std::initializer_list<std::size_t>{1, 4, 16, 64})
{
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(pieces, 1000));
auto rows = std::make_shared<std::vector<std::array<std::uint64_t, Degree + 1>>>(
horner_rows<Degree>(pieces));
const auto name = "horner_eval_d" + std::to_string(Degree)
+ "_p" + std::to_string(pieces);
works.push_back(cell("horner", name, 1, [mat, knots, rows, eta] {
const auto a = grotto::offset_horner_eval<0, Degree>(*mat, *knots, *rows, eta);
const auto b = grotto::offset_horner_eval<1, Degree>(*mat, *knots, *rows, eta);
return touch_word(a ^ b, 16);
}));
}
}
template <typename Input, typename BitPtr, typename GtPtr, std::size_t N>
void add_prefix_n(std::vector<work> & works, const char * width, Input step,
const BitPtr & bits, const GtPtr & gts)
{
const auto ends = ends_of<N>(Input{1}, step);
const auto ntag = std::to_string(N);
works.push_back(cell("prefix", std::string("prefix_") + width + "_e" + ntag, N,
[bits, ends] {
const auto a = grotto::prefix_parities(std::get<0>(*bits), ends);
const auto b = grotto::prefix_parities(std::get<1>(*bits), ends);
return touch_arr(std::get<0>(a)) + touch_arr(std::get<0>(b));
}));
works.push_back(cell("prefix", std::string("signed_prefix_") + width + "_e" + ntag, N,
[gts, ends] {
const auto a = grotto::signed_prefix_parities(std::get<0>(*gts), ends);
const auto b = grotto::signed_prefix_parities(std::get<1>(*gts), ends);
return touch_arr(a) + touch_arr(b);
}));
}
template <typename Input>
void add_prefix_width(std::vector<work> & works, const char * width, Input step)
{
const auto bit_keys = dpf::make_dpf(Input{1000}, dpf::bit{1});
const auto gt_keys = dpf::make_dpf(Input{1000}, dpf::gt(std::uint64_t{1}));
using bit_ptr = std::shared_ptr<std::decay_t<decltype(bit_keys)>>;
using gt_ptr = std::shared_ptr<std::decay_t<decltype(gt_keys)>>;
auto bits = std::make_shared<std::decay_t<decltype(bit_keys)>>(bit_keys);
auto gts = std::make_shared<std::decay_t<decltype(gt_keys)>>(gt_keys);
add_prefix_n<Input, bit_ptr, gt_ptr, 1>(works, width, step, bits, gts);
add_prefix_n<Input, bit_ptr, gt_ptr, 8>(works, width, step, bits, gts);
add_prefix_n<Input, bit_ptr, gt_ptr, 32>(works, width, step, bits, gts);
add_prefix_n<Input, bit_ptr, gt_ptr, 128>(works, width, step, bits, gts);
}
int main(int argc, char ** argv)
{
const auto opt = profile::parse_args(argc, argv, 30, 2, kUsage);
std::vector<work> works;
const unsigned precisions[] = {8u, 16u, 24u, 32u};
const auto add_window = [&](const char * stem, grotto::window which, unsigned bits) {
auto samples = probe(which, bits,
[](grotto::window w, unsigned b, std::int64_t raw) {
return grotto::eval_window(w, b, raw);
}, around_one(bits));
add_table(works, "window",
std::string("window_") + stem + "_p" + std::to_string(bits),
[which, bits](std::int64_t raw) {
return grotto::eval_window(which, bits, raw);
}, std::move(samples));
};
const auto add_principal = [&](const char * stem, grotto::principal which, unsigned bits) {
auto samples = probe(which, bits,
[](grotto::principal w, unsigned b, std::int64_t raw) {
return grotto::eval_principal(w, b, raw);
}, around_one(bits));
add_table(works, "principal",
std::string("principal_") + stem + "_p" + std::to_string(bits),
[which, bits](std::int64_t raw) {
return grotto::eval_principal(which, bits, raw);
}, std::move(samples));
};
const auto add_closed = [&](const char * stem, grotto::closed which, unsigned bits) {
auto samples = probe(which, bits,
[](grotto::closed w, unsigned b, std::int64_t raw) {
return grotto::eval_closed(w, b, raw);
}, around_one(bits));
add_table(works, "closed",
std::string("closed_") + stem + "_p" + std::to_string(bits),
[which, bits](std::int64_t raw) {
return grotto::eval_closed(which, bits, raw);
}, std::move(samples));
};
const auto add_reduced = [&](const char * stem, grotto::reduced which, unsigned bits) {
auto samples = probe(which, bits,
[](grotto::reduced w, unsigned b, std::int64_t raw) {
return grotto::eval_reduced(w, b, raw);
}, around_one(bits));
add_table(works, "reduced",
std::string("reduced_") + stem + "_p" + std::to_string(bits),
[which, bits](std::int64_t raw) {
return grotto::eval_reduced(which, bits, raw);
}, std::move(samples));
};
const struct { const char * stem; grotto::window which; } windows[] = {
{"gelu", grotto::window::gelu},
{"sigmoid", grotto::window::sigmoid},
{"tanh", grotto::window::tanh},
{"erf", grotto::window::erf},
{"erfc", grotto::window::erfc},
{"silu", grotto::window::silu},
{"mish", grotto::window::mish},
{"softplus", grotto::window::softplus},
{"probit", grotto::window::probit},
{"smoothstep", grotto::window::smoothstep},
{"asin", grotto::window::asin},
{"acos", grotto::window::acos},
{"hardelish", grotto::window::hardelish},
{"lecun_tanh", grotto::window::lecun_tanh},
};
for (const auto & spec : windows)
for (const unsigned bits : precisions)
add_window(spec.stem, spec.which, bits);
const struct { const char * stem; grotto::principal which; } principals[] = {
{"ln", grotto::principal::ln},
{"exp", grotto::principal::exp},
{"sin", grotto::principal::sin},
{"sqrt", grotto::principal::sqrt},
{"sinh", grotto::principal::sinh},
{"inv", grotto::principal::inv},
};
for (const auto & spec : principals)
for (const unsigned bits : precisions)
add_principal(spec.stem, spec.which, bits);
const struct { const char * stem; grotto::closed which; } closeds[] = {
{"atan", grotto::closed::atan},
{"cbrt", grotto::closed::cbrt},
{"sinc", grotto::closed::sinc},
{"softsign", grotto::closed::softsign},
{"logistic", grotto::closed::logistic},
};
for (const auto & spec : closeds)
for (const unsigned bits : precisions)
add_closed(spec.stem, spec.which, bits);
const struct { const char * stem; grotto::reduced which; } reduceds[] = {
{"ln", grotto::reduced::ln},
{"exp", grotto::reduced::exp},
{"sin", grotto::reduced::sin},
{"expm1", grotto::reduced::expm1},
{"log1p", grotto::reduced::log1p},
{"sqrt", grotto::reduced::sqrt},
};
for (const auto & spec : reduceds)
for (const unsigned bits : precisions)
add_reduced(spec.stem, spec.which, bits);
const auto add_exact = [&](const std::string & name, unsigned bits, auto fn) {
std::vector<std::int64_t> samples;
for (const std::int64_t raw : around_one(bits))
{
try
{
(void)fn(raw);
samples.push_back(raw);
}
catch (const std::exception &)
{
}
}
add_table(works, "exact", name, fn, std::move(samples));
};
for (const unsigned bits : {8u, 16u, 32u})
{
const auto tag = "_p" + std::to_string(bits);
add_exact("exact_dec_floor" + tag, bits, [bits](std::int64_t raw) {
return grotto::eval_dec_floor(raw, bits);
});
add_exact("exact_bit_width" + tag, bits, [bits](std::int64_t raw) {
return grotto::eval_bit_width<std::int64_t>(raw, bits);
});
add_exact("exact_dec_width" + tag, bits, [bits](std::int64_t raw) {
return grotto::eval_dec_width(raw, bits);
});
add_exact("exact_ilog256" + tag, bits, [bits](std::int64_t raw) {
return grotto::eval_ilog256<std::int64_t>(raw, bits);
});
}
add_horner_degree<1>(works);
add_horner_degree<3>(works);
using in_t = std::uint32_t;
const in_t center = 50;
const in_t eta = 7;
for (const std::size_t degree : std::initializer_list<std::size_t>{1, 4, 8, 16})
{
works.push_back(cell("keygen", "poly_keygen_d" + std::to_string(degree), 1,
[degree] {
auto built = grotto::make_offset_poly_keys(in_t{50}, degree);
return touch_word(built.wrap_share[0][0],
built.wrap_share.size() * sizeof(built.wrap_share[0]));
}));
using mat_t = decltype(grotto::make_offset_poly_keys(center, degree));
auto mat = std::make_shared<mat_t>(grotto::make_offset_poly_keys(center, degree));
for (const std::size_t pieces : std::initializer_list<std::size_t>{4, 8, 16})
{
if (degree == 16 && pieces == 16)
continue;
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(pieces, 800));
auto rows = std::make_shared<std::vector<std::vector<std::uint64_t>>>(
poly_rows(pieces, degree));
const auto name = "poly_eval_d" + std::to_string(degree)
+ "_p" + std::to_string(pieces);
works.push_back(cell("poly", name, 1, [mat, knots, rows] {
const auto a = grotto::offset_poly_eval<0>(*mat, *knots, *rows, eta, nullptr);
const auto b = grotto::offset_poly_eval<1>(*mat, *knots, *rows, eta, nullptr);
return touch_word(a ^ b, 16);
}));
}
}
for (const std::size_t degree : std::initializer_list<std::size_t>{1, 4, 8, 16})
{
works.push_back(cell("keygen", "jet_keygen_d" + std::to_string(degree), 1,
[degree] {
auto built = grotto::make_offset_jet_keys(in_t{50}, degree);
return touch_word(built.wrap_share[0][0],
built.wrap_share.size() * sizeof(built.wrap_share[0]));
}));
using mat_t = decltype(grotto::make_offset_jet_keys(center, degree));
auto mat = std::make_shared<mat_t>(grotto::make_offset_jet_keys(center, degree));
for (const std::size_t pieces : std::initializer_list<std::size_t>{4, 16})
{
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(pieces, 800));
auto col = std::make_shared<std::vector<std::uint64_t>>(coeff_col(degree));
const auto name = "jet_eval_d" + std::to_string(degree)
+ "_p" + std::to_string(pieces);
works.push_back(cell("jet", name, 1, [mat, knots, col] {
const auto a = grotto::offset_jet_eval<0>(*mat, *knots, *col, eta);
const auto b = grotto::offset_jet_eval<1>(*mat, *knots, *col, eta);
return touch_word(a ^ b, 16);
}));
}
}
for (const std::size_t degree : std::initializer_list<std::size_t>{1, 4, 8})
{
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(8, 900));
auto col = std::make_shared<std::vector<std::uint64_t>>(coeff_col(degree));
using odd_t = decltype(grotto::make_offset_twist_keys(center, degree, std::uint64_t{3}));
using half_t = decltype(grotto::make_offset_twist_keys(center, degree, grotto::twist_half));
auto odd = std::make_shared<odd_t>(
grotto::make_offset_twist_keys(center, degree, std::uint64_t{3}));
auto half = std::make_shared<half_t>(
grotto::make_offset_twist_keys(center, degree, grotto::twist_half));
const auto dtag = std::to_string(degree);
works.push_back(cell("keygen", "twist_keygen_odd_d" + dtag, 1, [degree] {
auto built = grotto::make_offset_twist_keys(in_t{50}, degree, std::uint64_t{3});
return touch_word(built.wrap_share[0][0],
built.wrap_share.size() * sizeof(built.wrap_share[0]));
}));
works.push_back(cell("twist", "twist_eval_odd_d" + dtag + "_p8", 1,
[odd, knots, col] {
const auto a = grotto::offset_twist_eval<0>(*odd, *knots, *col, eta);
const auto b = grotto::offset_twist_eval<1>(*odd, *knots, *col, eta);
return touch_word(a ^ b, 16);
}));
works.push_back(cell("twist", "twist_eval_half_d" + dtag + "_p8", 1,
[half, knots, col] {
const auto a = grotto::offset_twist_eval<0>(*half, *knots, *col, eta);
const auto b = grotto::offset_twist_eval<1>(*half, *knots, *col, eta);
return touch_word(a ^ b, 16);
}));
}
{
const std::vector<std::uint64_t> fib_state{1, 0};
const std::vector<std::uint64_t> wide_state{1, 2, 3, 4};
using fib_t = decltype(grotto::make_offset_repr_keys(center, fib_state));
using wide_t = decltype(grotto::make_offset_repr_keys(center, wide_state));
auto fib = std::make_shared<fib_t>(grotto::make_offset_repr_keys(center, fib_state));
auto wide = std::make_shared<wide_t>(grotto::make_offset_repr_keys(center, wide_state));
auto fib_m = std::make_shared<std::vector<std::vector<std::uint64_t>>>(
grotto::offset_repr_fibonacci_matrix());
auto tri_m = std::make_shared<std::vector<std::vector<std::uint64_t>>>(
std::vector<std::vector<std::uint64_t>>{
{1, 1, 0, 0},
{0, 1, 1, 0},
{0, 0, 1, 1},
{0, 0, 0, 1},
});
works.push_back(cell("keygen", "repr_keygen_dim2", 1, [] {
auto built = grotto::make_offset_repr_keys(in_t{50}, std::vector<std::uint64_t>{1, 0});
return touch_word(built.wrap_share[0][0],
built.wrap_share.size() * sizeof(built.wrap_share[0]));
}));
works.push_back(cell("keygen", "repr_keygen_dim4", 1, [] {
auto built = grotto::make_offset_repr_keys(in_t{50},
std::vector<std::uint64_t>{1, 2, 3, 4});
return touch_word(built.wrap_share[0][0],
built.wrap_share.size() * sizeof(built.wrap_share[0]));
}));
for (const std::size_t pieces : std::initializer_list<std::size_t>{4, 8, 16})
{
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(pieces, 700));
const auto tag = "_p" + std::to_string(pieces);
works.push_back(cell("repr", "repr_eval_fib" + tag, 1, [fib, fib_m, knots] {
const auto a = grotto::offset_repr_eval<0>(*fib, *fib_m, *knots, eta);
const auto b = grotto::offset_repr_eval<1>(*fib, *fib_m, *knots, eta);
return touch_vec(a) + touch_vec(b);
}));
works.push_back(cell("repr", "repr_eval_dim4" + tag, 1, [wide, tri_m, knots] {
const auto a = grotto::offset_repr_eval<0>(*wide, *tri_m, *knots, eta);
const auto b = grotto::offset_repr_eval<1>(*wide, *tri_m, *knots, eta);
return touch_vec(a) + touch_vec(b);
}));
}
works.push_back(cell("repr", "repr_crc32_jump", 64, [] {
std::uint32_t state = 0x12345678u;
for (unsigned i = 0; i < 64; ++i)
state = grotto::offset_repr_crc32_jump(state, 1ull << (i % 17));
return touch_word(state, 64u * 32u * sizeof(std::uint32_t));
}));
}
add_prefix_width<std::uint16_t>(works, "u16", 400);
add_prefix_width<std::uint32_t>(works, "u32", 100000);
return profile::run_works(opt, works);
}

365
test/profile/harness.hpp Normal file
View file

@ -0,0 +1,365 @@
/// @file test/profile/harness.hpp
/// @brief Timing loop shared by the in-process profile drivers.
#ifndef LIBDPF_TEST_PROFILE_HARNESS_HPP__
#define LIBDPF_TEST_PROFILE_HARNESS_HPP__
#include <algorithm>
#include <array>
#include <cstdint>
#include <cstring>
#include <functional>
#include <iostream>
#include <optional>
#include <stdexcept>
#include <string>
#include <type_traits>
#include <utility>
#include <vector>
#include <chrono>
#include "dpf/prg_count.hpp"
namespace profile
{
struct sample
{
std::uint64_t sink = 0;
std::uint64_t out_bytes = 0;
/// @brief Set when the case instruments the named cost; blank in TSV otherwise.
std::optional<std::uint64_t> prg_evals;
std::optional<std::uint64_t> preprocess_bytes;
std::optional<std::uint64_t> alloc_bytes;
std::optional<std::uint64_t> logical_bytes;
};
inline sample operator+(sample a, sample b)
{
a.sink ^= b.sink + 0x9e3779b97f4a7c15ull;
a.out_bytes += b.out_bytes;
if (a.prg_evals || b.prg_evals)
a.prg_evals = a.prg_evals.value_or(0) + b.prg_evals.value_or(0);
if (a.preprocess_bytes || b.preprocess_bytes)
a.preprocess_bytes = a.preprocess_bytes.value_or(0)
+ b.preprocess_bytes.value_or(0);
if (a.alloc_bytes || b.alloc_bytes)
a.alloc_bytes = a.alloc_bytes.value_or(0) + b.alloc_bytes.value_or(0);
if (a.logical_bytes || b.logical_bytes)
a.logical_bytes = a.logical_bytes.value_or(0)
+ b.logical_bytes.value_or(0);
return a;
}
/// @brief Keep `word` live. `out_bytes` is reported, not timed on its own.
inline sample touch_word(std::uint64_t word, std::uint64_t out_bytes = 0)
{
sample s;
s.sink = word;
s.out_bytes = out_bytes;
asm volatile("" : "+r"(s.sink)::"memory");
return s;
}
/// @brief Fold the ends of a buffer and publish a compiler barrier over it.
template <typename Buf>
sample touch_buf(const Buf & buf)
{
sample s;
using value_type = typename Buf::value_type;
s.out_bytes = buf.size() * sizeof(value_type);
s.sink = buf.size();
if (buf.size() != 0)
{
std::uint64_t a = 0;
std::uint64_t b = 0;
const std::size_t n = sizeof(value_type) < sizeof(a) ? sizeof(value_type) : sizeof(a);
std::memcpy(&a, &buf[0], n);
std::memcpy(&b, &buf[buf.size() - 1], n);
s.sink ^= a ^ (b + buf.size());
}
if (buf.size() != 0)
asm volatile("" : "+r"(s.sink) : "r"(buf.data()) : "memory");
else
asm volatile("" : "+r"(s.sink)::"memory");
return s;
}
template <typename T>
sample touch_vec(const std::vector<T> & v)
{
std::uint64_t h = v.size();
for (const T & x : v)
{
std::uint64_t w = 0;
if constexpr (std::is_integral_v<T>)
w = static_cast<std::uint64_t>(x);
else
{
const std::size_t n = sizeof(T) < sizeof(w) ? sizeof(T) : sizeof(w);
std::memcpy(&w, &x, n);
}
h ^= w + 0x9e3779b97f4a7c15ull;
h *= 0x100000001b3ull;
}
return touch_word(h, v.size() * sizeof(T));
}
template <typename T, std::size_t N>
sample touch_arr(const std::array<T, N> & a)
{
std::uint64_t h = N;
for (const T & x : a)
{
std::uint64_t w = 0;
if constexpr (std::is_integral_v<T>)
w = static_cast<std::uint64_t>(x);
else
w = x ? 1u : 0u;
h = (h << 1) ^ w;
}
return touch_word(h, N * sizeof(T));
}
/// @brief Reset the PRG counter, run `fn`, and attach the delta to the sample.
template <typename Fn>
sample with_prg(Fn && fn)
{
dpf::prg::reset_eval_count();
sample s = std::forward<Fn>(fn)();
s.prg_evals = dpf::prg::eval_count();
return s;
}
inline sample with_costs(sample s, std::uint64_t preprocess, std::uint64_t alloc,
std::uint64_t logical)
{
s.preprocess_bytes = preprocess;
s.alloc_bytes = alloc;
s.logical_bytes = logical;
return s;
}
inline std::string opt_field(const std::optional<std::uint64_t> & v)
{
return v ? std::to_string(*v) : std::string{};
}
inline std::uint64_t ticks()
{
#if defined(__x86_64__) || defined(__i386__)
unsigned lo = 0;
unsigned hi = 0;
asm volatile("rdtscp" : "=a"(lo), "=d"(hi)::"rcx");
return (static_cast<std::uint64_t>(hi) << 32) | lo;
#else
return 0;
#endif
}
struct work
{
std::string name;
std::uint64_t items = 1;
std::function<sample()> fn;
/// @brief Top-level group: `eval` or `grotto`.
std::string family;
/// @brief Piecewise rerun unit, such as `interval` or `horner`.
std::string slice;
/// @brief `std` is the default matrix. `heavy` is opt-in.
std::string tier = "std";
};
inline work make_work(std::string family, std::string slice, std::string name,
std::uint64_t items, std::function<sample()> fn, std::string tier = "std")
{
work w;
w.family = std::move(family);
w.slice = std::move(slice);
w.name = std::move(name);
w.items = items;
w.fn = std::move(fn);
w.tier = std::move(tier);
return w;
}
struct parsed
{
std::uint64_t repeat = 1;
std::uint64_t warmup = 0;
bool list = false;
std::vector<std::string> only;
std::vector<std::string> families;
std::vector<std::string> slices;
std::string tier;
};
inline parsed parse_args(int argc, char ** argv, std::uint64_t repeat,
std::uint64_t warmup, const char * usage)
{
parsed o;
o.repeat = repeat;
o.warmup = warmup;
for (int i = 1; i < argc; ++i)
{
const std::string a = argv[i];
auto need = [&](const char * flag) {
if (i + 1 >= argc)
{
std::cerr << "missing value for " << flag << "\n";
std::exit(2);
}
return std::string(argv[++i]);
};
if (a == "--list")
o.list = true;
else if (a == "--repeat")
o.repeat = std::stoull(need("--repeat"));
else if (a == "--warmup")
o.warmup = std::stoull(need("--warmup"));
else if (a == "--case")
o.only.push_back(need("--case"));
else if (a == "--family")
o.families.push_back(need("--family"));
else if (a == "--slice")
o.slices.push_back(need("--slice"));
else if (a == "--tier")
o.tier = need("--tier");
else if (a == "--help" || a == "-h")
{
std::cout << usage;
std::exit(0);
}
else
{
std::cerr << "unknown argument: " << a << "\n" << usage;
std::exit(2);
}
}
if (o.repeat == 0)
o.repeat = 1;
return o;
}
inline bool contains(const std::vector<std::string> & hay, const std::string & needle)
{
return std::find(hay.begin(), hay.end(), needle) != hay.end();
}
inline bool matches(const parsed & opt, const work & w)
{
if (!opt.tier.empty() && opt.tier != "all" && w.tier != opt.tier)
return false;
if (!opt.families.empty() && !contains(opt.families, w.family))
return false;
if (!opt.slices.empty() && !contains(opt.slices, w.slice))
return false;
if (!opt.only.empty() && !contains(opt.only, w.name))
return false;
return true;
}
inline int run_works(const parsed & opt, const std::vector<work> & all)
{
for (const auto & name : opt.only)
{
bool found = false;
for (const auto & w : all)
found = found || w.name == name;
if (!found)
{
std::cerr << "unknown case: " << name << "\n";
return 2;
}
}
std::vector<const work *> chosen;
for (const auto & w : all)
{
if (matches(opt, w))
chosen.push_back(&w);
}
if (chosen.empty())
{
std::cerr << "no cases match the requested family/slice/tier/case\n";
return 2;
}
if (opt.list)
{
std::cout << "family\tslice\tcase\titems\ttier\n";
for (const work * w : chosen)
std::cout << w->family << '\t' << w->slice << '\t' << w->name
<< '\t' << w->items << '\t' << w->tier << '\n';
return 0;
}
std::cout
<< "family\tslice\tcase\titems\trepeat\twarmup\tavg_ns\tmin_ns\tmax_ns\t"
<< "avg_cycles\tper_item_ns\tout_bytes\tsink\t"
<< "prg_evals\tpreprocess_bytes\talloc_bytes\tlogical_bytes\tlayout_waste\n";
int fails = 0;
std::uint64_t all_sink = 0;
for (const work * w : chosen)
{
try
{
for (std::uint64_t i = 0; i < opt.warmup; ++i)
all_sink ^= w->fn().sink;
std::uint64_t total_ns = 0;
std::uint64_t min_ns = ~std::uint64_t{0};
std::uint64_t max_ns = 0;
std::uint64_t total_cycles = 0;
sample last{};
for (std::uint64_t i = 0; i < opt.repeat; ++i)
{
const auto c0 = ticks();
const auto t0 = std::chrono::steady_clock::now();
const sample s = w->fn();
const auto t1 = std::chrono::steady_clock::now();
const auto c1 = ticks();
const auto ns = static_cast<std::uint64_t>(
std::chrono::duration_cast<std::chrono::nanoseconds>(t1 - t0).count());
total_ns += ns;
min_ns = std::min(min_ns, ns);
max_ns = std::max(max_ns, ns);
total_cycles += c1 - c0;
// Last sample, not an xor across repeats: identical samples
// would cancel and the column would read as zero.
last = s;
all_sink ^= s.sink + i;
}
const std::uint64_t avg_ns = total_ns / opt.repeat;
const std::uint64_t avg_cycles = total_cycles / opt.repeat;
const std::uint64_t per_item = w->items == 0 ? avg_ns : avg_ns / w->items;
std::string layout_waste;
if (last.alloc_bytes && last.logical_bytes
&& *last.alloc_bytes >= *last.logical_bytes)
{
layout_waste = std::to_string(*last.alloc_bytes - *last.logical_bytes);
}
std::cout << w->family << '\t' << w->slice << '\t' << w->name << '\t'
<< w->items << '\t' << opt.repeat << '\t'
<< opt.warmup << '\t' << avg_ns << '\t' << min_ns << '\t' << max_ns
<< '\t' << avg_cycles << '\t' << per_item << '\t' << last.out_bytes
<< '\t' << last.sink << '\t'
<< opt_field(last.prg_evals) << '\t'
<< opt_field(last.preprocess_bytes) << '\t'
<< opt_field(last.alloc_bytes) << '\t'
<< opt_field(last.logical_bytes) << '\t'
<< layout_waste << '\n';
}
catch (const std::exception & ex)
{
std::cerr << w->name << " failed: " << ex.what() << "\n";
++fails;
}
}
std::cout << "sink\t" << all_sink << "\n";
return fails == 0 ? 0 : 1;
}
} // namespace profile
#endif // LIBDPF_TEST_PROFILE_HARNESS_HPP__

View file

@ -0,0 +1,478 @@
/// @file test/profile/party_profile.cpp
/// @brief Workload for the (2+1) party protocols.
///
/// Spawns p0/p1/p2 the same way party_bench does. Each role reports CPU time
/// and the bytes and frames of the flow itself. The repeat barrier is not
/// included in those counters. `wall_ms` includes process startup.
///
/// profile_party --list
/// profile_party --suite core --repeat 5 --warmup 1
/// profile_party --suite extreme
/// profile_party --suite gadget --repeat 3 --warmup 1
/// profile_party --tag "beaver,bench" --repeat 3
/// profile_party --case beaver_dot_n32
///
/// Profile-guided build. The party binaries must be rebuilt with the same
/// mode, because the protocol code lives in those processes:
/// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release
/// cmake --build build-pgo --target profile_party p0 p1 p2
/// build-pgo/bin/profile_party --suite all --repeat 4 --warmup 1
/// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release
/// cmake --build build-pgo --target profile_party p0 p1 p2
#include "cases.hpp"
#include "registry.hpp"
#include "spawn.hpp"
#include <algorithm>
#include <cstdint>
#include <cstdlib>
#include <iostream>
#include <string>
#include <map>
#include <sstream>
#include <vector>
namespace
{
const char * const kCore[] = {
"beaver_dot_n32",
"beaver_product_100_200",
"beaver_stream_n128",
"beaver_horner_d4",
"beaver_batch_one_round",
"beaver_xor_mux",
"dcf_gt",
"dpf_point_2a_7",
"geneval_point",
"geneval_arith_point",
"blocked_dcf_point",
"ds_key_agrees",
"grotto_prefix_horner",
"verifiable_honest_point",
};
const char * const kExtreme[] = {
"beaver_dot_n128",
"beaver_stream_n2048",
"dcf_dense_gt",
"dcf_blocked_interval_ip",
"grotto_signed_prefix_dense",
"grotto_offset_horner_d3",
"grotto_offset_horner_multipiece",
"grotto_geneval_offset_horner",
"geneval_interval_dense",
"geneval_cmp_dense_gt",
"ds_cmp_many_points",
"recent_offset_poly",
"recent_offset_jet",
"recent_offset_twist",
"recent_offset_repr",
"recent_dist_dpf3_point",
"wildcard_single_leaf",
};
const char * const kGadget[] = {
"arith_proj_m5",
"arith_proj_m17",
"arith_proj_m64",
"arith_mul_p5",
"arith_mul_p7",
"arith_mul_p11",
"arith_thresh_b8",
"arith_thresh_b16",
"arith_chain_mul4",
"yao_if_4_2",
"yao_if_16_8",
"yao_onehot_k4",
"yao_onehot_k8",
"flute_d2",
"flute_d4",
"flute_d8",
"flute_d4_o8",
"shuffle_n16",
"shuffle_n64",
"shuffle_n256",
};
const char kUsage[] =
"profile_party [--list] [--slice S ...] [--tier std|heavy|all|smoke]\n"
" [--suite core|extreme|gadget|all] [--tag TAGS] [--case NAME ...]\n"
" [--repeat N=3] [--warmup W=1]\n"
"Default run is the core suite. --suite gadget is the word-garbling,\n"
"stacked-Yao, FLUTE, and hidden-shuffle battery on the party mesh.\n"
"--slice/--tier select the cost-matrix catalog instead.\n"
"--list with no filter prints every bench flow.\n"
"heavy: wildcard_single_leaf, beaver_stream_n512/n2048, dcf_full_*, *_domain.\n"
"smoke: flows that are not benchable (rejects and negative tests).\n";
bool starts_with(const std::string & s, const char * prefix)
{
const auto n = std::char_traits<char>::length(prefix);
return s.size() >= n && s.compare(0, n, prefix) == 0;
}
std::string slice_of(const std::string & name)
{
if (starts_with(name, "arith_"))
return "arith-garble";
if (starts_with(name, "yao_"))
return "yao-stack";
if (starts_with(name, "flute_"))
return "flute";
if (starts_with(name, "shuffle_"))
return "shuffle";
if (starts_with(name, "beaver_dot"))
return "beaver-dot";
if (starts_with(name, "beaver_stream"))
return "beaver-stream";
if (starts_with(name, "beaver_scale"))
return "beaver-scale";
if (starts_with(name, "beaver_horner") || name == "beaver_sign_horner")
return "beaver-horner";
if (starts_with(name, "beaver_product") || starts_with(name, "beaver_mul")
|| starts_with(name, "beaver_chained"))
return "beaver-product";
if (starts_with(name, "beaver_poly") || name == "beaver_like_terms"
|| starts_with(name, "beaver_nested") || starts_with(name, "beaver_tuple")
|| starts_with(name, "beaver_mixed") || starts_with(name, "beaver_factored"))
return "beaver-poly";
if (starts_with(name, "beaver_"))
return "beaver-misc";
if (starts_with(name, "dcf_full"))
return "dcf-full";
if (starts_with(name, "dcf_dense"))
return "dcf-dense";
if (starts_with(name, "dcf_blocked") || starts_with(name, "blocked_"))
return "blocked";
if (starts_with(name, "dcf_"))
return "dcf";
if (starts_with(name, "geneval_cmp"))
return "geneval-cmp";
if (starts_with(name, "geneval_"))
return "geneval";
if (starts_with(name, "grotto_"))
return "grotto-net";
if (starts_with(name, "ds_"))
return "ds";
if (starts_with(name, "recent_offset") || starts_with(name, "recent_ring")
|| starts_with(name, "recent_closed") || starts_with(name, "recent_exact"))
return "recent-grotto";
if (starts_with(name, "recent_dpf3") || starts_with(name, "recent_dist_dpf3"))
return "dpf3";
if (starts_with(name, "recent_"))
return "recent";
if (starts_with(name, "wildcard_"))
return "wildcard";
if (starts_with(name, "verifiable_"))
return "verifiable";
if (starts_with(name, "dpf_"))
return "dpf-point";
if (starts_with(name, "carry_") || starts_with(name, "extractable_"))
return "auth";
if (starts_with(name, "cov_paint") || starts_with(name, "cov_idcf")
|| name == "cov_eq_at" || name == "cov_ccmp")
return "coverage-cmp";
if (starts_with(name, "cov_geneval") || name == "cov_idpf_at"
|| name == "cov_eval_inner_product")
return "coverage-eval";
if (starts_with(name, "cov_"))
return "coverage";
return "other";
}
std::string tier_of(const std::string & name, bool)
{
const bool domain = name.size() >= 7
&& name.compare(name.size() - 7, 7, "_domain") == 0;
const bool dpf3_walk = name.find("dpf3") != std::string::npos
&& (name.find("update") != std::string::npos
|| name.find("proof") != std::string::npos
|| name.find("blocked") != std::string::npos);
if (name == "wildcard_single_leaf"
|| name == "beaver_stream_n512"
|| name == "beaver_stream_n2048"
|| starts_with(name, "dcf_full_")
|| domain
|| dpf3_walk
|| starts_with(name, "recent_dist_dpf3"))
return "heavy";
const bool negative = name.find("fail") != std::string::npos
|| name.find("tamper") != std::string::npos
|| name.find("reject") != std::string::npos
|| name.find("wrong") != std::string::npos
|| name.find("bad_use") != std::string::npos;
if (negative)
return "smoke";
return "std";
}
std::map<std::string, std::string> fields_of(const std::string & line)
{
std::map<std::string, std::string> out;
std::istringstream in(line);
std::string tok;
while (in >> tok)
{
const auto eq = tok.find('=');
if (eq == std::string::npos)
continue;
out.emplace(tok.substr(0, eq), tok.substr(eq + 1));
}
return out;
}
std::string field(const std::map<std::string, std::string> & m, const char * key)
{
const auto it = m.find(key);
return it == m.end() ? "0" : it->second;
}
} // namespace
int main(int argc, char ** argv)
{
dpf::party::register_all_flows();
std::string suite = "core";
std::string tag;
std::string tier;
std::vector<std::string> slices;
std::vector<std::string> case_names;
std::uint64_t repeat = 3;
std::uint64_t warmup = 1;
bool list_only = false;
bool suite_set = false;
bool tier_set = false;
for (int i = 1; i < argc; ++i)
{
const std::string a = argv[i];
auto need = [&](const char * flag) {
if (i + 1 >= argc)
{
std::cerr << "missing value for " << flag << "\n";
std::exit(2);
}
return std::string(argv[++i]);
};
if (a == "--list")
list_only = true;
else if (a == "--suite")
{
suite = need("--suite");
suite_set = true;
}
else if (a == "--tag")
tag = need("--tag");
else if (a == "--case")
case_names.push_back(need("--case"));
else if (a == "--slice")
slices.push_back(need("--slice"));
else if (a == "--tier")
{
tier = need("--tier");
tier_set = true;
}
else if (a == "--repeat")
repeat = std::stoull(need("--repeat"));
else if (a == "--warmup")
warmup = std::stoull(need("--warmup"));
else if (a == "--help" || a == "-h")
{
std::cout << kUsage;
return 0;
}
else
{
std::cerr << "unknown argument: " << a << "\n" << kUsage;
return 2;
}
}
if (repeat == 0)
repeat = 1;
const bool catalog_mode = tier_set || !slices.empty()
|| (list_only && !suite_set && tag.empty() && case_names.empty());
struct picked
{
std::string name;
std::string slice;
std::string tier;
};
std::vector<picked> picked_flows;
auto take_catalog = [&](const std::string & want_tier) {
for (const auto * f : dpf::party::select_flows(""))
{
const std::string slice = slice_of(f->name);
const std::string flow_tier = tier_of(f->name, f->bench);
if (want_tier == "std" && flow_tier != "std")
continue;
if (want_tier == "heavy" && flow_tier != "heavy")
continue;
if (want_tier == "smoke" && flow_tier != "smoke")
continue;
if (want_tier == "all" && flow_tier == "smoke")
continue;
if (!slices.empty()
&& std::find(slices.begin(), slices.end(), slice) == slices.end())
continue;
if (!case_names.empty()
&& std::find(case_names.begin(), case_names.end(), f->name) == case_names.end())
continue;
picked_flows.push_back(picked{f->name, slice, flow_tier});
}
};
if (catalog_mode)
{
const std::string want = tier_set ? tier : (list_only ? "all" : "std");
if (want != "std" && want != "heavy" && want != "all" && want != "smoke")
{
std::cerr << "unknown tier: " << want << "\n" << kUsage;
return 2;
}
take_catalog(want);
for (const auto & name : case_names)
{
bool found = false;
for (const auto & row : picked_flows)
found = found || row.name == name;
if (!found)
{
std::cerr << "unknown flow: " << name << "\n";
return 2;
}
}
if (picked_flows.empty())
{
std::cerr << "no flows match the requested slice/tier/case\n";
return 2;
}
}
else
{
std::vector<std::string> names;
if (!case_names.empty())
names = std::move(case_names);
else if (!tag.empty())
{
for (const auto * f : dpf::party::select_flows(tag))
names.emplace_back(f->name);
if (names.empty())
{
std::cerr << "no flows match tag: " << tag << "\n";
return 2;
}
}
else if (suite == "core" || suite == "all")
{
for (const char * n : kCore)
names.emplace_back(n);
if (suite == "all")
{
for (const char * n : kExtreme)
names.emplace_back(n);
for (const char * n : kGadget)
names.emplace_back(n);
}
}
else if (suite == "extreme")
{
for (const char * n : kExtreme)
names.emplace_back(n);
}
else if (suite == "gadget")
{
for (const char * n : kGadget)
names.emplace_back(n);
}
else
{
std::cerr << "unknown suite: " << suite << "\n" << kUsage;
return 2;
}
for (const auto & name : names)
{
const auto * f = dpf::party::find_flow(name);
if (!f)
{
std::cerr << "unknown flow: " << name << "\n";
return 2;
}
picked_flows.push_back(picked{name, slice_of(name), tier_of(name, f->bench)});
}
}
if (list_only)
{
std::cout << "family\tslice\tcase\titems\ttier\n";
for (const auto & row : picked_flows)
std::cout << "party\t" << row.slice << '\t' << row.name
<< "\t1\t" << row.tier << '\n';
return 0;
}
dpf::party::spawn_opts opts;
opts.repeat = repeat;
opts.warmup = warmup;
opts.metrics = true;
std::cout
<< "family\tslice\ttier\tflow\trole\twall_ms\tavg_ns\tmin_ns\tmax_ns\t"
<< "bytes_sent\tbytes_recv\tframes_sent\tframes_recv\t"
<< "bytes_recv_from_p2\tbytes_recv_from_peer\t"
<< "bytes_sent_to_p2\tbytes_sent_to_peer\t"
<< "payload_sent\tpayload_recv\twire_overhead\trounds\tprg_evals\t"
<< "avg_bytes_sent\tavg_frames_sent\trc\n";
int fails = 0;
for (const auto & row : picked_flows)
{
const auto result = dpf::party::spawn_trio_flow(row.name, opts);
if (result.metrics_lines.empty())
{
std::cout << "party\t" << row.slice << '\t' << row.tier << '\t'
<< row.name << "\t-\t" << result.wall_ms
<< "\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t"
<< result.rc[0] << "," << result.rc[1] << "," << result.rc[2]
<< "\n";
}
for (const auto & line : result.metrics_lines)
{
const auto f = fields_of(line);
const auto role = field(f, "role");
const int rc = role == "p0" ? result.rc[0]
: role == "p1" ? result.rc[1]
: role == "p2" ? result.rc[2] : 0;
std::cout << "party\t" << row.slice << '\t' << row.tier << '\t'
<< row.name << '\t' << role << '\t' << result.wall_ms << '\t'
<< field(f, "avg_ns") << '\t'
<< field(f, "min_ns") << '\t'
<< field(f, "max_ns") << '\t'
<< field(f, "bytes_sent") << '\t'
<< field(f, "bytes_recv") << '\t'
<< field(f, "frames_sent") << '\t'
<< field(f, "frames_recv") << '\t'
<< field(f, "bytes_recv_from_p2") << '\t'
<< field(f, "bytes_recv_from_peer") << '\t'
<< field(f, "bytes_sent_to_p2") << '\t'
<< field(f, "bytes_sent_to_peer") << '\t'
<< field(f, "payload_sent") << '\t'
<< field(f, "payload_recv") << '\t'
<< field(f, "wire_overhead") << '\t'
<< field(f, "rounds") << '\t'
<< field(f, "prg_evals") << '\t'
<< field(f, "avg_bytes_sent") << '\t'
<< field(f, "avg_frames_sent") << '\t'
<< rc << '\n';
}
if (result.rc[0] || result.rc[1] || result.rc[2])
++fails;
}
return fails ? 1 : 0;
}