Checkpoint the party/runtime stack before share-program and malicious-mode work.
Ship the TLS mesh, composer, Beaver/Yao/leaf MPC, prep/online paths, apps, and docs so the tree is pushable before elevating share_expr, security_mode, and prep resume. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
695f8e84f7
commit
0d22946a0e
1835 changed files with 170291 additions and 2849 deletions
BIN
test/profile/__pycache__/cost_matrix.cpython-314.pyc
Normal file
BIN
test/profile/__pycache__/cost_matrix.cpython-314.pyc
Normal file
Binary file not shown.
505
test/profile/cost_matrix.py
Executable file
505
test/profile/cost_matrix.py
Executable file
|
|
@ -0,0 +1,505 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Cost matrix for libdpf profile harnesses.
|
||||
|
||||
Each cell is one named case. Cells are grouped into slices so an optimization
|
||||
pass can rerun only the code it touched. A label is a directory of results.
|
||||
Rerunning a slice under the same label replaces those rows and leaves the rest.
|
||||
|
||||
Examples:
|
||||
test/profile/cost_matrix.py --list
|
||||
test/profile/cost_matrix.py --label baseline
|
||||
test/profile/cost_matrix.py --label baseline --slice interval
|
||||
test/profile/cost_matrix.py --label baseline --slice memoizer --slice bits
|
||||
test/profile/cost_matrix.py --label baseline --family grotto --slice horner
|
||||
test/profile/cost_matrix.py --label baseline --case interval_u32_L4096
|
||||
test/profile/cost_matrix.py --label after-horner --slice horner
|
||||
test/profile/cost_matrix.py --compare baseline after-horner
|
||||
|
||||
Tiers:
|
||||
std default. Benchable party flows, plus every in-process cell.
|
||||
heavy wildcard_single_leaf, long beaver streams, full-domain DCFs,
|
||||
and party flows whose names end in _domain.
|
||||
all std and heavy.
|
||||
smoke party flows that are negative tests, not cost points.
|
||||
|
||||
Default repeats are per slice (party 3, tables 20, walks 12). --repeat
|
||||
overrides every selected cell.
|
||||
|
||||
Columns (blank when not instrumented):
|
||||
prg_evals, bytes_sent, bytes_recv_from_p2, bytes_recv_from_peer,
|
||||
rounds (channel exchange barriers; not frames), preprocess_bytes,
|
||||
alloc_bytes, logical_bytes, layout_waste (alloc-logical),
|
||||
payload_sent/recv, wire_overhead (framed bytes minus payload).
|
||||
For p2, bytes_*_from_peer / bytes_*_to_peer sum the p0 and p1 links.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
LONG_FIELDS = [
|
||||
"family",
|
||||
"slice",
|
||||
"tier",
|
||||
"case",
|
||||
"role",
|
||||
"items",
|
||||
"repeat",
|
||||
"warmup",
|
||||
"avg_ns",
|
||||
"min_ns",
|
||||
"max_ns",
|
||||
"avg_cycles",
|
||||
"per_item_ns",
|
||||
"out_bytes",
|
||||
"prg_evals",
|
||||
"preprocess_bytes",
|
||||
"alloc_bytes",
|
||||
"logical_bytes",
|
||||
"layout_waste",
|
||||
"bytes_sent",
|
||||
"bytes_recv",
|
||||
"bytes_recv_from_p2",
|
||||
"bytes_recv_from_peer",
|
||||
"bytes_sent_to_p2",
|
||||
"bytes_sent_to_peer",
|
||||
"payload_sent",
|
||||
"payload_recv",
|
||||
"wire_overhead",
|
||||
"rounds",
|
||||
"frames_sent",
|
||||
"frames_recv",
|
||||
"avg_bytes_sent",
|
||||
"avg_frames_sent",
|
||||
"wall_ms",
|
||||
"rc",
|
||||
"sink",
|
||||
]
|
||||
|
||||
MATRIX_FIELDS = [
|
||||
"family",
|
||||
"slice",
|
||||
"tier",
|
||||
"case",
|
||||
"role",
|
||||
"avg_ns",
|
||||
"per_item_ns",
|
||||
"avg_cycles",
|
||||
"out_bytes",
|
||||
"prg_evals",
|
||||
"preprocess_bytes",
|
||||
"alloc_bytes",
|
||||
"logical_bytes",
|
||||
"layout_waste",
|
||||
"bytes_sent",
|
||||
"bytes_recv",
|
||||
"bytes_recv_from_p2",
|
||||
"bytes_recv_from_peer",
|
||||
"rounds",
|
||||
"wire_overhead",
|
||||
"frames_sent",
|
||||
"frames_recv",
|
||||
"wall_ms",
|
||||
"repeat",
|
||||
]
|
||||
|
||||
COMPARE_NUM = [
|
||||
"avg_ns",
|
||||
"prg_evals",
|
||||
"bytes_sent",
|
||||
"bytes_recv_from_p2",
|
||||
"bytes_recv_from_peer",
|
||||
"rounds",
|
||||
"preprocess_bytes",
|
||||
"alloc_bytes",
|
||||
"wire_overhead",
|
||||
]
|
||||
|
||||
|
||||
def default_bin_dir() -> Path:
|
||||
here = Path(__file__).resolve().parent
|
||||
candidate = here.parents[1] / "build" / "test" / "bin"
|
||||
return candidate
|
||||
|
||||
|
||||
def timing_for(family: str, slice_name: str, tier: str) -> tuple[int, int]:
|
||||
if tier == "heavy":
|
||||
return 1, 0
|
||||
if family == "party":
|
||||
return 3, 1
|
||||
if slice_name in {"window", "principal", "closed", "reduced", "exact"}:
|
||||
return 20, 2
|
||||
if slice_name == "keygen":
|
||||
return 8, 1
|
||||
if slice_name in {"memoizer", "bits", "inner-product", "helpers"}:
|
||||
return 12, 2
|
||||
return 12, 2
|
||||
|
||||
|
||||
def run_capture(cmd: list[str]) -> tuple[int, str, str]:
|
||||
proc = subprocess.run(cmd, text=True, capture_output=True)
|
||||
return proc.returncode, proc.stdout, proc.stderr
|
||||
|
||||
|
||||
def parse_list(text: str) -> list[dict[str, str]]:
|
||||
rows = []
|
||||
reader = csv.DictReader(text.splitlines(), delimiter="\t")
|
||||
for row in reader:
|
||||
if not row.get("case"):
|
||||
continue
|
||||
rows.append(row)
|
||||
return rows
|
||||
|
||||
|
||||
def discover(bin_dir: Path) -> list[dict[str, str]]:
|
||||
cells: list[dict[str, str]] = []
|
||||
for binary in ("profile_eval", "profile_grotto", "profile_party"):
|
||||
path = bin_dir / binary
|
||||
if not path.is_file():
|
||||
raise SystemExit(f"missing {path}; build the profile targets first")
|
||||
cmd = [str(path), "--list"]
|
||||
if binary == "profile_party":
|
||||
cmd.extend(["--tier", "all"])
|
||||
rc, out, err = run_capture(cmd)
|
||||
if rc != 0:
|
||||
raise SystemExit(err or out or f"{binary} --list failed")
|
||||
if err.strip():
|
||||
print(err, file=sys.stderr, end="" if err.endswith("\n") else "\n")
|
||||
for row in parse_list(out):
|
||||
row["driver"] = binary
|
||||
row.setdefault("tier", "std")
|
||||
cells.append(row)
|
||||
heavy_party = [
|
||||
"wildcard_single_leaf",
|
||||
"beaver_stream_n512",
|
||||
"beaver_stream_n2048",
|
||||
]
|
||||
for row in cells:
|
||||
if row["family"] == "party" and (
|
||||
row["case"] in heavy_party or row["case"].startswith("dcf_full_")
|
||||
):
|
||||
row["tier"] = "heavy"
|
||||
return cells
|
||||
|
||||
|
||||
def select_cells(cells, args) -> list[dict[str, str]]:
|
||||
chosen = []
|
||||
for row in cells:
|
||||
if args.family and row["family"] not in args.family:
|
||||
continue
|
||||
if args.slice and row["slice"] not in args.slice:
|
||||
continue
|
||||
if args.case and row["case"] not in args.case:
|
||||
continue
|
||||
tier = row.get("tier") or "std"
|
||||
if args.tier == "std" and tier != "std":
|
||||
continue
|
||||
if args.tier == "heavy" and tier != "heavy":
|
||||
continue
|
||||
if args.tier == "smoke" and tier != "smoke":
|
||||
continue
|
||||
if args.tier == "all" and tier == "smoke":
|
||||
continue
|
||||
chosen.append(row)
|
||||
return chosen
|
||||
|
||||
|
||||
def blank_cost_fields() -> dict[str, str]:
|
||||
return {
|
||||
"prg_evals": "",
|
||||
"preprocess_bytes": "",
|
||||
"alloc_bytes": "",
|
||||
"logical_bytes": "",
|
||||
"layout_waste": "",
|
||||
"bytes_sent": "",
|
||||
"bytes_recv": "",
|
||||
"bytes_recv_from_p2": "",
|
||||
"bytes_recv_from_peer": "",
|
||||
"bytes_sent_to_p2": "",
|
||||
"bytes_sent_to_peer": "",
|
||||
"payload_sent": "",
|
||||
"payload_recv": "",
|
||||
"wire_overhead": "",
|
||||
"rounds": "",
|
||||
"frames_sent": "",
|
||||
"frames_recv": "",
|
||||
"avg_bytes_sent": "",
|
||||
"avg_frames_sent": "",
|
||||
"wall_ms": "",
|
||||
}
|
||||
|
||||
|
||||
def parse_inprocess(text: str, tier_of: dict[tuple[str, str], str]) -> list[dict[str, str]]:
|
||||
rows = []
|
||||
reader = csv.DictReader(text.splitlines(), delimiter="\t")
|
||||
for row in reader:
|
||||
if not row or row.get("family") == "sink" or not row.get("case"):
|
||||
continue
|
||||
# Skip the trailing sink summary line if DictReader mis-parses it.
|
||||
if row.get("family") == "sink":
|
||||
continue
|
||||
out = blank_cost_fields()
|
||||
out.update({
|
||||
"family": row.get("family") or "",
|
||||
"slice": row.get("slice") or "",
|
||||
"tier": tier_of.get((row.get("family") or "", row.get("case") or ""), "std"),
|
||||
"case": row.get("case") or "",
|
||||
"role": "-",
|
||||
"items": row.get("items") or "",
|
||||
"repeat": row.get("repeat") or "",
|
||||
"warmup": row.get("warmup") or "",
|
||||
"avg_ns": row.get("avg_ns") or "",
|
||||
"min_ns": row.get("min_ns") or "",
|
||||
"max_ns": row.get("max_ns") or "",
|
||||
"avg_cycles": row.get("avg_cycles") or "",
|
||||
"per_item_ns": row.get("per_item_ns") or "",
|
||||
"out_bytes": row.get("out_bytes") or "",
|
||||
"prg_evals": row.get("prg_evals") or "",
|
||||
"preprocess_bytes": row.get("preprocess_bytes") or "",
|
||||
"alloc_bytes": row.get("alloc_bytes") or "",
|
||||
"logical_bytes": row.get("logical_bytes") or "",
|
||||
"layout_waste": row.get("layout_waste") or "",
|
||||
"rc": "0",
|
||||
"sink": row.get("sink") or "",
|
||||
})
|
||||
rows.append(out)
|
||||
return rows
|
||||
|
||||
|
||||
def parse_party(text: str) -> list[dict[str, str]]:
|
||||
rows = []
|
||||
reader = csv.DictReader(text.splitlines(), delimiter="\t")
|
||||
for row in reader:
|
||||
if not row.get("flow"):
|
||||
continue
|
||||
out = blank_cost_fields()
|
||||
out.update({
|
||||
"family": row.get("family") or "party",
|
||||
"slice": row.get("slice") or "",
|
||||
"tier": row.get("tier") or "std",
|
||||
"case": row["flow"],
|
||||
"role": row.get("role") or "-",
|
||||
"items": "1",
|
||||
"repeat": "",
|
||||
"warmup": "",
|
||||
"avg_ns": row.get("avg_ns") or "",
|
||||
"min_ns": row.get("min_ns") or "",
|
||||
"max_ns": row.get("max_ns") or "",
|
||||
"avg_cycles": "",
|
||||
"per_item_ns": row.get("avg_ns") or "",
|
||||
"out_bytes": "",
|
||||
"prg_evals": row.get("prg_evals") or "",
|
||||
"bytes_sent": row.get("bytes_sent") or "",
|
||||
"bytes_recv": row.get("bytes_recv") or "",
|
||||
"bytes_recv_from_p2": row.get("bytes_recv_from_p2") or "",
|
||||
"bytes_recv_from_peer": row.get("bytes_recv_from_peer") or "",
|
||||
"bytes_sent_to_p2": row.get("bytes_sent_to_p2") or "",
|
||||
"bytes_sent_to_peer": row.get("bytes_sent_to_peer") or "",
|
||||
"payload_sent": row.get("payload_sent") or "",
|
||||
"payload_recv": row.get("payload_recv") or "",
|
||||
"wire_overhead": row.get("wire_overhead") or "",
|
||||
"rounds": row.get("rounds") or "",
|
||||
"frames_sent": row.get("frames_sent") or "",
|
||||
"frames_recv": row.get("frames_recv") or "",
|
||||
"avg_bytes_sent": row.get("avg_bytes_sent") or "",
|
||||
"avg_frames_sent": row.get("avg_frames_sent") or "",
|
||||
"wall_ms": row.get("wall_ms") or "",
|
||||
"rc": row.get("rc") or "",
|
||||
"sink": "",
|
||||
})
|
||||
rows.append(out)
|
||||
return rows
|
||||
|
||||
|
||||
def fill_timing(rows: list[dict[str, str]], repeat: int, warmup: int) -> None:
|
||||
for row in rows:
|
||||
if not row["repeat"]:
|
||||
row["repeat"] = str(repeat)
|
||||
if not row["warmup"]:
|
||||
row["warmup"] = str(warmup)
|
||||
|
||||
|
||||
def invoke_group(bin_dir: Path, driver: str, cases: list[dict[str, str]],
|
||||
repeat: int, warmup: int) -> tuple[int, list[dict[str, str]], str]:
|
||||
cmd = [str(bin_dir / driver), "--repeat", str(repeat), "--warmup", str(warmup)]
|
||||
for case in cases:
|
||||
cmd.extend(["--case", case["case"]])
|
||||
rc, out, err = run_capture(cmd)
|
||||
tier_of = {(c["family"], c["case"]): c.get("tier") or "std" for c in cases}
|
||||
if driver == "profile_party":
|
||||
parsed = parse_party(out)
|
||||
else:
|
||||
parsed = parse_inprocess(out, tier_of)
|
||||
fill_timing(parsed, repeat, warmup)
|
||||
return rc, parsed, err
|
||||
|
||||
|
||||
def write_tsv(path: Path, fields: list[str], rows: list[dict[str, str]]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", newline="") as fh:
|
||||
writer = csv.DictWriter(fh, fieldnames=fields, delimiter="\t", extrasaction="ignore")
|
||||
writer.writeheader()
|
||||
for row in rows:
|
||||
writer.writerow(row)
|
||||
|
||||
|
||||
def read_tsv(path: Path) -> list[dict[str, str]]:
|
||||
if not path.is_file():
|
||||
return []
|
||||
with path.open(newline="") as fh:
|
||||
return list(csv.DictReader(fh, delimiter="\t"))
|
||||
|
||||
|
||||
def merge_rows(old: list[dict[str, str]], new: list[dict[str, str]]) -> list[dict[str, str]]:
|
||||
replaced = {(row["family"], row["slice"], row["case"], row["role"]) for row in new}
|
||||
kept = [
|
||||
row for row in old
|
||||
if (row.get("family"), row.get("slice"), row.get("case"), row.get("role")) not in replaced
|
||||
]
|
||||
# Older long.tsv rows may lack new columns; normalize on merge.
|
||||
merged = []
|
||||
for row in kept + new:
|
||||
full = blank_cost_fields()
|
||||
full.update({k: "" for k in LONG_FIELDS})
|
||||
full.update({k: v for k, v in row.items() if v is not None})
|
||||
merged.append(full)
|
||||
return merged
|
||||
|
||||
|
||||
def matrix_view(rows: list[dict[str, str]]) -> list[dict[str, str]]:
|
||||
view = []
|
||||
for row in rows:
|
||||
view.append({key: row.get(key, "") for key in MATRIX_FIELDS})
|
||||
view.sort(key=lambda r: (r["family"], r["slice"], r["case"], r["role"]))
|
||||
return view
|
||||
|
||||
|
||||
def print_catalog(cells: list[dict[str, str]]) -> None:
|
||||
groups: dict[tuple[str, str, str], int] = defaultdict(int)
|
||||
for row in cells:
|
||||
groups[(row["family"], row["slice"], row.get("tier") or "std")] += 1
|
||||
print("family\tslice\ttier\tcases")
|
||||
for (family, slice_name, tier), count in sorted(groups.items()):
|
||||
print(f"{family}\t{slice_name}\t{tier}\t{count}")
|
||||
print(f"# {len(cells)} cases in {len(groups)} slices")
|
||||
|
||||
|
||||
def compare_labels(out_root: Path, left: str, right: str) -> int:
|
||||
a_rows = read_tsv(out_root / left / "matrix.tsv")
|
||||
b_rows = read_tsv(out_root / right / "matrix.tsv")
|
||||
if not a_rows or not b_rows:
|
||||
raise SystemExit(f"need matrix.tsv in both {left} and {right}")
|
||||
b_index = {
|
||||
(row["family"], row["slice"], row["case"], row["role"]): row
|
||||
for row in b_rows
|
||||
}
|
||||
header = ["family", "slice", "case", "role"]
|
||||
for col in COMPARE_NUM:
|
||||
header.extend([f"{col}_a", f"{col}_b", f"delta_{col}"])
|
||||
print("\t".join(header))
|
||||
missing = 0
|
||||
for row in a_rows:
|
||||
key = (row["family"], row["slice"], row["case"], row["role"])
|
||||
other = b_index.get(key)
|
||||
if other is None:
|
||||
missing += 1
|
||||
continue
|
||||
parts = [row["family"], row["slice"], row["case"], row["role"]]
|
||||
for col in COMPARE_NUM:
|
||||
av_s = row.get(col, "") or ""
|
||||
bv_s = other.get(col, "") or ""
|
||||
try:
|
||||
av = float(av_s)
|
||||
bv = float(bv_s)
|
||||
delta = f"{bv - av:.0f}"
|
||||
except ValueError:
|
||||
delta = ""
|
||||
parts.extend([av_s, bv_s, delta])
|
||||
print("\t".join(parts))
|
||||
if missing:
|
||||
print(f"# {missing} rows in {left} have no match in {right}", file=sys.stderr)
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--bin-dir", type=Path, default=default_bin_dir())
|
||||
parser.add_argument("--out", type=Path, default=None,
|
||||
help="directory that holds one subdirectory per label")
|
||||
parser.add_argument("--label", default="baseline")
|
||||
parser.add_argument("--family", action="append", default=[])
|
||||
parser.add_argument("--slice", action="append", default=[])
|
||||
parser.add_argument("--case", action="append", default=[])
|
||||
parser.add_argument("--tier", choices=("std", "heavy", "all", "smoke"), default="std")
|
||||
parser.add_argument("--repeat", type=int, default=None)
|
||||
parser.add_argument("--warmup", type=int, default=None)
|
||||
parser.add_argument("--list", action="store_true")
|
||||
parser.add_argument("--compare", nargs=2, metavar=("LABEL_A", "LABEL_B"))
|
||||
args = parser.parse_args()
|
||||
|
||||
out_root = args.out
|
||||
if out_root is None:
|
||||
out_root = args.bin_dir.parent / "cost-matrix"
|
||||
|
||||
if args.compare:
|
||||
return compare_labels(out_root, args.compare[0], args.compare[1])
|
||||
|
||||
cells = discover(args.bin_dir)
|
||||
chosen = select_cells(cells, args)
|
||||
if args.list:
|
||||
print_catalog(chosen)
|
||||
return 0
|
||||
if not chosen:
|
||||
print("no cells match", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
by_driver: dict[str, list[dict[str, str]]] = defaultdict(list)
|
||||
for row in chosen:
|
||||
by_driver[row["driver"]].append(row)
|
||||
|
||||
fresh: list[dict[str, str]] = []
|
||||
failures = 0
|
||||
for driver, rows in by_driver.items():
|
||||
buckets: dict[tuple[str, str, str], list[dict[str, str]]] = defaultdict(list)
|
||||
for row in rows:
|
||||
buckets[(row["family"], row["slice"], row.get("tier") or "std")].append(row)
|
||||
for (family, slice_name, tier), group in buckets.items():
|
||||
repeat, warmup = timing_for(family, slice_name, tier)
|
||||
if args.repeat is not None:
|
||||
repeat = args.repeat
|
||||
if args.warmup is not None:
|
||||
warmup = args.warmup
|
||||
print(f"# {driver} {family}/{slice_name} tier={tier} "
|
||||
f"cases={len(group)} repeat={repeat} warmup={warmup}",
|
||||
file=sys.stderr)
|
||||
rc, parsed, err = invoke_group(args.bin_dir, driver, group, repeat, warmup)
|
||||
if err.strip():
|
||||
print(err, file=sys.stderr, end="" if err.endswith("\n") else "\n")
|
||||
got = {(row["case"], row["role"]) for row in parsed}
|
||||
expected = {row["case"] for row in group}
|
||||
have_cases = {case for case, _role in got}
|
||||
missing = sorted(expected - have_cases)
|
||||
if missing:
|
||||
print(f"# missing results: {', '.join(missing)}", file=sys.stderr)
|
||||
failures += len(missing)
|
||||
if rc != 0:
|
||||
failures += 1
|
||||
fresh.extend(parsed)
|
||||
|
||||
label_dir = out_root / args.label
|
||||
merged = merge_rows(read_tsv(label_dir / "long.tsv"), fresh)
|
||||
write_tsv(label_dir / "long.tsv", LONG_FIELDS, merged)
|
||||
write_tsv(label_dir / "matrix.tsv", MATRIX_FIELDS, matrix_view(merged))
|
||||
print(f"# wrote {label_dir / 'matrix.tsv'} ({len(merged)} rows)", file=sys.stderr)
|
||||
return 1 if failures else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
907
test/profile/eval_profile.cpp
Normal file
907
test/profile/eval_profile.cpp
Normal file
|
|
@ -0,0 +1,907 @@
|
|||
/// @file test/profile/eval_profile.cpp
|
||||
/// @brief Workload for sequence and interval evaluation.
|
||||
///
|
||||
/// Both party keys are evaluated inside each sample. Key generation is its
|
||||
/// own case, not part of the eval samples. `out_bytes` is the output buffer
|
||||
/// volume of one sample (both parties). `per_item_ns` divides by the point
|
||||
/// count, not by the number of parties.
|
||||
///
|
||||
/// profile_eval --list
|
||||
/// profile_eval --case interval_u32_L4096 --repeat 50 --warmup 3
|
||||
///
|
||||
/// Profile-guided build, separate from COVERAGE:
|
||||
/// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release
|
||||
/// cmake --build build-pgo --target profile_eval
|
||||
/// build-pgo/bin/profile_eval --repeat 40 --warmup 2
|
||||
/// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release
|
||||
/// cmake --build build-pgo --target profile_eval
|
||||
|
||||
#include "harness.hpp"
|
||||
|
||||
#include "dpf.hpp"
|
||||
|
||||
#include <cstdint>
|
||||
#include <initializer_list>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
|
||||
namespace
|
||||
{
|
||||
|
||||
using profile::sample;
|
||||
using profile::touch_buf;
|
||||
using profile::touch_word;
|
||||
using profile::work;
|
||||
|
||||
template <typename Input>
|
||||
std::shared_ptr<std::vector<Input>> clustered(Input start, std::size_t n)
|
||||
{
|
||||
auto pts = std::make_shared<std::vector<Input>>(n);
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
(*pts)[i] = static_cast<Input>(start + static_cast<Input>(i));
|
||||
return pts;
|
||||
}
|
||||
|
||||
template <typename Input>
|
||||
std::shared_ptr<std::vector<Input>> strided(Input start, Input step, std::size_t n)
|
||||
{
|
||||
auto pts = std::make_shared<std::vector<Input>>(n);
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
(*pts)[i] = static_cast<Input>(start + step * static_cast<Input>(i));
|
||||
return pts;
|
||||
}
|
||||
|
||||
template <typename Keys>
|
||||
sample hash_roots(const Keys & keys)
|
||||
{
|
||||
std::uint64_t w0 = 0;
|
||||
std::uint64_t w1 = 0;
|
||||
const auto r0 = std::get<0>(keys).root();
|
||||
const auto r1 = std::get<1>(keys).root();
|
||||
const std::size_t n0 = sizeof(r0) < sizeof(w0) ? sizeof(r0) : sizeof(w0);
|
||||
const std::size_t n1 = sizeof(r1) < sizeof(w1) ? sizeof(r1) : sizeof(w1);
|
||||
std::memcpy(&w0, &r0, n0);
|
||||
std::memcpy(&w1, &r1, n1);
|
||||
return touch_word(w0 ^ w1, sizeof(r0) + sizeof(r1));
|
||||
}
|
||||
|
||||
template <typename Keys, typename Input>
|
||||
sample interval_packed(const Keys & keys, Input from, Input to, unsigned lane_bits)
|
||||
{
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_interval(std::get<0>(keys), from, to);
|
||||
auto b = dpf::eval_interval(std::get<1>(keys), from, to);
|
||||
auto fold = [lane_bits](const auto & buf) {
|
||||
std::uint64_t w = 0;
|
||||
const std::size_t bytes = (buf.size() * lane_bits + 7u) / 8u;
|
||||
if (bytes != 0 && buf.data() != nullptr)
|
||||
{
|
||||
const std::size_t n = bytes < sizeof(w) ? bytes : sizeof(w);
|
||||
std::memcpy(&w, buf.data(), n);
|
||||
}
|
||||
return touch_word(w, bytes);
|
||||
};
|
||||
return fold(a.first) + fold(b.first);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename Keys, typename Input>
|
||||
sample interval_both(const Keys & keys, Input from, Input to)
|
||||
{
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_interval(std::get<0>(keys), from, to);
|
||||
auto b = dpf::eval_interval(std::get<1>(keys), from, to);
|
||||
return touch_buf(a.first) + touch_buf(b.first);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename Keys, typename Input>
|
||||
sample interval_reuse(const Keys & keys, Input from, Input to,
|
||||
decltype(dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to)) & buf0,
|
||||
decltype(dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to)) & buf1)
|
||||
{
|
||||
return profile::with_prg([&] {
|
||||
auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0);
|
||||
auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1);
|
||||
(void)i0;
|
||||
(void)i1;
|
||||
return touch_buf(buf0) + touch_buf(buf1);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename Keys, typename Input>
|
||||
sample interval_prove(const Keys & keys, Input from, Input to)
|
||||
{
|
||||
auto buf0 = dpf::make_output_buffer_for_interval(std::get<0>(keys), from, to);
|
||||
auto buf1 = dpf::make_output_buffer_for_interval(std::get<1>(keys), from, to);
|
||||
dpf::proof_token p0{};
|
||||
dpf::proof_token p1{};
|
||||
auto i0 = dpf::eval_interval(std::get<0>(keys), from, to, buf0, dpf::prove(p0));
|
||||
auto i1 = dpf::eval_interval(std::get<1>(keys), from, to, buf1, dpf::prove(p1));
|
||||
(void)i0;
|
||||
(void)i1;
|
||||
std::uint64_t extra = 0;
|
||||
std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0));
|
||||
return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0);
|
||||
}
|
||||
|
||||
template <typename Keys, typename Pts>
|
||||
sample sequence_both(const Keys & keys, const Pts & pts, bool output_only)
|
||||
{
|
||||
return profile::with_prg([&] {
|
||||
sample s;
|
||||
if (output_only)
|
||||
{
|
||||
auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(),
|
||||
dpf::return_output_only_tag_{});
|
||||
auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(),
|
||||
dpf::return_output_only_tag_{});
|
||||
s = touch_buf(a.first) + touch_buf(b.first);
|
||||
}
|
||||
else
|
||||
{
|
||||
auto a = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end());
|
||||
auto b = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end());
|
||||
s = touch_buf(a.first) + touch_buf(b.first);
|
||||
}
|
||||
return s;
|
||||
});
|
||||
}
|
||||
|
||||
template <typename Keys, typename Pts>
|
||||
sample sequence_breadth(const Keys & keys, const Pts & pts)
|
||||
{
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_sequence_breadth_first(std::get<0>(keys), pts.begin(), pts.end());
|
||||
auto b = dpf::eval_sequence_breadth_first(std::get<1>(keys), pts.begin(), pts.end());
|
||||
return touch_buf(a.first) + touch_buf(b.first);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename Keys, typename Recipe0, typename Recipe1>
|
||||
sample sequence_recipe(const Keys & keys, const Recipe0 & r0, const Recipe1 & r1)
|
||||
{
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_sequence(std::get<0>(keys), r0);
|
||||
auto b = dpf::eval_sequence(std::get<1>(keys), r1);
|
||||
return touch_buf(a.first) + touch_buf(b.first);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename Keys, typename Pts>
|
||||
sample sequence_prove(const Keys & keys, const Pts & pts)
|
||||
{
|
||||
auto buf0 = dpf::make_output_buffer_for_subsequence(std::get<0>(keys),
|
||||
pts.begin(), pts.end(), dpf::return_output_only_tag_{});
|
||||
auto buf1 = dpf::make_output_buffer_for_subsequence(std::get<1>(keys),
|
||||
pts.begin(), pts.end(), dpf::return_output_only_tag_{});
|
||||
dpf::proof_token p0{};
|
||||
dpf::proof_token p1{};
|
||||
auto i0 = dpf::eval_sequence(std::get<0>(keys), pts.begin(), pts.end(), buf0,
|
||||
dpf::prove(p0), dpf::return_output_only_tag_{});
|
||||
auto i1 = dpf::eval_sequence(std::get<1>(keys), pts.begin(), pts.end(), buf1,
|
||||
dpf::prove(p1), dpf::return_output_only_tag_{});
|
||||
(void)i0;
|
||||
(void)i1;
|
||||
std::uint64_t extra = 0;
|
||||
std::memcpy(&extra, &p0, sizeof(extra) < sizeof(p0) ? sizeof(extra) : sizeof(p0));
|
||||
return touch_buf(buf0) + touch_buf(buf1) + touch_word(extra, 0);
|
||||
}
|
||||
|
||||
template <typename Input, typename... Extra>
|
||||
auto make_keys(Input alpha, Extra && ...extra)
|
||||
{
|
||||
auto keys = dpf::make_dpf(alpha, std::uint64_t{0x9e3779b97f4a7c15ull},
|
||||
std::forward<Extra>(extra)...);
|
||||
using keys_t = std::decay_t<decltype(keys)>;
|
||||
return std::make_shared<keys_t>(std::move(keys));
|
||||
}
|
||||
|
||||
template <typename Input, typename Output, typename... Extra>
|
||||
auto make_out_keys(Input alpha, Output beta, Extra && ...extra)
|
||||
{
|
||||
auto keys = dpf::make_dpf(alpha, beta, std::forward<Extra>(extra)...);
|
||||
using keys_t = std::decay_t<decltype(keys)>;
|
||||
return std::make_shared<keys_t>(std::move(keys));
|
||||
}
|
||||
|
||||
const char kUsage[] =
|
||||
"profile_eval [--list] [--family F] [--slice S] [--case NAME] [--tier std]\n"
|
||||
" [--repeat N=20] [--warmup W=2]\n"
|
||||
"Slices: keygen, interval, interval-shape, sequence, sequence-algo,\n"
|
||||
" memoizer, bits, gf, inner-product, helpers, defer.\n"
|
||||
"Times sequence and interval evaluation on both party keys.\n";
|
||||
|
||||
template <typename Input, typename KeysPtr>
|
||||
void add_interval_at(std::vector<work> & works, const char * slice,
|
||||
const std::string & name, Input from, Input to, const KeysPtr & keys)
|
||||
{
|
||||
const auto items = static_cast<std::uint64_t>(to) - static_cast<std::uint64_t>(from) + 1;
|
||||
works.push_back(profile::make_work("eval", slice, name, items, [keys, from, to] {
|
||||
return interval_both(*keys, from, to);
|
||||
}));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char ** argv)
|
||||
{
|
||||
const auto opt = profile::parse_args(argc, argv, 20, 2, kUsage);
|
||||
|
||||
using u8 = std::uint8_t;
|
||||
using u16 = std::uint16_t;
|
||||
using u32 = std::uint32_t;
|
||||
|
||||
const auto k8 = make_keys(u8{40});
|
||||
const auto k16 = make_keys(u16{1512});
|
||||
const auto k16v = make_keys(u16{1512}, dpf::verifiable{});
|
||||
const auto k32 = make_keys(u32{1002048u});
|
||||
const auto k32v = make_keys(u32{1002048u}, dpf::verifiable{});
|
||||
|
||||
std::vector<work> works;
|
||||
|
||||
works.push_back(profile::make_work("eval", "keygen", "keygen_u8", 1, [] {
|
||||
return hash_roots(*make_keys(u8{40}));
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "keygen", "keygen_u16", 1, [] {
|
||||
return hash_roots(*make_keys(u16{1512}));
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "keygen", "keygen_u32", 1, [] {
|
||||
return hash_roots(*make_keys(u32{0x01000000u}));
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "keygen", "keygen_u32_verifiable", 1, [] {
|
||||
return hash_roots(*make_keys(u32{0x01000000u}, dpf::verifiable{}));
|
||||
}));
|
||||
|
||||
const auto add_gf = [&](const char * name, auto beta) {
|
||||
using out = std::decay_t<decltype(beta)>;
|
||||
works.push_back(profile::make_work("eval", "gf",
|
||||
std::string("keygen_") + name, 1, [beta] {
|
||||
return hash_roots(*make_out_keys(u8{40}, beta));
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "gf",
|
||||
std::string("keygen_") + name + "_verifiable", 1, [beta] {
|
||||
return hash_roots(*make_out_keys(u8{40}, beta, dpf::verifiable{}));
|
||||
}));
|
||||
const auto keys = make_out_keys(u8{40}, beta);
|
||||
if constexpr (dpf::utils::is_packed_subbyte_v<out>)
|
||||
{
|
||||
constexpr unsigned bits = dpf::utils::packed_lane_bits_v<out>;
|
||||
works.push_back(profile::make_work("eval", "gf",
|
||||
std::string("interval_") + name + "_u8_L256", 256,
|
||||
[keys, bits] {
|
||||
return interval_packed(*keys, u8{0}, u8{255}, bits);
|
||||
}));
|
||||
}
|
||||
else
|
||||
{
|
||||
add_interval_at(works, "gf", std::string("interval_") + name + "_u8_L256",
|
||||
u8{0}, u8{255}, keys);
|
||||
}
|
||||
};
|
||||
add_gf("gf2", dpf::gf2{1});
|
||||
add_gf("gf22", dpf::gf22{3});
|
||||
add_gf("gf24", dpf::gf24{0xa});
|
||||
add_gf("gf28", dpf::gf28{0x1b});
|
||||
add_gf("gf216", dpf::gf216{0x2d});
|
||||
add_gf("gf232", dpf::gf232{0x90200001u});
|
||||
add_gf("gf264", dpf::gf264{0x11});
|
||||
|
||||
const auto add_lengths = [&](auto from0, auto keys, const char * width,
|
||||
std::initializer_list<std::uint64_t> lengths) {
|
||||
using input = decltype(from0);
|
||||
for (const std::uint64_t n : lengths)
|
||||
{
|
||||
const input from = from0;
|
||||
const input to = static_cast<input>(from + static_cast<input>(n - 1));
|
||||
add_interval_at(works, "interval",
|
||||
std::string("interval_") + width + "_L" + std::to_string(n),
|
||||
from, to, keys);
|
||||
}
|
||||
};
|
||||
add_lengths(u8{0}, k8, "u8", {1, 16, 64, 256});
|
||||
add_lengths(u16{1000}, k16, "u16", {1, 16, 256, 1024, 4096});
|
||||
add_lengths(u32{1000000}, k32, "u32", {1, 16, 64, 256, 1024, 4096, 16384});
|
||||
|
||||
add_interval_at(works, "interval-shape", "interval_u32_L256_unaligned",
|
||||
u32{1000003}, u32{1000258}, k32);
|
||||
add_interval_at(works, "interval-shape", "interval_u32_L4096_unaligned",
|
||||
u32{1000003}, u32{1004098}, k32);
|
||||
|
||||
const auto add_reuse = [&](u32 from, u32 to, const char * name) {
|
||||
using buf0_t = std::decay_t<decltype(dpf::make_output_buffer_for_interval(
|
||||
std::get<0>(*k32), from, to))>;
|
||||
using buf1_t = std::decay_t<decltype(dpf::make_output_buffer_for_interval(
|
||||
std::get<1>(*k32), from, to))>;
|
||||
auto buf0 = std::make_shared<buf0_t>(
|
||||
dpf::make_output_buffer_for_interval(std::get<0>(*k32), from, to));
|
||||
auto buf1 = std::make_shared<buf1_t>(
|
||||
dpf::make_output_buffer_for_interval(std::get<1>(*k32), from, to));
|
||||
const auto items = static_cast<std::uint64_t>(to - from + 1);
|
||||
works.push_back(profile::make_work("eval", "interval-shape", name, items,
|
||||
[k32, buf0, buf1, from, to] {
|
||||
return interval_reuse(*k32, from, to, *buf0, *buf1);
|
||||
}));
|
||||
};
|
||||
add_reuse(1000000, 1000255, "interval_u32_L256_reuse");
|
||||
add_reuse(1000000, 1004095, "interval_u32_L4096_reuse");
|
||||
|
||||
works.push_back(profile::make_work("eval", "interval-shape",
|
||||
"interval_u16_L1024_verifiable", 1024, [k16v] {
|
||||
return interval_prove(*k16v, u16{1000}, u16{2023});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "interval-shape",
|
||||
"interval_u32_L256_verifiable", 256, [k32v] {
|
||||
return interval_prove(*k32v, u32{1000000}, u32{1000255});
|
||||
}));
|
||||
|
||||
const auto add_seq = [&](auto keys, const char * width, const char * shape,
|
||||
auto pts) {
|
||||
const auto n = static_cast<std::uint64_t>(pts->size());
|
||||
const std::string name = std::string("sequence_") + width + "_" + shape
|
||||
+ "_" + std::to_string(n);
|
||||
works.push_back(profile::make_work("eval", "sequence", name, n,
|
||||
[keys, pts] {
|
||||
return sequence_both(*keys, *pts, false);
|
||||
}));
|
||||
};
|
||||
const std::size_t seq32[] = {8, 32, 64, 128, 256, 512, 1024, 2048};
|
||||
for (const std::size_t n : seq32)
|
||||
{
|
||||
add_seq(k32, "u32", "cluster", clustered<u32>(1u << 20, n));
|
||||
const u32 step = n <= 64 ? u32{1u << 20} : u32{1u << 12};
|
||||
add_seq(k32, "u32", "stride", strided<u32>(16u, step, n));
|
||||
}
|
||||
const std::size_t seq16[] = {8, 64, 256, 1024};
|
||||
for (const std::size_t n : seq16)
|
||||
add_seq(k16, "u16", "cluster", clustered<u16>(1000, n));
|
||||
const std::size_t seq8[] = {8, 32, 64};
|
||||
for (const std::size_t n : seq8)
|
||||
{
|
||||
add_seq(k8, "u8", "cluster", clustered<u8>(0, n));
|
||||
add_seq(k8, "u8", "stride", strided<u8>(0, 3, n));
|
||||
}
|
||||
|
||||
const auto c256 = clustered<u32>(1u << 20, 256);
|
||||
const auto s64 = strided<u32>(16u, 1u << 20, 64);
|
||||
const auto c16s = clustered<u16>(1000, 64);
|
||||
const auto c32s = clustered<u32>(1u << 20, 64);
|
||||
|
||||
works.push_back(profile::make_work("eval", "sequence-algo",
|
||||
"sequence_u32_cluster_256_output_only", 256, [k32, c256] {
|
||||
return sequence_both(*k32, *c256, true);
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "sequence-algo",
|
||||
"sequence_u32_cluster_256_breadth", 256, [k32, c256] {
|
||||
return sequence_breadth(*k32, *c256);
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "sequence-algo",
|
||||
"sequence_u32_stride_64_breadth", 64, [k32, s64] {
|
||||
return sequence_breadth(*k32, *s64);
|
||||
}));
|
||||
|
||||
using recipe0_t = std::decay_t<decltype(dpf::make_sequence_recipe(
|
||||
std::get<0>(*k32), c256->begin(), c256->end()))>;
|
||||
using recipe1_t = std::decay_t<decltype(dpf::make_sequence_recipe(
|
||||
std::get<1>(*k32), c256->begin(), c256->end()))>;
|
||||
auto recipe0 = std::make_shared<recipe0_t>(dpf::make_sequence_recipe(
|
||||
std::get<0>(*k32), c256->begin(), c256->end()));
|
||||
auto recipe1 = std::make_shared<recipe1_t>(dpf::make_sequence_recipe(
|
||||
std::get<1>(*k32), c256->begin(), c256->end()));
|
||||
works.push_back(profile::make_work("eval", "sequence-algo",
|
||||
"sequence_u32_cluster_256_recipe", 256, [k32, recipe0, recipe1] {
|
||||
return sequence_recipe(*k32, *recipe0, *recipe1);
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "sequence-algo",
|
||||
"sequence_u16_cluster_64_verifiable", 64, [k16v, c16s] {
|
||||
return sequence_prove(*k16v, *c16s);
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "sequence-algo",
|
||||
"sequence_u32_cluster_64_verifiable", 64, [k32v, c32s] {
|
||||
return sequence_prove(*k32v, *c32s);
|
||||
}));
|
||||
|
||||
// --- memoizer: built-in reuse vs hand-rolled pointwise / fresh memo ---
|
||||
{
|
||||
using u32 = std::uint32_t;
|
||||
const u32 from = 1000000u;
|
||||
const u32 to = 1000255u; // L256
|
||||
const auto items = static_cast<std::uint64_t>(to - from + 1);
|
||||
const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32));
|
||||
using dpf_t = std::decay_t<decltype(std::get<0>(*k32))>;
|
||||
using node_t = typename dpf_t::interior_node;
|
||||
const auto out_elem = sizeof(std::uint64_t);
|
||||
const auto logical = items * out_elem * 2;
|
||||
// Match `basic_interval_memoizer` capacity (leaf nodes, not output slots).
|
||||
const auto leaf_nodes =
|
||||
dpf::utils::get_leafnodes_in_output_interval<dpf_t>(from, to);
|
||||
const auto slots = leaf_nodes == 0 ? std::size_t{1} : leaf_nodes;
|
||||
const auto pivot = std::max((slots >> 1) + (slots & 1) - 1,
|
||||
(slots + 6) >> 2);
|
||||
const auto memo_nodes = pivot + ((slots + 2) >> 1);
|
||||
const auto memo_bytes = 2ull * memo_nodes * sizeof(node_t);
|
||||
|
||||
auto memo0 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to))>>(
|
||||
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to));
|
||||
auto memo1 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to))>>(
|
||||
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to));
|
||||
|
||||
works.push_back(profile::make_work("eval", "memoizer",
|
||||
"memo_interval_reuse", items, [k32, memo0, memo1, from, to, prep, memo_bytes, logical] {
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0);
|
||||
auto b = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1);
|
||||
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||||
// Warm reuse: second pass on the same memoizers.
|
||||
auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, *memo0);
|
||||
auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, *memo1);
|
||||
s = s + touch_buf(a2.first) + touch_buf(b2.first);
|
||||
return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2);
|
||||
});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "memoizer",
|
||||
"memo_interval_fresh", items, [k32, from, to, prep, logical, memo_bytes] {
|
||||
return profile::with_prg([&] {
|
||||
auto m0 = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to);
|
||||
auto m1 = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to);
|
||||
auto a = dpf::eval_interval(std::get<0>(*k32), from, to, m0);
|
||||
auto b = dpf::eval_interval(std::get<1>(*k32), from, to, m1);
|
||||
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||||
auto m0b = dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to);
|
||||
auto m1b = dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to);
|
||||
auto a2 = dpf::eval_interval(std::get<0>(*k32), from, to, m0b);
|
||||
auto b2 = dpf::eval_interval(std::get<1>(*k32), from, to, m1b);
|
||||
s = s + touch_buf(a2.first) + touch_buf(b2.first);
|
||||
return profile::with_costs(s, prep, memo_bytes + s.out_bytes, logical * 2);
|
||||
});
|
||||
}));
|
||||
|
||||
const auto pts = clustered<u32>(1u << 20, 64);
|
||||
auto path0 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_basic_path_memoizer(std::get<0>(*k32)))>>(
|
||||
dpf::make_basic_path_memoizer(std::get<0>(*k32)));
|
||||
auto path1 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_basic_path_memoizer(std::get<1>(*k32)))>>(
|
||||
dpf::make_basic_path_memoizer(std::get<1>(*k32)));
|
||||
works.push_back(profile::make_work("eval", "memoizer",
|
||||
"memo_path_reuse", 64, [k32, pts, path0, path1, prep] {
|
||||
return profile::with_prg([&] {
|
||||
std::uint64_t h = 0;
|
||||
for (auto x : *pts)
|
||||
{
|
||||
h ^= static_cast<std::uint64_t>(
|
||||
(*dpf::eval_point(std::get<0>(*k32), x, *path0)).raw());
|
||||
h ^= static_cast<std::uint64_t>(
|
||||
(*dpf::eval_point(std::get<1>(*k32), x, *path1)).raw());
|
||||
}
|
||||
auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2);
|
||||
return profile::with_costs(s, prep,
|
||||
2 * sizeof(node_t) * 32, 64 * sizeof(std::uint64_t) * 2);
|
||||
});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "memoizer",
|
||||
"memo_path_pointwise", 64, [k32, pts, prep] {
|
||||
return profile::with_prg([&] {
|
||||
std::uint64_t h = 0;
|
||||
for (auto x : *pts)
|
||||
{
|
||||
h ^= static_cast<std::uint64_t>(
|
||||
(*dpf::eval_point(std::get<0>(*k32), x)).raw());
|
||||
h ^= static_cast<std::uint64_t>(
|
||||
(*dpf::eval_point(std::get<1>(*k32), x)).raw());
|
||||
}
|
||||
auto s = touch_word(h, 64 * sizeof(std::uint64_t) * 2);
|
||||
return profile::with_costs(s, prep, 0,
|
||||
64 * sizeof(std::uint64_t) * 2);
|
||||
});
|
||||
}));
|
||||
|
||||
using recipe0_t = std::decay_t<decltype(dpf::make_sequence_recipe(
|
||||
std::get<0>(*k32), pts->begin(), pts->end()))>;
|
||||
using recipe1_t = std::decay_t<decltype(dpf::make_sequence_recipe(
|
||||
std::get<1>(*k32), pts->begin(), pts->end()))>;
|
||||
auto rec0 = std::make_shared<recipe0_t>(dpf::make_sequence_recipe(
|
||||
std::get<0>(*k32), pts->begin(), pts->end()));
|
||||
auto rec1 = std::make_shared<recipe1_t>(dpf::make_sequence_recipe(
|
||||
std::get<1>(*k32), pts->begin(), pts->end()));
|
||||
auto smemo0 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0))>>(
|
||||
dpf::make_inplace_reversing_sequence_memoizer(std::get<0>(*k32), *rec0));
|
||||
auto smemo1 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1))>>(
|
||||
dpf::make_inplace_reversing_sequence_memoizer(std::get<1>(*k32), *rec1));
|
||||
works.push_back(profile::make_work("eval", "memoizer",
|
||||
"memo_recipe_reuse", 64, [k32, rec0, rec1, smemo0, smemo1, prep] {
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0);
|
||||
auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1);
|
||||
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||||
auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, *smemo0);
|
||||
auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, *smemo1);
|
||||
s = s + touch_buf(a2.first) + touch_buf(b2.first);
|
||||
return profile::with_costs(s, prep, s.out_bytes,
|
||||
64 * sizeof(std::uint64_t) * 4);
|
||||
});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "memoizer",
|
||||
"memo_recipe_fresh", 64, [k32, rec0, rec1, prep] {
|
||||
return profile::with_prg([&] {
|
||||
auto m0 = dpf::make_inplace_reversing_sequence_memoizer(
|
||||
std::get<0>(*k32), *rec0);
|
||||
auto m1 = dpf::make_inplace_reversing_sequence_memoizer(
|
||||
std::get<1>(*k32), *rec1);
|
||||
auto a = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0);
|
||||
auto b = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1);
|
||||
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||||
auto m0b = dpf::make_inplace_reversing_sequence_memoizer(
|
||||
std::get<0>(*k32), *rec0);
|
||||
auto m1b = dpf::make_inplace_reversing_sequence_memoizer(
|
||||
std::get<1>(*k32), *rec1);
|
||||
auto a2 = dpf::eval_sequence(std::get<0>(*k32), *rec0, m0b);
|
||||
auto b2 = dpf::eval_sequence(std::get<1>(*k32), *rec1, m1b);
|
||||
s = s + touch_buf(a2.first) + touch_buf(b2.first);
|
||||
return profile::with_costs(s, prep, s.out_bytes,
|
||||
64 * sizeof(std::uint64_t) * 4);
|
||||
});
|
||||
}));
|
||||
}
|
||||
|
||||
// --- bits: built-in iterators vs hand-rolled scans (u8 domain) ---
|
||||
{
|
||||
using input_type = std::uint8_t;
|
||||
using output_type = dpf::bit;
|
||||
const input_type alpha = 40;
|
||||
auto bit_keys = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_dpf(alpha, output_type::one))>>(
|
||||
dpf::make_dpf(alpha, output_type::one));
|
||||
using dpf_type = std::decay_t<decltype(std::get<0>(*bit_keys))>;
|
||||
auto memo0 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_basic_full_memoizer<dpf_type>())>>(
|
||||
dpf::make_basic_full_memoizer<dpf_type>());
|
||||
auto memo1 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_basic_full_memoizer<dpf_type>())>>(
|
||||
dpf::make_basic_full_memoizer<dpf_type>());
|
||||
auto full0 = dpf::eval_full(std::get<0>(*bit_keys), *memo0);
|
||||
auto full1 = dpf::eval_full(std::get<1>(*bit_keys), *memo1);
|
||||
auto buf0 = std::make_shared<std::decay_t<decltype(full0.first)>>(std::move(full0.first));
|
||||
auto buf1 = std::make_shared<std::decay_t<decltype(full1.first)>>(std::move(full1.first));
|
||||
auto iter0 = std::make_shared<std::decay_t<decltype(full0.second)>>(std::move(full0.second));
|
||||
auto iter1 = std::make_shared<std::decay_t<decltype(full1.second)>>(std::move(full1.second));
|
||||
const auto leaf_nodes = static_cast<std::uint64_t>(1) << dpf_type::depth;
|
||||
const auto bit_items = static_cast<std::uint64_t>(buf0->size());
|
||||
const auto prep = sizeof(std::get<0>(*bit_keys)) + sizeof(std::get<1>(*bit_keys));
|
||||
|
||||
works.push_back(profile::make_work("eval", "bits",
|
||||
"bits_advice_builtin", leaf_nodes, [memo0, memo1, prep, leaf_nodes] {
|
||||
std::uint64_t h = 0;
|
||||
std::uint64_t n = 0;
|
||||
for (auto b : dpf::advice_bits_of(*memo0))
|
||||
{
|
||||
h = (h << 1) ^ (b ? 1u : 0u);
|
||||
++n;
|
||||
}
|
||||
for (auto b : dpf::advice_bits_of(*memo1))
|
||||
h ^= (b ? 1u : 0u);
|
||||
auto s = touch_word(h ^ n, n);
|
||||
return profile::with_costs(s, prep, 0, leaf_nodes);
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "bits",
|
||||
"bits_advice_handroll", leaf_nodes, [memo0, memo1, prep, leaf_nodes] {
|
||||
std::uint64_t h = 0;
|
||||
std::uint64_t n = 0;
|
||||
const auto * p0 = memo0->begin();
|
||||
const auto * e0 = memo0->end();
|
||||
for (; p0 != e0; ++p0)
|
||||
{
|
||||
const auto * bytes = reinterpret_cast<const char *>(p0);
|
||||
h = (h << 1) ^ (bytes[0] & 1u);
|
||||
++n;
|
||||
}
|
||||
const auto * p1 = memo1->begin();
|
||||
const auto * e1 = memo1->end();
|
||||
for (; p1 != e1; ++p1)
|
||||
{
|
||||
const auto * bytes = reinterpret_cast<const char *>(p1);
|
||||
h ^= (bytes[0] & 1u);
|
||||
}
|
||||
auto s = touch_word(h ^ n, n);
|
||||
return profile::with_costs(s, prep, 0, leaf_nodes);
|
||||
}));
|
||||
|
||||
works.push_back(profile::make_work("eval", "bits",
|
||||
"bits_setbit_builtin", bit_items, [iter0, iter1, prep] {
|
||||
std::uint64_t h = 0;
|
||||
std::uint64_t n = 0;
|
||||
for (auto i : dpf::indices_set_in(*iter0))
|
||||
{
|
||||
h ^= static_cast<std::uint64_t>(i) + 0x9e3779b97f4a7c15ull;
|
||||
++n;
|
||||
}
|
||||
for (auto i : dpf::indices_set_in(*iter1))
|
||||
h ^= static_cast<std::uint64_t>(i);
|
||||
auto s = touch_word(h ^ n, n * sizeof(std::size_t));
|
||||
return profile::with_costs(s, prep, 0, n * sizeof(std::size_t));
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "bits",
|
||||
"bits_setbit_handroll", bit_items, [buf0, buf1, prep] {
|
||||
std::uint64_t h = 0;
|
||||
std::uint64_t n = 0;
|
||||
const auto scan = [&](const auto & buf) {
|
||||
for (std::size_t i = 0; i < buf.size(); ++i)
|
||||
{
|
||||
if (static_cast<bool>(buf[i]))
|
||||
{
|
||||
h ^= i + 0x9e3779b97f4a7c15ull;
|
||||
++n;
|
||||
}
|
||||
}
|
||||
};
|
||||
scan(*buf0);
|
||||
scan(*buf1);
|
||||
auto s = touch_word(h ^ n, n * sizeof(std::size_t));
|
||||
return profile::with_costs(s, prep, 0, n * sizeof(std::size_t));
|
||||
}));
|
||||
|
||||
works.push_back(profile::make_work("eval", "bits",
|
||||
"bits_parallel_builtin", bit_items, [buf0, buf1, prep, bit_items] {
|
||||
std::uint64_t h = 0;
|
||||
std::uint64_t n = 0;
|
||||
for (auto word : dpf::batch_of(*buf0, *buf1))
|
||||
{
|
||||
h ^= static_cast<std::uint64_t>(word[0])
|
||||
^ static_cast<std::uint64_t>(word[1]);
|
||||
++n;
|
||||
}
|
||||
auto s = touch_word(h ^ n, n * sizeof(std::uint64_t));
|
||||
return profile::with_costs(s, prep, 0, bit_items);
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "bits",
|
||||
"bits_parallel_handroll", bit_items, [buf0, buf1, prep, bit_items] {
|
||||
std::uint64_t h = 0;
|
||||
const std::size_t n = std::min(buf0->size(), buf1->size());
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
h ^= (static_cast<bool>((*buf0)[i]) ? 1ull : 0ull)
|
||||
^ (static_cast<bool>((*buf1)[i]) ? 2ull : 0ull);
|
||||
auto s = touch_word(h ^ n, n);
|
||||
return profile::with_costs(s, prep, 0, bit_items);
|
||||
}));
|
||||
}
|
||||
|
||||
// --- inner-product: built-in vs hand-rolled interval + dot ---
|
||||
{
|
||||
using u32 = std::uint32_t;
|
||||
const u32 from = 1000000u;
|
||||
const u32 to = 1000255u;
|
||||
const auto items = static_cast<std::uint64_t>(to - from + 1);
|
||||
auto weights = std::make_shared<std::vector<std::uint64_t>>(items);
|
||||
for (std::size_t i = 0; i < items; ++i)
|
||||
(*weights)[i] = 0x9e3779b97f4a7c15ull * (i + 1);
|
||||
const auto prep = sizeof(std::get<0>(*k32)) + sizeof(std::get<1>(*k32));
|
||||
auto ip_memo0 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to))>>(
|
||||
dpf::make_basic_interval_memoizer(std::get<0>(*k32), from, to));
|
||||
auto ip_memo1 = std::make_shared<std::decay_t<decltype(
|
||||
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to))>>(
|
||||
dpf::make_basic_interval_memoizer(std::get<1>(*k32), from, to));
|
||||
|
||||
works.push_back(profile::make_work("eval", "inner-product",
|
||||
"ip_interval_builtin", items,
|
||||
[k32, weights, from, to, prep, items, ip_memo0, ip_memo1] {
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_inner_product(std::get<0>(*k32), from, to,
|
||||
*weights, *ip_memo0);
|
||||
auto b = dpf::eval_inner_product(std::get<1>(*k32), from, to,
|
||||
*weights, *ip_memo1);
|
||||
std::uint64_t ha = 0;
|
||||
std::uint64_t hb = 0;
|
||||
std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a));
|
||||
std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b));
|
||||
auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b));
|
||||
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
|
||||
});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "inner-product",
|
||||
"ip_interval_handroll", items, [k32, weights, from, to, prep, items] {
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_interval(std::get<0>(*k32), from, to);
|
||||
auto b = dpf::eval_interval(std::get<1>(*k32), from, to);
|
||||
std::uint64_t dot0 = 0;
|
||||
std::uint64_t dot1 = 0;
|
||||
const std::size_t n = std::min(items, a.first.size());
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
{
|
||||
dot0 += static_cast<std::uint64_t>(a.first[i].raw()) * (*weights)[i];
|
||||
dot1 += static_cast<std::uint64_t>(b.first[i].raw()) * (*weights)[i];
|
||||
}
|
||||
auto s = touch_word(dot0 ^ dot1, a.first.size() * sizeof(a.first[0])
|
||||
+ b.first.size() * sizeof(b.first[0]));
|
||||
return profile::with_costs(s, prep, s.out_bytes,
|
||||
items * sizeof(std::uint64_t) * 2);
|
||||
});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "inner-product",
|
||||
"ip_interval_paired", items,
|
||||
[k32, weights, from, to, prep, items] {
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_inner_product(dpf::paired,
|
||||
std::get<0>(*k32), from, to, *weights);
|
||||
auto b = dpf::eval_inner_product(dpf::paired,
|
||||
std::get<1>(*k32), from, to, *weights);
|
||||
std::uint64_t ha = 0;
|
||||
std::uint64_t hb = 0;
|
||||
std::memcpy(&ha, &a, sizeof(ha) < sizeof(a) ? sizeof(ha) : sizeof(a));
|
||||
std::memcpy(&hb, &b, sizeof(hb) < sizeof(b) ? sizeof(hb) : sizeof(b));
|
||||
auto s = touch_word(ha ^ hb, sizeof(a) + sizeof(b));
|
||||
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
|
||||
});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "inner-product",
|
||||
"ip_interval_columns", items,
|
||||
[k32, weights, from, to, prep, items] {
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_inner_product(dpf::columns,
|
||||
std::get<0>(*k32), from, to, std::tie(*weights));
|
||||
auto b = dpf::eval_inner_product(dpf::columns,
|
||||
std::get<1>(*k32), from, to, std::tie(*weights));
|
||||
std::uint64_t ha = 0;
|
||||
std::uint64_t hb = 0;
|
||||
const auto & a0 = std::get<0>(a);
|
||||
const auto & b0 = std::get<0>(b);
|
||||
std::memcpy(&ha, &a0,
|
||||
sizeof(ha) < sizeof(a0) ? sizeof(ha) : sizeof(a0));
|
||||
std::memcpy(&hb, &b0,
|
||||
sizeof(hb) < sizeof(b0) ? sizeof(hb) : sizeof(b0));
|
||||
auto s = touch_word(ha ^ hb, sizeof(a0) + sizeof(b0));
|
||||
return profile::with_costs(s, prep, 0, items * sizeof(std::uint64_t));
|
||||
});
|
||||
}));
|
||||
}
|
||||
|
||||
// --- helpers: point vs interval vs full (representative widths) ---
|
||||
{
|
||||
using u16 = std::uint16_t;
|
||||
const auto prep16 = sizeof(std::get<0>(*k16)) + sizeof(std::get<1>(*k16));
|
||||
works.push_back(profile::make_work("eval", "helpers",
|
||||
"helper_point_u16", 1, [k16, prep16] {
|
||||
return profile::with_prg([&] {
|
||||
auto a = *dpf::eval_point(std::get<0>(*k16), u16{1512});
|
||||
auto b = *dpf::eval_point(std::get<1>(*k16), u16{1512});
|
||||
auto s = touch_word(
|
||||
static_cast<std::uint64_t>(a.raw()) ^ static_cast<std::uint64_t>(b.raw()),
|
||||
sizeof(a) + sizeof(b));
|
||||
return profile::with_costs(s, prep16, 0, sizeof(a) + sizeof(b));
|
||||
});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "helpers",
|
||||
"helper_interval_u16_L256", 256, [k16, prep16] {
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_interval(std::get<0>(*k16), u16{1000}, u16{1255});
|
||||
auto b = dpf::eval_interval(std::get<1>(*k16), u16{1000}, u16{1255});
|
||||
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||||
return profile::with_costs(s, prep16, s.out_bytes,
|
||||
256 * sizeof(std::uint64_t) * 2);
|
||||
});
|
||||
}));
|
||||
works.push_back(profile::make_work("eval", "helpers",
|
||||
"helper_full_u8", 256, [k8] {
|
||||
const auto prep = sizeof(std::get<0>(*k8)) + sizeof(std::get<1>(*k8));
|
||||
return profile::with_prg([&] {
|
||||
auto a = dpf::eval_full(std::get<0>(*k8));
|
||||
auto b = dpf::eval_full(std::get<1>(*k8));
|
||||
auto s = touch_buf(a.first) + touch_buf(b.first);
|
||||
return profile::with_costs(s, prep, s.out_bytes, s.out_bytes);
|
||||
});
|
||||
}));
|
||||
}
|
||||
|
||||
// --- defer: pre-assign full-domain expand vs rotate-after-assign ---
|
||||
{
|
||||
using u8 = std::uint8_t;
|
||||
using out_t = std::uint64_t;
|
||||
const u8 alpha = 40;
|
||||
const out_t beta = 0x9e3779b97f4a7c15ull;
|
||||
|
||||
works.push_back(profile::make_work("eval", "defer",
|
||||
"defer_full_u8_expand", 256, [beta] {
|
||||
return profile::with_prg([&] {
|
||||
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
|
||||
auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys));
|
||||
auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys));
|
||||
auto d0 = dpf::defer_eval_full(std::get<0>(keys), b0);
|
||||
auto d1 = dpf::defer_eval_full(std::get<1>(keys), b1);
|
||||
(void)d0;
|
||||
(void)d1;
|
||||
return touch_buf(b0) + touch_buf(b1);
|
||||
});
|
||||
}));
|
||||
|
||||
works.push_back(profile::make_work("eval", "defer",
|
||||
"defer_interval_u8_L64_expand", 64, [beta] {
|
||||
return profile::with_prg([&] {
|
||||
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
|
||||
auto b0 = dpf::make_output_buffer_for_full(std::get<0>(keys));
|
||||
auto b1 = dpf::make_output_buffer_for_full(std::get<1>(keys));
|
||||
auto d0 = dpf::defer_eval_interval(std::get<0>(keys),
|
||||
u8{16}, u8{79}, b0);
|
||||
auto d1 = dpf::defer_eval_interval(std::get<1>(keys),
|
||||
u8{16}, u8{79}, b1);
|
||||
(void)d0;
|
||||
(void)d1;
|
||||
return touch_buf(b0) + touch_buf(b1);
|
||||
});
|
||||
}));
|
||||
|
||||
// Expand once against stable key storage, assign, then time .get().
|
||||
{
|
||||
using keys_t = std::decay_t<decltype(
|
||||
dpf::make_dpf(dpf::wildcard_value<u8>{}, beta))>;
|
||||
auto keys = std::make_shared<keys_t>(
|
||||
dpf::make_dpf(dpf::wildcard_value<u8>{}, beta));
|
||||
using buf0_t = std::decay_t<decltype(
|
||||
dpf::make_output_buffer_for_full(std::get<0>(*keys)))>;
|
||||
using buf1_t = std::decay_t<decltype(
|
||||
dpf::make_output_buffer_for_full(std::get<1>(*keys)))>;
|
||||
auto b0 = std::make_shared<buf0_t>(
|
||||
dpf::make_output_buffer_for_full(std::get<0>(*keys)));
|
||||
auto b1 = std::make_shared<buf1_t>(
|
||||
dpf::make_output_buffer_for_full(std::get<1>(*keys)));
|
||||
using def0_t = std::decay_t<decltype(
|
||||
dpf::defer_eval_full(std::get<0>(*keys), *b0))>;
|
||||
using def1_t = std::decay_t<decltype(
|
||||
dpf::defer_eval_full(std::get<1>(*keys), *b1))>;
|
||||
auto def0 = std::make_shared<def0_t>(
|
||||
dpf::defer_eval_full(std::get<0>(*keys), *b0));
|
||||
auto def1 = std::make_shared<def1_t>(
|
||||
dpf::defer_eval_full(std::get<1>(*keys), *b1));
|
||||
{
|
||||
auto & k0 = std::get<0>(*keys);
|
||||
auto & k1 = std::get<1>(*keys);
|
||||
const u8 a0 = 0x12;
|
||||
const u8 a1 = static_cast<u8>(alpha - a0);
|
||||
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
|
||||
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
|
||||
k0.offset_x.reconstruct(sh1);
|
||||
k1.offset_x.reconstruct(sh0);
|
||||
}
|
||||
works.push_back(profile::make_work("eval", "defer",
|
||||
"defer_full_u8_get", 256, [keys, def0, def1, b0, b1] {
|
||||
(void)keys;
|
||||
auto v0 = def0->get();
|
||||
auto v1 = def1->get();
|
||||
std::uint64_t sink = 0;
|
||||
for (auto it = std::begin(v0); it != std::end(v0); ++it)
|
||||
sink ^= static_cast<std::uint64_t>((*it).raw());
|
||||
for (auto it = std::begin(v1); it != std::end(v1); ++it)
|
||||
sink ^= static_cast<std::uint64_t>((*it).raw());
|
||||
return touch_word(sink, b0->size() * sizeof((*b0)[0])
|
||||
+ b1->size() * sizeof((*b1)[0]));
|
||||
}));
|
||||
}
|
||||
|
||||
works.push_back(profile::make_work("eval", "defer",
|
||||
"eager_full_u8_after_assign", 256, [beta, alpha] {
|
||||
return profile::with_prg([&] {
|
||||
auto keys = dpf::make_dpf(dpf::wildcard_value<u8>{}, beta);
|
||||
auto & k0 = std::get<0>(keys);
|
||||
auto & k1 = std::get<1>(keys);
|
||||
const u8 a0 = 0x12;
|
||||
const u8 a1 = static_cast<u8>(alpha - a0);
|
||||
const auto sh0 = k0.offset_x.compute_and_get_share(a0);
|
||||
const auto sh1 = k1.offset_x.compute_and_get_share(a1);
|
||||
k0.offset_x.reconstruct(sh1);
|
||||
k1.offset_x.reconstruct(sh0);
|
||||
auto a = dpf::eval_full(k0);
|
||||
auto b = dpf::eval_full(k1);
|
||||
return touch_buf(a.first) + touch_buf(b.first);
|
||||
});
|
||||
}));
|
||||
}
|
||||
|
||||
return profile::run_works(opt, works);
|
||||
}
|
||||
509
test/profile/grotto_profile.cpp
Normal file
509
test/profile/grotto_profile.cpp
Normal file
|
|
@ -0,0 +1,509 @@
|
|||
/// @file test/profile/grotto_profile.cpp
|
||||
/// @brief Workload for grotto evaluation: tables, offset walks, prefix parity.
|
||||
///
|
||||
/// Key generation is a separate case from the eval that consumes those keys.
|
||||
/// Table cases probe the domain once, then time only inputs that evaluate.
|
||||
/// `items` is the number of table evaluations inside one sample, or 1 for a
|
||||
/// single offset / prefix walk (both parties).
|
||||
///
|
||||
/// profile_grotto --list
|
||||
/// profile_grotto --case window_gelu_p16 --repeat 100
|
||||
///
|
||||
/// Profile-guided build, separate from COVERAGE:
|
||||
/// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release
|
||||
/// cmake --build build-pgo --target profile_grotto
|
||||
/// build-pgo/bin/profile_grotto --repeat 40 --warmup 2
|
||||
/// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release
|
||||
/// cmake --build build-pgo --target profile_grotto
|
||||
|
||||
#include "harness.hpp"
|
||||
|
||||
#include "dpf.hpp"
|
||||
#include "grotto/closed_form.hpp"
|
||||
#include "grotto/exact_steps.hpp"
|
||||
#include "grotto/offset_horner.hpp"
|
||||
#include "grotto/offset_jet.hpp"
|
||||
#include "grotto/offset_poly.hpp"
|
||||
#include "grotto/offset_repr.hpp"
|
||||
#include "grotto/offset_twist.hpp"
|
||||
#include "grotto/prefix_parity.hpp"
|
||||
#include "grotto/range_lut.hpp"
|
||||
#include "grotto/window_lut.hpp"
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <initializer_list>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace
|
||||
{
|
||||
|
||||
using profile::sample;
|
||||
using profile::touch_arr;
|
||||
using profile::touch_vec;
|
||||
using profile::touch_word;
|
||||
using profile::work;
|
||||
|
||||
const char kUsage[] =
|
||||
"profile_grotto [--list] [--family grotto] [--slice S] [--case NAME]\n"
|
||||
" [--repeat N=30] [--warmup W=2]\n"
|
||||
"Slices: window, principal, closed, reduced, exact, horner, poly, jet,\n"
|
||||
" twist, repr, prefix, keygen.\n"
|
||||
"Times grotto table evals and offset / prefix walks.\n";
|
||||
|
||||
inline work cell(const char * slice, std::string name, std::uint64_t items,
|
||||
std::function<sample()> fn)
|
||||
{
|
||||
return profile::make_work("grotto", slice, std::move(name), items, std::move(fn));
|
||||
}
|
||||
|
||||
template <typename Enum, typename Fn>
|
||||
std::vector<std::int64_t> probe(Enum which, unsigned bits, Fn fn,
|
||||
const std::vector<std::int64_t> & cand)
|
||||
{
|
||||
std::vector<std::int64_t> ok;
|
||||
ok.reserve(cand.size());
|
||||
for (const std::int64_t raw : cand)
|
||||
{
|
||||
try
|
||||
{
|
||||
(void)fn(which, bits, raw);
|
||||
ok.push_back(raw);
|
||||
}
|
||||
catch (const std::exception &)
|
||||
{
|
||||
}
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
std::vector<std::int64_t> around_one(unsigned bits)
|
||||
{
|
||||
const std::int64_t one = std::int64_t{1} << bits;
|
||||
return {
|
||||
-4 * one, -2 * one, -one, -one / 2, -1, 0, 1,
|
||||
one / 20, one / 5, one / 3, one / 2, one, (3 * one) / 2, 2 * one, 4 * one
|
||||
};
|
||||
}
|
||||
|
||||
template <typename Fn>
|
||||
void add_table(std::vector<work> & works, const char * slice, std::string name, Fn fn,
|
||||
std::vector<std::int64_t> samples)
|
||||
{
|
||||
if (samples.empty())
|
||||
{
|
||||
std::cerr << name << " skipped: no in-domain sample\n";
|
||||
return;
|
||||
}
|
||||
auto held = std::make_shared<std::vector<std::int64_t>>(std::move(samples));
|
||||
const std::uint64_t items = static_cast<std::uint64_t>(held->size()) * 32u;
|
||||
std::function<sample()> body = [held, fn]() -> sample {
|
||||
std::uint64_t h = 0x84222325cbf29ce4ull;
|
||||
for (int rep = 0; rep < 32; ++rep)
|
||||
{
|
||||
for (const std::int64_t raw : *held)
|
||||
{
|
||||
// Multiply-add, not a plain xor: an even number of identical
|
||||
// xors cancels, and the compiler will delete the evals.
|
||||
h = (h * 0x100000001b3ull) ^ static_cast<std::uint64_t>(fn(raw))
|
||||
^ static_cast<std::uint64_t>(rep);
|
||||
}
|
||||
}
|
||||
return touch_word(h, held->size() * sizeof(std::int64_t));
|
||||
};
|
||||
works.push_back(cell(slice, std::move(name), items, std::move(body)));
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
std::vector<T> knots_of(std::size_t n, T gap)
|
||||
{
|
||||
std::vector<T> knots(n);
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
knots[i] = static_cast<T>(1 + gap * static_cast<T>(i));
|
||||
return knots;
|
||||
}
|
||||
|
||||
template <std::size_t Degree>
|
||||
std::vector<std::array<std::uint64_t, Degree + 1>> horner_rows(std::size_t n)
|
||||
{
|
||||
std::vector<std::array<std::uint64_t, Degree + 1>> rows(n);
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
for (std::size_t m = 0; m <= Degree; ++m)
|
||||
rows[i][m] = (i + 1) * (m + 3);
|
||||
return rows;
|
||||
}
|
||||
|
||||
std::vector<std::vector<std::uint64_t>> poly_rows(std::size_t n, std::size_t degree)
|
||||
{
|
||||
std::vector<std::vector<std::uint64_t>> rows(n, std::vector<std::uint64_t>(degree + 1));
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
for (std::size_t m = 0; m <= degree; ++m)
|
||||
rows[i][m] = (i + 1) * (m + 3);
|
||||
return rows;
|
||||
}
|
||||
|
||||
std::vector<std::uint64_t> coeff_col(std::size_t degree)
|
||||
{
|
||||
std::vector<std::uint64_t> c(degree + 1);
|
||||
for (std::size_t m = 0; m <= degree; ++m)
|
||||
c[m] = m + 5;
|
||||
return c;
|
||||
}
|
||||
|
||||
template <std::size_t N, typename T>
|
||||
std::array<T, N> ends_of(T start, T step)
|
||||
{
|
||||
std::array<T, N> ends{};
|
||||
for (std::size_t i = 0; i < N; ++i)
|
||||
ends[i] = static_cast<T>(start + step * static_cast<T>(i));
|
||||
return ends;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
template <std::size_t Degree>
|
||||
void add_horner_degree(std::vector<work> & works)
|
||||
{
|
||||
using in_t = std::uint32_t;
|
||||
const in_t center = 50;
|
||||
const in_t eta = 7;
|
||||
works.push_back(cell("keygen", "horner_keygen_d" + std::to_string(Degree), 1, [] {
|
||||
auto built = grotto::make_offset_horner_keys<in_t, Degree>(in_t{50});
|
||||
return touch_word(built.wrap_share[0][0] ^ built.wrap_share[Degree][1],
|
||||
sizeof(built.wrap_share));
|
||||
}));
|
||||
using mat_t = decltype(grotto::make_offset_horner_keys<in_t, Degree>(center));
|
||||
auto mat = std::make_shared<mat_t>(grotto::make_offset_horner_keys<in_t, Degree>(center));
|
||||
for (const std::size_t pieces : std::initializer_list<std::size_t>{1, 4, 16, 64})
|
||||
{
|
||||
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(pieces, 1000));
|
||||
auto rows = std::make_shared<std::vector<std::array<std::uint64_t, Degree + 1>>>(
|
||||
horner_rows<Degree>(pieces));
|
||||
const auto name = "horner_eval_d" + std::to_string(Degree)
|
||||
+ "_p" + std::to_string(pieces);
|
||||
works.push_back(cell("horner", name, 1, [mat, knots, rows, eta] {
|
||||
const auto a = grotto::offset_horner_eval<0, Degree>(*mat, *knots, *rows, eta);
|
||||
const auto b = grotto::offset_horner_eval<1, Degree>(*mat, *knots, *rows, eta);
|
||||
return touch_word(a ^ b, 16);
|
||||
}));
|
||||
}
|
||||
}
|
||||
|
||||
template <typename Input, typename BitPtr, typename GtPtr, std::size_t N>
|
||||
void add_prefix_n(std::vector<work> & works, const char * width, Input step,
|
||||
const BitPtr & bits, const GtPtr & gts)
|
||||
{
|
||||
const auto ends = ends_of<N>(Input{1}, step);
|
||||
const auto ntag = std::to_string(N);
|
||||
works.push_back(cell("prefix", std::string("prefix_") + width + "_e" + ntag, N,
|
||||
[bits, ends] {
|
||||
const auto a = grotto::prefix_parities(std::get<0>(*bits), ends);
|
||||
const auto b = grotto::prefix_parities(std::get<1>(*bits), ends);
|
||||
return touch_arr(std::get<0>(a)) + touch_arr(std::get<0>(b));
|
||||
}));
|
||||
works.push_back(cell("prefix", std::string("signed_prefix_") + width + "_e" + ntag, N,
|
||||
[gts, ends] {
|
||||
const auto a = grotto::signed_prefix_parities(std::get<0>(*gts), ends);
|
||||
const auto b = grotto::signed_prefix_parities(std::get<1>(*gts), ends);
|
||||
return touch_arr(a) + touch_arr(b);
|
||||
}));
|
||||
}
|
||||
|
||||
template <typename Input>
|
||||
void add_prefix_width(std::vector<work> & works, const char * width, Input step)
|
||||
{
|
||||
const auto bit_keys = dpf::make_dpf(Input{1000}, dpf::bit{1});
|
||||
const auto gt_keys = dpf::make_dpf(Input{1000}, dpf::gt(std::uint64_t{1}));
|
||||
using bit_ptr = std::shared_ptr<std::decay_t<decltype(bit_keys)>>;
|
||||
using gt_ptr = std::shared_ptr<std::decay_t<decltype(gt_keys)>>;
|
||||
auto bits = std::make_shared<std::decay_t<decltype(bit_keys)>>(bit_keys);
|
||||
auto gts = std::make_shared<std::decay_t<decltype(gt_keys)>>(gt_keys);
|
||||
add_prefix_n<Input, bit_ptr, gt_ptr, 1>(works, width, step, bits, gts);
|
||||
add_prefix_n<Input, bit_ptr, gt_ptr, 8>(works, width, step, bits, gts);
|
||||
add_prefix_n<Input, bit_ptr, gt_ptr, 32>(works, width, step, bits, gts);
|
||||
add_prefix_n<Input, bit_ptr, gt_ptr, 128>(works, width, step, bits, gts);
|
||||
}
|
||||
|
||||
int main(int argc, char ** argv)
|
||||
{
|
||||
const auto opt = profile::parse_args(argc, argv, 30, 2, kUsage);
|
||||
std::vector<work> works;
|
||||
|
||||
const unsigned precisions[] = {8u, 16u, 24u, 32u};
|
||||
const auto add_window = [&](const char * stem, grotto::window which, unsigned bits) {
|
||||
auto samples = probe(which, bits,
|
||||
[](grotto::window w, unsigned b, std::int64_t raw) {
|
||||
return grotto::eval_window(w, b, raw);
|
||||
}, around_one(bits));
|
||||
add_table(works, "window",
|
||||
std::string("window_") + stem + "_p" + std::to_string(bits),
|
||||
[which, bits](std::int64_t raw) {
|
||||
return grotto::eval_window(which, bits, raw);
|
||||
}, std::move(samples));
|
||||
};
|
||||
const auto add_principal = [&](const char * stem, grotto::principal which, unsigned bits) {
|
||||
auto samples = probe(which, bits,
|
||||
[](grotto::principal w, unsigned b, std::int64_t raw) {
|
||||
return grotto::eval_principal(w, b, raw);
|
||||
}, around_one(bits));
|
||||
add_table(works, "principal",
|
||||
std::string("principal_") + stem + "_p" + std::to_string(bits),
|
||||
[which, bits](std::int64_t raw) {
|
||||
return grotto::eval_principal(which, bits, raw);
|
||||
}, std::move(samples));
|
||||
};
|
||||
const auto add_closed = [&](const char * stem, grotto::closed which, unsigned bits) {
|
||||
auto samples = probe(which, bits,
|
||||
[](grotto::closed w, unsigned b, std::int64_t raw) {
|
||||
return grotto::eval_closed(w, b, raw);
|
||||
}, around_one(bits));
|
||||
add_table(works, "closed",
|
||||
std::string("closed_") + stem + "_p" + std::to_string(bits),
|
||||
[which, bits](std::int64_t raw) {
|
||||
return grotto::eval_closed(which, bits, raw);
|
||||
}, std::move(samples));
|
||||
};
|
||||
const auto add_reduced = [&](const char * stem, grotto::reduced which, unsigned bits) {
|
||||
auto samples = probe(which, bits,
|
||||
[](grotto::reduced w, unsigned b, std::int64_t raw) {
|
||||
return grotto::eval_reduced(w, b, raw);
|
||||
}, around_one(bits));
|
||||
add_table(works, "reduced",
|
||||
std::string("reduced_") + stem + "_p" + std::to_string(bits),
|
||||
[which, bits](std::int64_t raw) {
|
||||
return grotto::eval_reduced(which, bits, raw);
|
||||
}, std::move(samples));
|
||||
};
|
||||
|
||||
const struct { const char * stem; grotto::window which; } windows[] = {
|
||||
{"gelu", grotto::window::gelu},
|
||||
{"sigmoid", grotto::window::sigmoid},
|
||||
{"tanh", grotto::window::tanh},
|
||||
{"erf", grotto::window::erf},
|
||||
{"erfc", grotto::window::erfc},
|
||||
{"silu", grotto::window::silu},
|
||||
{"mish", grotto::window::mish},
|
||||
{"softplus", grotto::window::softplus},
|
||||
{"probit", grotto::window::probit},
|
||||
{"smoothstep", grotto::window::smoothstep},
|
||||
{"asin", grotto::window::asin},
|
||||
{"acos", grotto::window::acos},
|
||||
{"hardelish", grotto::window::hardelish},
|
||||
{"lecun_tanh", grotto::window::lecun_tanh},
|
||||
};
|
||||
for (const auto & spec : windows)
|
||||
for (const unsigned bits : precisions)
|
||||
add_window(spec.stem, spec.which, bits);
|
||||
|
||||
const struct { const char * stem; grotto::principal which; } principals[] = {
|
||||
{"ln", grotto::principal::ln},
|
||||
{"exp", grotto::principal::exp},
|
||||
{"sin", grotto::principal::sin},
|
||||
{"sqrt", grotto::principal::sqrt},
|
||||
{"sinh", grotto::principal::sinh},
|
||||
{"inv", grotto::principal::inv},
|
||||
};
|
||||
for (const auto & spec : principals)
|
||||
for (const unsigned bits : precisions)
|
||||
add_principal(spec.stem, spec.which, bits);
|
||||
|
||||
const struct { const char * stem; grotto::closed which; } closeds[] = {
|
||||
{"atan", grotto::closed::atan},
|
||||
{"cbrt", grotto::closed::cbrt},
|
||||
{"sinc", grotto::closed::sinc},
|
||||
{"softsign", grotto::closed::softsign},
|
||||
{"logistic", grotto::closed::logistic},
|
||||
};
|
||||
for (const auto & spec : closeds)
|
||||
for (const unsigned bits : precisions)
|
||||
add_closed(spec.stem, spec.which, bits);
|
||||
|
||||
const struct { const char * stem; grotto::reduced which; } reduceds[] = {
|
||||
{"ln", grotto::reduced::ln},
|
||||
{"exp", grotto::reduced::exp},
|
||||
{"sin", grotto::reduced::sin},
|
||||
{"expm1", grotto::reduced::expm1},
|
||||
{"log1p", grotto::reduced::log1p},
|
||||
{"sqrt", grotto::reduced::sqrt},
|
||||
};
|
||||
for (const auto & spec : reduceds)
|
||||
for (const unsigned bits : precisions)
|
||||
add_reduced(spec.stem, spec.which, bits);
|
||||
|
||||
const auto add_exact = [&](const std::string & name, unsigned bits, auto fn) {
|
||||
std::vector<std::int64_t> samples;
|
||||
for (const std::int64_t raw : around_one(bits))
|
||||
{
|
||||
try
|
||||
{
|
||||
(void)fn(raw);
|
||||
samples.push_back(raw);
|
||||
}
|
||||
catch (const std::exception &)
|
||||
{
|
||||
}
|
||||
}
|
||||
add_table(works, "exact", name, fn, std::move(samples));
|
||||
};
|
||||
for (const unsigned bits : {8u, 16u, 32u})
|
||||
{
|
||||
const auto tag = "_p" + std::to_string(bits);
|
||||
add_exact("exact_dec_floor" + tag, bits, [bits](std::int64_t raw) {
|
||||
return grotto::eval_dec_floor(raw, bits);
|
||||
});
|
||||
add_exact("exact_bit_width" + tag, bits, [bits](std::int64_t raw) {
|
||||
return grotto::eval_bit_width<std::int64_t>(raw, bits);
|
||||
});
|
||||
add_exact("exact_dec_width" + tag, bits, [bits](std::int64_t raw) {
|
||||
return grotto::eval_dec_width(raw, bits);
|
||||
});
|
||||
add_exact("exact_ilog256" + tag, bits, [bits](std::int64_t raw) {
|
||||
return grotto::eval_ilog256<std::int64_t>(raw, bits);
|
||||
});
|
||||
}
|
||||
|
||||
add_horner_degree<1>(works);
|
||||
add_horner_degree<3>(works);
|
||||
|
||||
using in_t = std::uint32_t;
|
||||
const in_t center = 50;
|
||||
const in_t eta = 7;
|
||||
for (const std::size_t degree : std::initializer_list<std::size_t>{1, 4, 8, 16})
|
||||
{
|
||||
works.push_back(cell("keygen", "poly_keygen_d" + std::to_string(degree), 1,
|
||||
[degree] {
|
||||
auto built = grotto::make_offset_poly_keys(in_t{50}, degree);
|
||||
return touch_word(built.wrap_share[0][0],
|
||||
built.wrap_share.size() * sizeof(built.wrap_share[0]));
|
||||
}));
|
||||
using mat_t = decltype(grotto::make_offset_poly_keys(center, degree));
|
||||
auto mat = std::make_shared<mat_t>(grotto::make_offset_poly_keys(center, degree));
|
||||
for (const std::size_t pieces : std::initializer_list<std::size_t>{4, 8, 16})
|
||||
{
|
||||
if (degree == 16 && pieces == 16)
|
||||
continue;
|
||||
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(pieces, 800));
|
||||
auto rows = std::make_shared<std::vector<std::vector<std::uint64_t>>>(
|
||||
poly_rows(pieces, degree));
|
||||
const auto name = "poly_eval_d" + std::to_string(degree)
|
||||
+ "_p" + std::to_string(pieces);
|
||||
works.push_back(cell("poly", name, 1, [mat, knots, rows] {
|
||||
const auto a = grotto::offset_poly_eval<0>(*mat, *knots, *rows, eta, nullptr);
|
||||
const auto b = grotto::offset_poly_eval<1>(*mat, *knots, *rows, eta, nullptr);
|
||||
return touch_word(a ^ b, 16);
|
||||
}));
|
||||
}
|
||||
}
|
||||
|
||||
for (const std::size_t degree : std::initializer_list<std::size_t>{1, 4, 8, 16})
|
||||
{
|
||||
works.push_back(cell("keygen", "jet_keygen_d" + std::to_string(degree), 1,
|
||||
[degree] {
|
||||
auto built = grotto::make_offset_jet_keys(in_t{50}, degree);
|
||||
return touch_word(built.wrap_share[0][0],
|
||||
built.wrap_share.size() * sizeof(built.wrap_share[0]));
|
||||
}));
|
||||
using mat_t = decltype(grotto::make_offset_jet_keys(center, degree));
|
||||
auto mat = std::make_shared<mat_t>(grotto::make_offset_jet_keys(center, degree));
|
||||
for (const std::size_t pieces : std::initializer_list<std::size_t>{4, 16})
|
||||
{
|
||||
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(pieces, 800));
|
||||
auto col = std::make_shared<std::vector<std::uint64_t>>(coeff_col(degree));
|
||||
const auto name = "jet_eval_d" + std::to_string(degree)
|
||||
+ "_p" + std::to_string(pieces);
|
||||
works.push_back(cell("jet", name, 1, [mat, knots, col] {
|
||||
const auto a = grotto::offset_jet_eval<0>(*mat, *knots, *col, eta);
|
||||
const auto b = grotto::offset_jet_eval<1>(*mat, *knots, *col, eta);
|
||||
return touch_word(a ^ b, 16);
|
||||
}));
|
||||
}
|
||||
}
|
||||
|
||||
for (const std::size_t degree : std::initializer_list<std::size_t>{1, 4, 8})
|
||||
{
|
||||
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(8, 900));
|
||||
auto col = std::make_shared<std::vector<std::uint64_t>>(coeff_col(degree));
|
||||
using odd_t = decltype(grotto::make_offset_twist_keys(center, degree, std::uint64_t{3}));
|
||||
using half_t = decltype(grotto::make_offset_twist_keys(center, degree, grotto::twist_half));
|
||||
auto odd = std::make_shared<odd_t>(
|
||||
grotto::make_offset_twist_keys(center, degree, std::uint64_t{3}));
|
||||
auto half = std::make_shared<half_t>(
|
||||
grotto::make_offset_twist_keys(center, degree, grotto::twist_half));
|
||||
const auto dtag = std::to_string(degree);
|
||||
works.push_back(cell("keygen", "twist_keygen_odd_d" + dtag, 1, [degree] {
|
||||
auto built = grotto::make_offset_twist_keys(in_t{50}, degree, std::uint64_t{3});
|
||||
return touch_word(built.wrap_share[0][0],
|
||||
built.wrap_share.size() * sizeof(built.wrap_share[0]));
|
||||
}));
|
||||
works.push_back(cell("twist", "twist_eval_odd_d" + dtag + "_p8", 1,
|
||||
[odd, knots, col] {
|
||||
const auto a = grotto::offset_twist_eval<0>(*odd, *knots, *col, eta);
|
||||
const auto b = grotto::offset_twist_eval<1>(*odd, *knots, *col, eta);
|
||||
return touch_word(a ^ b, 16);
|
||||
}));
|
||||
works.push_back(cell("twist", "twist_eval_half_d" + dtag + "_p8", 1,
|
||||
[half, knots, col] {
|
||||
const auto a = grotto::offset_twist_eval<0>(*half, *knots, *col, eta);
|
||||
const auto b = grotto::offset_twist_eval<1>(*half, *knots, *col, eta);
|
||||
return touch_word(a ^ b, 16);
|
||||
}));
|
||||
}
|
||||
|
||||
{
|
||||
const std::vector<std::uint64_t> fib_state{1, 0};
|
||||
const std::vector<std::uint64_t> wide_state{1, 2, 3, 4};
|
||||
using fib_t = decltype(grotto::make_offset_repr_keys(center, fib_state));
|
||||
using wide_t = decltype(grotto::make_offset_repr_keys(center, wide_state));
|
||||
auto fib = std::make_shared<fib_t>(grotto::make_offset_repr_keys(center, fib_state));
|
||||
auto wide = std::make_shared<wide_t>(grotto::make_offset_repr_keys(center, wide_state));
|
||||
auto fib_m = std::make_shared<std::vector<std::vector<std::uint64_t>>>(
|
||||
grotto::offset_repr_fibonacci_matrix());
|
||||
auto tri_m = std::make_shared<std::vector<std::vector<std::uint64_t>>>(
|
||||
std::vector<std::vector<std::uint64_t>>{
|
||||
{1, 1, 0, 0},
|
||||
{0, 1, 1, 0},
|
||||
{0, 0, 1, 1},
|
||||
{0, 0, 0, 1},
|
||||
});
|
||||
works.push_back(cell("keygen", "repr_keygen_dim2", 1, [] {
|
||||
auto built = grotto::make_offset_repr_keys(in_t{50}, std::vector<std::uint64_t>{1, 0});
|
||||
return touch_word(built.wrap_share[0][0],
|
||||
built.wrap_share.size() * sizeof(built.wrap_share[0]));
|
||||
}));
|
||||
works.push_back(cell("keygen", "repr_keygen_dim4", 1, [] {
|
||||
auto built = grotto::make_offset_repr_keys(in_t{50},
|
||||
std::vector<std::uint64_t>{1, 2, 3, 4});
|
||||
return touch_word(built.wrap_share[0][0],
|
||||
built.wrap_share.size() * sizeof(built.wrap_share[0]));
|
||||
}));
|
||||
for (const std::size_t pieces : std::initializer_list<std::size_t>{4, 8, 16})
|
||||
{
|
||||
auto knots = std::make_shared<std::vector<in_t>>(knots_of<in_t>(pieces, 700));
|
||||
const auto tag = "_p" + std::to_string(pieces);
|
||||
works.push_back(cell("repr", "repr_eval_fib" + tag, 1, [fib, fib_m, knots] {
|
||||
const auto a = grotto::offset_repr_eval<0>(*fib, *fib_m, *knots, eta);
|
||||
const auto b = grotto::offset_repr_eval<1>(*fib, *fib_m, *knots, eta);
|
||||
return touch_vec(a) + touch_vec(b);
|
||||
}));
|
||||
works.push_back(cell("repr", "repr_eval_dim4" + tag, 1, [wide, tri_m, knots] {
|
||||
const auto a = grotto::offset_repr_eval<0>(*wide, *tri_m, *knots, eta);
|
||||
const auto b = grotto::offset_repr_eval<1>(*wide, *tri_m, *knots, eta);
|
||||
return touch_vec(a) + touch_vec(b);
|
||||
}));
|
||||
}
|
||||
works.push_back(cell("repr", "repr_crc32_jump", 64, [] {
|
||||
std::uint32_t state = 0x12345678u;
|
||||
for (unsigned i = 0; i < 64; ++i)
|
||||
state = grotto::offset_repr_crc32_jump(state, 1ull << (i % 17));
|
||||
return touch_word(state, 64u * 32u * sizeof(std::uint32_t));
|
||||
}));
|
||||
}
|
||||
|
||||
add_prefix_width<std::uint16_t>(works, "u16", 400);
|
||||
add_prefix_width<std::uint32_t>(works, "u32", 100000);
|
||||
|
||||
return profile::run_works(opt, works);
|
||||
}
|
||||
365
test/profile/harness.hpp
Normal file
365
test/profile/harness.hpp
Normal file
|
|
@ -0,0 +1,365 @@
|
|||
/// @file test/profile/harness.hpp
|
||||
/// @brief Timing loop shared by the in-process profile drivers.
|
||||
#ifndef LIBDPF_TEST_PROFILE_HARNESS_HPP__
|
||||
#define LIBDPF_TEST_PROFILE_HARNESS_HPP__
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <functional>
|
||||
#include <iostream>
|
||||
#include <optional>
|
||||
#include <stdexcept>
|
||||
#include <string>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <chrono>
|
||||
|
||||
#include "dpf/prg_count.hpp"
|
||||
|
||||
namespace profile
|
||||
{
|
||||
|
||||
struct sample
|
||||
{
|
||||
std::uint64_t sink = 0;
|
||||
std::uint64_t out_bytes = 0;
|
||||
/// @brief Set when the case instruments the named cost; blank in TSV otherwise.
|
||||
std::optional<std::uint64_t> prg_evals;
|
||||
std::optional<std::uint64_t> preprocess_bytes;
|
||||
std::optional<std::uint64_t> alloc_bytes;
|
||||
std::optional<std::uint64_t> logical_bytes;
|
||||
};
|
||||
|
||||
inline sample operator+(sample a, sample b)
|
||||
{
|
||||
a.sink ^= b.sink + 0x9e3779b97f4a7c15ull;
|
||||
a.out_bytes += b.out_bytes;
|
||||
if (a.prg_evals || b.prg_evals)
|
||||
a.prg_evals = a.prg_evals.value_or(0) + b.prg_evals.value_or(0);
|
||||
if (a.preprocess_bytes || b.preprocess_bytes)
|
||||
a.preprocess_bytes = a.preprocess_bytes.value_or(0)
|
||||
+ b.preprocess_bytes.value_or(0);
|
||||
if (a.alloc_bytes || b.alloc_bytes)
|
||||
a.alloc_bytes = a.alloc_bytes.value_or(0) + b.alloc_bytes.value_or(0);
|
||||
if (a.logical_bytes || b.logical_bytes)
|
||||
a.logical_bytes = a.logical_bytes.value_or(0)
|
||||
+ b.logical_bytes.value_or(0);
|
||||
return a;
|
||||
}
|
||||
|
||||
/// @brief Keep `word` live. `out_bytes` is reported, not timed on its own.
|
||||
inline sample touch_word(std::uint64_t word, std::uint64_t out_bytes = 0)
|
||||
{
|
||||
sample s;
|
||||
s.sink = word;
|
||||
s.out_bytes = out_bytes;
|
||||
asm volatile("" : "+r"(s.sink)::"memory");
|
||||
return s;
|
||||
}
|
||||
|
||||
/// @brief Fold the ends of a buffer and publish a compiler barrier over it.
|
||||
template <typename Buf>
|
||||
sample touch_buf(const Buf & buf)
|
||||
{
|
||||
sample s;
|
||||
using value_type = typename Buf::value_type;
|
||||
s.out_bytes = buf.size() * sizeof(value_type);
|
||||
s.sink = buf.size();
|
||||
if (buf.size() != 0)
|
||||
{
|
||||
std::uint64_t a = 0;
|
||||
std::uint64_t b = 0;
|
||||
const std::size_t n = sizeof(value_type) < sizeof(a) ? sizeof(value_type) : sizeof(a);
|
||||
std::memcpy(&a, &buf[0], n);
|
||||
std::memcpy(&b, &buf[buf.size() - 1], n);
|
||||
s.sink ^= a ^ (b + buf.size());
|
||||
}
|
||||
if (buf.size() != 0)
|
||||
asm volatile("" : "+r"(s.sink) : "r"(buf.data()) : "memory");
|
||||
else
|
||||
asm volatile("" : "+r"(s.sink)::"memory");
|
||||
return s;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
sample touch_vec(const std::vector<T> & v)
|
||||
{
|
||||
std::uint64_t h = v.size();
|
||||
for (const T & x : v)
|
||||
{
|
||||
std::uint64_t w = 0;
|
||||
if constexpr (std::is_integral_v<T>)
|
||||
w = static_cast<std::uint64_t>(x);
|
||||
else
|
||||
{
|
||||
const std::size_t n = sizeof(T) < sizeof(w) ? sizeof(T) : sizeof(w);
|
||||
std::memcpy(&w, &x, n);
|
||||
}
|
||||
h ^= w + 0x9e3779b97f4a7c15ull;
|
||||
h *= 0x100000001b3ull;
|
||||
}
|
||||
return touch_word(h, v.size() * sizeof(T));
|
||||
}
|
||||
|
||||
template <typename T, std::size_t N>
|
||||
sample touch_arr(const std::array<T, N> & a)
|
||||
{
|
||||
std::uint64_t h = N;
|
||||
for (const T & x : a)
|
||||
{
|
||||
std::uint64_t w = 0;
|
||||
if constexpr (std::is_integral_v<T>)
|
||||
w = static_cast<std::uint64_t>(x);
|
||||
else
|
||||
w = x ? 1u : 0u;
|
||||
h = (h << 1) ^ w;
|
||||
}
|
||||
return touch_word(h, N * sizeof(T));
|
||||
}
|
||||
|
||||
/// @brief Reset the PRG counter, run `fn`, and attach the delta to the sample.
|
||||
template <typename Fn>
|
||||
sample with_prg(Fn && fn)
|
||||
{
|
||||
dpf::prg::reset_eval_count();
|
||||
sample s = std::forward<Fn>(fn)();
|
||||
s.prg_evals = dpf::prg::eval_count();
|
||||
return s;
|
||||
}
|
||||
|
||||
inline sample with_costs(sample s, std::uint64_t preprocess, std::uint64_t alloc,
|
||||
std::uint64_t logical)
|
||||
{
|
||||
s.preprocess_bytes = preprocess;
|
||||
s.alloc_bytes = alloc;
|
||||
s.logical_bytes = logical;
|
||||
return s;
|
||||
}
|
||||
|
||||
inline std::string opt_field(const std::optional<std::uint64_t> & v)
|
||||
{
|
||||
return v ? std::to_string(*v) : std::string{};
|
||||
}
|
||||
|
||||
inline std::uint64_t ticks()
|
||||
{
|
||||
#if defined(__x86_64__) || defined(__i386__)
|
||||
unsigned lo = 0;
|
||||
unsigned hi = 0;
|
||||
asm volatile("rdtscp" : "=a"(lo), "=d"(hi)::"rcx");
|
||||
return (static_cast<std::uint64_t>(hi) << 32) | lo;
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
struct work
|
||||
{
|
||||
std::string name;
|
||||
std::uint64_t items = 1;
|
||||
std::function<sample()> fn;
|
||||
/// @brief Top-level group: `eval` or `grotto`.
|
||||
std::string family;
|
||||
/// @brief Piecewise rerun unit, such as `interval` or `horner`.
|
||||
std::string slice;
|
||||
/// @brief `std` is the default matrix. `heavy` is opt-in.
|
||||
std::string tier = "std";
|
||||
};
|
||||
|
||||
inline work make_work(std::string family, std::string slice, std::string name,
|
||||
std::uint64_t items, std::function<sample()> fn, std::string tier = "std")
|
||||
{
|
||||
work w;
|
||||
w.family = std::move(family);
|
||||
w.slice = std::move(slice);
|
||||
w.name = std::move(name);
|
||||
w.items = items;
|
||||
w.fn = std::move(fn);
|
||||
w.tier = std::move(tier);
|
||||
return w;
|
||||
}
|
||||
|
||||
struct parsed
|
||||
{
|
||||
std::uint64_t repeat = 1;
|
||||
std::uint64_t warmup = 0;
|
||||
bool list = false;
|
||||
std::vector<std::string> only;
|
||||
std::vector<std::string> families;
|
||||
std::vector<std::string> slices;
|
||||
std::string tier;
|
||||
};
|
||||
|
||||
inline parsed parse_args(int argc, char ** argv, std::uint64_t repeat,
|
||||
std::uint64_t warmup, const char * usage)
|
||||
{
|
||||
parsed o;
|
||||
o.repeat = repeat;
|
||||
o.warmup = warmup;
|
||||
for (int i = 1; i < argc; ++i)
|
||||
{
|
||||
const std::string a = argv[i];
|
||||
auto need = [&](const char * flag) {
|
||||
if (i + 1 >= argc)
|
||||
{
|
||||
std::cerr << "missing value for " << flag << "\n";
|
||||
std::exit(2);
|
||||
}
|
||||
return std::string(argv[++i]);
|
||||
};
|
||||
if (a == "--list")
|
||||
o.list = true;
|
||||
else if (a == "--repeat")
|
||||
o.repeat = std::stoull(need("--repeat"));
|
||||
else if (a == "--warmup")
|
||||
o.warmup = std::stoull(need("--warmup"));
|
||||
else if (a == "--case")
|
||||
o.only.push_back(need("--case"));
|
||||
else if (a == "--family")
|
||||
o.families.push_back(need("--family"));
|
||||
else if (a == "--slice")
|
||||
o.slices.push_back(need("--slice"));
|
||||
else if (a == "--tier")
|
||||
o.tier = need("--tier");
|
||||
else if (a == "--help" || a == "-h")
|
||||
{
|
||||
std::cout << usage;
|
||||
std::exit(0);
|
||||
}
|
||||
else
|
||||
{
|
||||
std::cerr << "unknown argument: " << a << "\n" << usage;
|
||||
std::exit(2);
|
||||
}
|
||||
}
|
||||
if (o.repeat == 0)
|
||||
o.repeat = 1;
|
||||
return o;
|
||||
}
|
||||
|
||||
inline bool contains(const std::vector<std::string> & hay, const std::string & needle)
|
||||
{
|
||||
return std::find(hay.begin(), hay.end(), needle) != hay.end();
|
||||
}
|
||||
|
||||
inline bool matches(const parsed & opt, const work & w)
|
||||
{
|
||||
if (!opt.tier.empty() && opt.tier != "all" && w.tier != opt.tier)
|
||||
return false;
|
||||
if (!opt.families.empty() && !contains(opt.families, w.family))
|
||||
return false;
|
||||
if (!opt.slices.empty() && !contains(opt.slices, w.slice))
|
||||
return false;
|
||||
if (!opt.only.empty() && !contains(opt.only, w.name))
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
inline int run_works(const parsed & opt, const std::vector<work> & all)
|
||||
{
|
||||
for (const auto & name : opt.only)
|
||||
{
|
||||
bool found = false;
|
||||
for (const auto & w : all)
|
||||
found = found || w.name == name;
|
||||
if (!found)
|
||||
{
|
||||
std::cerr << "unknown case: " << name << "\n";
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<const work *> chosen;
|
||||
for (const auto & w : all)
|
||||
{
|
||||
if (matches(opt, w))
|
||||
chosen.push_back(&w);
|
||||
}
|
||||
if (chosen.empty())
|
||||
{
|
||||
std::cerr << "no cases match the requested family/slice/tier/case\n";
|
||||
return 2;
|
||||
}
|
||||
|
||||
if (opt.list)
|
||||
{
|
||||
std::cout << "family\tslice\tcase\titems\ttier\n";
|
||||
for (const work * w : chosen)
|
||||
std::cout << w->family << '\t' << w->slice << '\t' << w->name
|
||||
<< '\t' << w->items << '\t' << w->tier << '\n';
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::cout
|
||||
<< "family\tslice\tcase\titems\trepeat\twarmup\tavg_ns\tmin_ns\tmax_ns\t"
|
||||
<< "avg_cycles\tper_item_ns\tout_bytes\tsink\t"
|
||||
<< "prg_evals\tpreprocess_bytes\talloc_bytes\tlogical_bytes\tlayout_waste\n";
|
||||
int fails = 0;
|
||||
std::uint64_t all_sink = 0;
|
||||
for (const work * w : chosen)
|
||||
{
|
||||
try
|
||||
{
|
||||
for (std::uint64_t i = 0; i < opt.warmup; ++i)
|
||||
all_sink ^= w->fn().sink;
|
||||
|
||||
std::uint64_t total_ns = 0;
|
||||
std::uint64_t min_ns = ~std::uint64_t{0};
|
||||
std::uint64_t max_ns = 0;
|
||||
std::uint64_t total_cycles = 0;
|
||||
sample last{};
|
||||
for (std::uint64_t i = 0; i < opt.repeat; ++i)
|
||||
{
|
||||
const auto c0 = ticks();
|
||||
const auto t0 = std::chrono::steady_clock::now();
|
||||
const sample s = w->fn();
|
||||
const auto t1 = std::chrono::steady_clock::now();
|
||||
const auto c1 = ticks();
|
||||
const auto ns = static_cast<std::uint64_t>(
|
||||
std::chrono::duration_cast<std::chrono::nanoseconds>(t1 - t0).count());
|
||||
total_ns += ns;
|
||||
min_ns = std::min(min_ns, ns);
|
||||
max_ns = std::max(max_ns, ns);
|
||||
total_cycles += c1 - c0;
|
||||
// Last sample, not an xor across repeats: identical samples
|
||||
// would cancel and the column would read as zero.
|
||||
last = s;
|
||||
all_sink ^= s.sink + i;
|
||||
}
|
||||
const std::uint64_t avg_ns = total_ns / opt.repeat;
|
||||
const std::uint64_t avg_cycles = total_cycles / opt.repeat;
|
||||
const std::uint64_t per_item = w->items == 0 ? avg_ns : avg_ns / w->items;
|
||||
std::string layout_waste;
|
||||
if (last.alloc_bytes && last.logical_bytes
|
||||
&& *last.alloc_bytes >= *last.logical_bytes)
|
||||
{
|
||||
layout_waste = std::to_string(*last.alloc_bytes - *last.logical_bytes);
|
||||
}
|
||||
std::cout << w->family << '\t' << w->slice << '\t' << w->name << '\t'
|
||||
<< w->items << '\t' << opt.repeat << '\t'
|
||||
<< opt.warmup << '\t' << avg_ns << '\t' << min_ns << '\t' << max_ns
|
||||
<< '\t' << avg_cycles << '\t' << per_item << '\t' << last.out_bytes
|
||||
<< '\t' << last.sink << '\t'
|
||||
<< opt_field(last.prg_evals) << '\t'
|
||||
<< opt_field(last.preprocess_bytes) << '\t'
|
||||
<< opt_field(last.alloc_bytes) << '\t'
|
||||
<< opt_field(last.logical_bytes) << '\t'
|
||||
<< layout_waste << '\n';
|
||||
}
|
||||
catch (const std::exception & ex)
|
||||
{
|
||||
std::cerr << w->name << " failed: " << ex.what() << "\n";
|
||||
++fails;
|
||||
}
|
||||
}
|
||||
std::cout << "sink\t" << all_sink << "\n";
|
||||
return fails == 0 ? 0 : 1;
|
||||
}
|
||||
|
||||
} // namespace profile
|
||||
|
||||
#endif // LIBDPF_TEST_PROFILE_HARNESS_HPP__
|
||||
478
test/profile/party_profile.cpp
Normal file
478
test/profile/party_profile.cpp
Normal file
|
|
@ -0,0 +1,478 @@
|
|||
/// @file test/profile/party_profile.cpp
|
||||
/// @brief Workload for the (2+1) party protocols.
|
||||
///
|
||||
/// Spawns p0/p1/p2 the same way party_bench does. Each role reports CPU time
|
||||
/// and the bytes and frames of the flow itself. The repeat barrier is not
|
||||
/// included in those counters. `wall_ms` includes process startup.
|
||||
///
|
||||
/// profile_party --list
|
||||
/// profile_party --suite core --repeat 5 --warmup 1
|
||||
/// profile_party --suite extreme
|
||||
/// profile_party --suite gadget --repeat 3 --warmup 1
|
||||
/// profile_party --tag "beaver,bench" --repeat 3
|
||||
/// profile_party --case beaver_dot_n32
|
||||
///
|
||||
/// Profile-guided build. The party binaries must be rebuilt with the same
|
||||
/// mode, because the protocol code lives in those processes:
|
||||
/// cmake -S test -B build-pgo -DLIBDPF_PGO=generate -DCMAKE_BUILD_TYPE=Release
|
||||
/// cmake --build build-pgo --target profile_party p0 p1 p2
|
||||
/// build-pgo/bin/profile_party --suite all --repeat 4 --warmup 1
|
||||
/// cmake -S test -B build-pgo -DLIBDPF_PGO=use -DCMAKE_BUILD_TYPE=Release
|
||||
/// cmake --build build-pgo --target profile_party p0 p1 p2
|
||||
|
||||
#include "cases.hpp"
|
||||
#include "registry.hpp"
|
||||
#include "spawn.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdint>
|
||||
#include <cstdlib>
|
||||
#include <iostream>
|
||||
#include <string>
|
||||
#include <map>
|
||||
#include <sstream>
|
||||
#include <vector>
|
||||
|
||||
namespace
|
||||
{
|
||||
|
||||
const char * const kCore[] = {
|
||||
"beaver_dot_n32",
|
||||
"beaver_product_100_200",
|
||||
"beaver_stream_n128",
|
||||
"beaver_horner_d4",
|
||||
"beaver_batch_one_round",
|
||||
"beaver_xor_mux",
|
||||
"dcf_gt",
|
||||
"dpf_point_2a_7",
|
||||
"geneval_point",
|
||||
"geneval_arith_point",
|
||||
"blocked_dcf_point",
|
||||
"ds_key_agrees",
|
||||
"grotto_prefix_horner",
|
||||
"verifiable_honest_point",
|
||||
};
|
||||
|
||||
const char * const kExtreme[] = {
|
||||
"beaver_dot_n128",
|
||||
"beaver_stream_n2048",
|
||||
"dcf_dense_gt",
|
||||
"dcf_blocked_interval_ip",
|
||||
"grotto_signed_prefix_dense",
|
||||
"grotto_offset_horner_d3",
|
||||
"grotto_offset_horner_multipiece",
|
||||
"grotto_geneval_offset_horner",
|
||||
"geneval_interval_dense",
|
||||
"geneval_cmp_dense_gt",
|
||||
"ds_cmp_many_points",
|
||||
"recent_offset_poly",
|
||||
"recent_offset_jet",
|
||||
"recent_offset_twist",
|
||||
"recent_offset_repr",
|
||||
"recent_dist_dpf3_point",
|
||||
"wildcard_single_leaf",
|
||||
};
|
||||
|
||||
const char * const kGadget[] = {
|
||||
"arith_proj_m5",
|
||||
"arith_proj_m17",
|
||||
"arith_proj_m64",
|
||||
"arith_mul_p5",
|
||||
"arith_mul_p7",
|
||||
"arith_mul_p11",
|
||||
"arith_thresh_b8",
|
||||
"arith_thresh_b16",
|
||||
"arith_chain_mul4",
|
||||
"yao_if_4_2",
|
||||
"yao_if_16_8",
|
||||
"yao_onehot_k4",
|
||||
"yao_onehot_k8",
|
||||
"flute_d2",
|
||||
"flute_d4",
|
||||
"flute_d8",
|
||||
"flute_d4_o8",
|
||||
"shuffle_n16",
|
||||
"shuffle_n64",
|
||||
"shuffle_n256",
|
||||
};
|
||||
|
||||
const char kUsage[] =
|
||||
"profile_party [--list] [--slice S ...] [--tier std|heavy|all|smoke]\n"
|
||||
" [--suite core|extreme|gadget|all] [--tag TAGS] [--case NAME ...]\n"
|
||||
" [--repeat N=3] [--warmup W=1]\n"
|
||||
"Default run is the core suite. --suite gadget is the word-garbling,\n"
|
||||
"stacked-Yao, FLUTE, and hidden-shuffle battery on the party mesh.\n"
|
||||
"--slice/--tier select the cost-matrix catalog instead.\n"
|
||||
"--list with no filter prints every bench flow.\n"
|
||||
"heavy: wildcard_single_leaf, beaver_stream_n512/n2048, dcf_full_*, *_domain.\n"
|
||||
"smoke: flows that are not benchable (rejects and negative tests).\n";
|
||||
|
||||
bool starts_with(const std::string & s, const char * prefix)
|
||||
{
|
||||
const auto n = std::char_traits<char>::length(prefix);
|
||||
return s.size() >= n && s.compare(0, n, prefix) == 0;
|
||||
}
|
||||
|
||||
std::string slice_of(const std::string & name)
|
||||
{
|
||||
if (starts_with(name, "arith_"))
|
||||
return "arith-garble";
|
||||
if (starts_with(name, "yao_"))
|
||||
return "yao-stack";
|
||||
if (starts_with(name, "flute_"))
|
||||
return "flute";
|
||||
if (starts_with(name, "shuffle_"))
|
||||
return "shuffle";
|
||||
if (starts_with(name, "beaver_dot"))
|
||||
return "beaver-dot";
|
||||
if (starts_with(name, "beaver_stream"))
|
||||
return "beaver-stream";
|
||||
if (starts_with(name, "beaver_scale"))
|
||||
return "beaver-scale";
|
||||
if (starts_with(name, "beaver_horner") || name == "beaver_sign_horner")
|
||||
return "beaver-horner";
|
||||
if (starts_with(name, "beaver_product") || starts_with(name, "beaver_mul")
|
||||
|| starts_with(name, "beaver_chained"))
|
||||
return "beaver-product";
|
||||
if (starts_with(name, "beaver_poly") || name == "beaver_like_terms"
|
||||
|| starts_with(name, "beaver_nested") || starts_with(name, "beaver_tuple")
|
||||
|| starts_with(name, "beaver_mixed") || starts_with(name, "beaver_factored"))
|
||||
return "beaver-poly";
|
||||
if (starts_with(name, "beaver_"))
|
||||
return "beaver-misc";
|
||||
if (starts_with(name, "dcf_full"))
|
||||
return "dcf-full";
|
||||
if (starts_with(name, "dcf_dense"))
|
||||
return "dcf-dense";
|
||||
if (starts_with(name, "dcf_blocked") || starts_with(name, "blocked_"))
|
||||
return "blocked";
|
||||
if (starts_with(name, "dcf_"))
|
||||
return "dcf";
|
||||
if (starts_with(name, "geneval_cmp"))
|
||||
return "geneval-cmp";
|
||||
if (starts_with(name, "geneval_"))
|
||||
return "geneval";
|
||||
if (starts_with(name, "grotto_"))
|
||||
return "grotto-net";
|
||||
if (starts_with(name, "ds_"))
|
||||
return "ds";
|
||||
if (starts_with(name, "recent_offset") || starts_with(name, "recent_ring")
|
||||
|| starts_with(name, "recent_closed") || starts_with(name, "recent_exact"))
|
||||
return "recent-grotto";
|
||||
if (starts_with(name, "recent_dpf3") || starts_with(name, "recent_dist_dpf3"))
|
||||
return "dpf3";
|
||||
if (starts_with(name, "recent_"))
|
||||
return "recent";
|
||||
if (starts_with(name, "wildcard_"))
|
||||
return "wildcard";
|
||||
if (starts_with(name, "verifiable_"))
|
||||
return "verifiable";
|
||||
if (starts_with(name, "dpf_"))
|
||||
return "dpf-point";
|
||||
if (starts_with(name, "carry_") || starts_with(name, "extractable_"))
|
||||
return "auth";
|
||||
if (starts_with(name, "cov_paint") || starts_with(name, "cov_idcf")
|
||||
|| name == "cov_eq_at" || name == "cov_ccmp")
|
||||
return "coverage-cmp";
|
||||
if (starts_with(name, "cov_geneval") || name == "cov_idpf_at"
|
||||
|| name == "cov_eval_inner_product")
|
||||
return "coverage-eval";
|
||||
if (starts_with(name, "cov_"))
|
||||
return "coverage";
|
||||
return "other";
|
||||
}
|
||||
|
||||
std::string tier_of(const std::string & name, bool)
|
||||
{
|
||||
const bool domain = name.size() >= 7
|
||||
&& name.compare(name.size() - 7, 7, "_domain") == 0;
|
||||
const bool dpf3_walk = name.find("dpf3") != std::string::npos
|
||||
&& (name.find("update") != std::string::npos
|
||||
|| name.find("proof") != std::string::npos
|
||||
|| name.find("blocked") != std::string::npos);
|
||||
if (name == "wildcard_single_leaf"
|
||||
|| name == "beaver_stream_n512"
|
||||
|| name == "beaver_stream_n2048"
|
||||
|| starts_with(name, "dcf_full_")
|
||||
|| domain
|
||||
|| dpf3_walk
|
||||
|| starts_with(name, "recent_dist_dpf3"))
|
||||
return "heavy";
|
||||
const bool negative = name.find("fail") != std::string::npos
|
||||
|| name.find("tamper") != std::string::npos
|
||||
|| name.find("reject") != std::string::npos
|
||||
|| name.find("wrong") != std::string::npos
|
||||
|| name.find("bad_use") != std::string::npos;
|
||||
if (negative)
|
||||
return "smoke";
|
||||
return "std";
|
||||
}
|
||||
|
||||
std::map<std::string, std::string> fields_of(const std::string & line)
|
||||
{
|
||||
std::map<std::string, std::string> out;
|
||||
std::istringstream in(line);
|
||||
std::string tok;
|
||||
while (in >> tok)
|
||||
{
|
||||
const auto eq = tok.find('=');
|
||||
if (eq == std::string::npos)
|
||||
continue;
|
||||
out.emplace(tok.substr(0, eq), tok.substr(eq + 1));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
std::string field(const std::map<std::string, std::string> & m, const char * key)
|
||||
{
|
||||
const auto it = m.find(key);
|
||||
return it == m.end() ? "0" : it->second;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char ** argv)
|
||||
{
|
||||
dpf::party::register_all_flows();
|
||||
|
||||
std::string suite = "core";
|
||||
std::string tag;
|
||||
std::string tier;
|
||||
std::vector<std::string> slices;
|
||||
std::vector<std::string> case_names;
|
||||
std::uint64_t repeat = 3;
|
||||
std::uint64_t warmup = 1;
|
||||
bool list_only = false;
|
||||
bool suite_set = false;
|
||||
bool tier_set = false;
|
||||
|
||||
for (int i = 1; i < argc; ++i)
|
||||
{
|
||||
const std::string a = argv[i];
|
||||
auto need = [&](const char * flag) {
|
||||
if (i + 1 >= argc)
|
||||
{
|
||||
std::cerr << "missing value for " << flag << "\n";
|
||||
std::exit(2);
|
||||
}
|
||||
return std::string(argv[++i]);
|
||||
};
|
||||
if (a == "--list")
|
||||
list_only = true;
|
||||
else if (a == "--suite")
|
||||
{
|
||||
suite = need("--suite");
|
||||
suite_set = true;
|
||||
}
|
||||
else if (a == "--tag")
|
||||
tag = need("--tag");
|
||||
else if (a == "--case")
|
||||
case_names.push_back(need("--case"));
|
||||
else if (a == "--slice")
|
||||
slices.push_back(need("--slice"));
|
||||
else if (a == "--tier")
|
||||
{
|
||||
tier = need("--tier");
|
||||
tier_set = true;
|
||||
}
|
||||
else if (a == "--repeat")
|
||||
repeat = std::stoull(need("--repeat"));
|
||||
else if (a == "--warmup")
|
||||
warmup = std::stoull(need("--warmup"));
|
||||
else if (a == "--help" || a == "-h")
|
||||
{
|
||||
std::cout << kUsage;
|
||||
return 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
std::cerr << "unknown argument: " << a << "\n" << kUsage;
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
if (repeat == 0)
|
||||
repeat = 1;
|
||||
|
||||
const bool catalog_mode = tier_set || !slices.empty()
|
||||
|| (list_only && !suite_set && tag.empty() && case_names.empty());
|
||||
|
||||
struct picked
|
||||
{
|
||||
std::string name;
|
||||
std::string slice;
|
||||
std::string tier;
|
||||
};
|
||||
std::vector<picked> picked_flows;
|
||||
|
||||
auto take_catalog = [&](const std::string & want_tier) {
|
||||
for (const auto * f : dpf::party::select_flows(""))
|
||||
{
|
||||
const std::string slice = slice_of(f->name);
|
||||
const std::string flow_tier = tier_of(f->name, f->bench);
|
||||
if (want_tier == "std" && flow_tier != "std")
|
||||
continue;
|
||||
if (want_tier == "heavy" && flow_tier != "heavy")
|
||||
continue;
|
||||
if (want_tier == "smoke" && flow_tier != "smoke")
|
||||
continue;
|
||||
if (want_tier == "all" && flow_tier == "smoke")
|
||||
continue;
|
||||
if (!slices.empty()
|
||||
&& std::find(slices.begin(), slices.end(), slice) == slices.end())
|
||||
continue;
|
||||
if (!case_names.empty()
|
||||
&& std::find(case_names.begin(), case_names.end(), f->name) == case_names.end())
|
||||
continue;
|
||||
picked_flows.push_back(picked{f->name, slice, flow_tier});
|
||||
}
|
||||
};
|
||||
|
||||
if (catalog_mode)
|
||||
{
|
||||
const std::string want = tier_set ? tier : (list_only ? "all" : "std");
|
||||
if (want != "std" && want != "heavy" && want != "all" && want != "smoke")
|
||||
{
|
||||
std::cerr << "unknown tier: " << want << "\n" << kUsage;
|
||||
return 2;
|
||||
}
|
||||
take_catalog(want);
|
||||
for (const auto & name : case_names)
|
||||
{
|
||||
bool found = false;
|
||||
for (const auto & row : picked_flows)
|
||||
found = found || row.name == name;
|
||||
if (!found)
|
||||
{
|
||||
std::cerr << "unknown flow: " << name << "\n";
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
if (picked_flows.empty())
|
||||
{
|
||||
std::cerr << "no flows match the requested slice/tier/case\n";
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
std::vector<std::string> names;
|
||||
if (!case_names.empty())
|
||||
names = std::move(case_names);
|
||||
else if (!tag.empty())
|
||||
{
|
||||
for (const auto * f : dpf::party::select_flows(tag))
|
||||
names.emplace_back(f->name);
|
||||
if (names.empty())
|
||||
{
|
||||
std::cerr << "no flows match tag: " << tag << "\n";
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
else if (suite == "core" || suite == "all")
|
||||
{
|
||||
for (const char * n : kCore)
|
||||
names.emplace_back(n);
|
||||
if (suite == "all")
|
||||
{
|
||||
for (const char * n : kExtreme)
|
||||
names.emplace_back(n);
|
||||
for (const char * n : kGadget)
|
||||
names.emplace_back(n);
|
||||
}
|
||||
}
|
||||
else if (suite == "extreme")
|
||||
{
|
||||
for (const char * n : kExtreme)
|
||||
names.emplace_back(n);
|
||||
}
|
||||
else if (suite == "gadget")
|
||||
{
|
||||
for (const char * n : kGadget)
|
||||
names.emplace_back(n);
|
||||
}
|
||||
else
|
||||
{
|
||||
std::cerr << "unknown suite: " << suite << "\n" << kUsage;
|
||||
return 2;
|
||||
}
|
||||
|
||||
for (const auto & name : names)
|
||||
{
|
||||
const auto * f = dpf::party::find_flow(name);
|
||||
if (!f)
|
||||
{
|
||||
std::cerr << "unknown flow: " << name << "\n";
|
||||
return 2;
|
||||
}
|
||||
picked_flows.push_back(picked{name, slice_of(name), tier_of(name, f->bench)});
|
||||
}
|
||||
}
|
||||
|
||||
if (list_only)
|
||||
{
|
||||
std::cout << "family\tslice\tcase\titems\ttier\n";
|
||||
for (const auto & row : picked_flows)
|
||||
std::cout << "party\t" << row.slice << '\t' << row.name
|
||||
<< "\t1\t" << row.tier << '\n';
|
||||
return 0;
|
||||
}
|
||||
|
||||
dpf::party::spawn_opts opts;
|
||||
opts.repeat = repeat;
|
||||
opts.warmup = warmup;
|
||||
opts.metrics = true;
|
||||
|
||||
std::cout
|
||||
<< "family\tslice\ttier\tflow\trole\twall_ms\tavg_ns\tmin_ns\tmax_ns\t"
|
||||
<< "bytes_sent\tbytes_recv\tframes_sent\tframes_recv\t"
|
||||
<< "bytes_recv_from_p2\tbytes_recv_from_peer\t"
|
||||
<< "bytes_sent_to_p2\tbytes_sent_to_peer\t"
|
||||
<< "payload_sent\tpayload_recv\twire_overhead\trounds\tprg_evals\t"
|
||||
<< "avg_bytes_sent\tavg_frames_sent\trc\n";
|
||||
|
||||
int fails = 0;
|
||||
for (const auto & row : picked_flows)
|
||||
{
|
||||
const auto result = dpf::party::spawn_trio_flow(row.name, opts);
|
||||
if (result.metrics_lines.empty())
|
||||
{
|
||||
std::cout << "party\t" << row.slice << '\t' << row.tier << '\t'
|
||||
<< row.name << "\t-\t" << result.wall_ms
|
||||
<< "\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t0\t"
|
||||
<< result.rc[0] << "," << result.rc[1] << "," << result.rc[2]
|
||||
<< "\n";
|
||||
}
|
||||
for (const auto & line : result.metrics_lines)
|
||||
{
|
||||
const auto f = fields_of(line);
|
||||
const auto role = field(f, "role");
|
||||
const int rc = role == "p0" ? result.rc[0]
|
||||
: role == "p1" ? result.rc[1]
|
||||
: role == "p2" ? result.rc[2] : 0;
|
||||
std::cout << "party\t" << row.slice << '\t' << row.tier << '\t'
|
||||
<< row.name << '\t' << role << '\t' << result.wall_ms << '\t'
|
||||
<< field(f, "avg_ns") << '\t'
|
||||
<< field(f, "min_ns") << '\t'
|
||||
<< field(f, "max_ns") << '\t'
|
||||
<< field(f, "bytes_sent") << '\t'
|
||||
<< field(f, "bytes_recv") << '\t'
|
||||
<< field(f, "frames_sent") << '\t'
|
||||
<< field(f, "frames_recv") << '\t'
|
||||
<< field(f, "bytes_recv_from_p2") << '\t'
|
||||
<< field(f, "bytes_recv_from_peer") << '\t'
|
||||
<< field(f, "bytes_sent_to_p2") << '\t'
|
||||
<< field(f, "bytes_sent_to_peer") << '\t'
|
||||
<< field(f, "payload_sent") << '\t'
|
||||
<< field(f, "payload_recv") << '\t'
|
||||
<< field(f, "wire_overhead") << '\t'
|
||||
<< field(f, "rounds") << '\t'
|
||||
<< field(f, "prg_evals") << '\t'
|
||||
<< field(f, "avg_bytes_sent") << '\t'
|
||||
<< field(f, "avg_frames_sent") << '\t'
|
||||
<< rc << '\n';
|
||||
}
|
||||
if (result.rc[0] || result.rc[1] || result.rc[2])
|
||||
++fails;
|
||||
}
|
||||
return fails ? 1 : 0;
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue