/// @file dpf/app_flow.hpp /// @brief Measure an application plan on an explicit `run_config`. /// @details `exercise_plan(p, ex, cfg)` drives both parties of `p` over /// `cfg.kind` (in-process memory, sync streams, async memory, unix /// sockets, TCP mux, parallel TCP, or SCTP) with the configured lanes, /// framing, instances, window, chunking, pipelining, compute threads, /// warmup, and trials. Link setup is outside the timed region. The /// overloads without a config read `run_config::from_env()` once, for /// the example binaries. #ifndef LIBDPF_INCLUDE_DPF_APP_FLOW_HPP__ #define LIBDPF_INCLUDE_DPF_APP_FLOW_HPP__ #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include "dpf/net/asio_ns.hpp" #include "dpf/app_runtime.hpp" #include "dpf/compose.hpp" #include "dpf/compose_async.hpp" #include "dpf/experiment.hpp" #include "dpf/net/async_round_sink.hpp" #include "dpf/net/memory_sink.hpp" #include "dpf/net/round_lane.hpp" #include "dpf/net/stream_array.hpp" #include "dpf/online_session.hpp" #include "dpf/party_runner.hpp" #include "dpf/run_config.hpp" #include "dpf/run_log.hpp" namespace dpf { namespace app { using transport_kind = net::transport; inline const char * transport_name(transport_kind t) noexcept { return net::transport_name(t); } /// @brief `DPF_TRANSPORT` (default `async`). Unknown names throw. inline transport_kind transport_from_env() { return run_config::from_env().kind; } /// @brief Drive a plan through `schedule_session` on an `async_round_sink`. inline void drive_via_schedule_async(const protocol::plan & p, net::async_round_sink & sink, std::vector> & values, const std::map & kernels, std::size_t party, const protocol::drive_options & opt = {}) { protocol::drive_via_schedule(p, sink, values, kernels, party, opt); } /// @brief What one `exercise_plan` measured. /// @details `bytes` is the plan's slot bytes (one instance). `wire_*` are /// party 0's link counters including every header. `wall_ns` is the /// median party-0 drive time over the timed trials. struct cost { std::size_t rounds = 0; std::size_t bytes = 0; std::uint64_t wire_out = 0; std::uint64_t wire_in = 0; std::uint64_t frames_out = 0; std::uint64_t frames_in = 0; std::uint64_t write_calls = 0; std::uint64_t wall_ns = 0; }; inline cost plan_cost(const protocol::plan & p) { cost out; out.rounds = p.rounds(); for (auto n : p.slot_bytes_all()) out.bytes += n; return out; } namespace detail { struct once { std::uint64_t wall_ns = 0; net::stream_stats wire; /// Every party's drive time (`party_walls[0] == wall_ns`). std::vector party_walls; }; /// @brief Parties 0 and 1 on two threads over a paired in-process sink. Party /// `p` draws from `seeds->derive_party(p)` when `seeds` is set; `ex` /// records party 0 and receives both parties' noted seeds. template inline once two_threads(experiment * ex, const experiment * seeds, Drive drive) { once out; out.party_walls.assign(2, 0); std::exception_ptr err[2]; std::vector noted[2]; start_gate gate(2); auto side = [&](unsigned party) { try { std::optional stream; if (seeds != nullptr) stream.emplace(seeds->derive_party(party)); if (!gate.arrive_and_wait()) throw gate_broken(); experiment * mine = party == 0 ? ex : nullptr; protocol::round_probe probe{}; if (mine != nullptr) { probe = mine->probe(); mine->begin_timing(); } const auto a = std::chrono::steady_clock::now(); drive(party, mine != nullptr ? &probe : nullptr); out.party_walls[party] = static_cast( std::chrono::duration_cast( std::chrono::steady_clock::now() - a) .count()); if (mine != nullptr) mine->end_timing(); if (stream) noted[party] = stream->seeds(); } catch (...) { err[party] = std::current_exception(); gate.fail(); } }; std::thread t0(side, 0u); std::thread t1(side, 1u); t0.join(); t1.join(); for (auto & e : err) { if (!e) continue; try { std::rethrow_exception(e); } catch (const gate_broken &) { continue; } catch (...) { } std::rethrow_exception(e); } out.wall_ns = out.party_walls[0]; if (ex != nullptr && seeds != nullptr) for (unsigned p = 0; p < 2; ++p) ex->fold_seeds("p" + std::to_string(p), noted[p]); return out; } /// @brief One run of `plans` on `cfg.kind` from fresh copies of `inputs`; /// `ex` records party 0 when set, and parties draw from `seeds`'s /// derived streams when it is set. inline once run_once(const std::vector & plans, const std::vector & inputs, const std::map & kernels, const run_config & cfg, experiment * ex, const experiment * seeds = nullptr) { std::vector values = inputs; values.resize(plans.size()); const auto slots = plans[0].slot_bytes_all(); auto opt = cfg.drive(); const bool in_memory = cfg.kind == net::transport::memory_sink || cfg.kind == net::transport::memory_stream; if (in_memory && plans.size() != 2) throw std::invalid_argument(std::string("transport ") + net::transport_name(cfg.kind) + " runs two parties"); switch (cfg.kind) { case net::transport::memory_sink: { auto sl = slots.empty() ? std::vector{0} : slots; auto sinks = net::make_memory_sink_pair(cfg.instances, sl); return two_threads(ex, seeds, [&](unsigned party, const protocol::round_probe * pr) { auto o = opt; o.probe = pr; protocol::drive_via_schedule(plans[party], party == 0 ? sinks.first : sinks.second, values[party], kernels, party, o); }); } case net::transport::memory_stream: { const std::size_t nstreams = slots.empty() ? 1 : protocol::lanes_for_plan(slots.size(), opt); auto ends = net::make_memory_stream_pair(nstreams); return two_threads(ex, seeds, [&](unsigned party, const protocol::round_probe * pr) { auto o = opt; o.probe = pr; protocol::drive_plan_on_streams(plans[party], party == 0 ? ends.first : ends.second, values[party], kernels, party, cfg.instances, o); }); } default: { const auto r = run_parties(plans, values, kernels, cfg, ex, nullptr, seeds); once out; out.wall_ns = r.party0_wall_ns; out.wire = r.wire[0]; out.party_walls = r.party_wall_ns; return out; } } } } // namespace detail /// @brief Drive every party's plan over `cfg` and return party 0's cost. /// @details `plans[i]` is party `i`'s plan (2 or 3 parties); each trial starts /// from a fresh copy of `inputs` (missing entries start empty). /// Warmup runs are not timed; `wall_ns` is the median timed trial. inline cost exercise_parties(const std::vector & plans, const std::vector & inputs, const std::map & kernels, experiment * ex, const run_config & cfg) { if (plans.size() < 2 || plans.size() > 3) throw std::invalid_argument("exercise_parties needs 2 or 3 plans"); const protocol::plan & p = plans[0]; cost out = plan_cost(p); if (ex != nullptr) { ex->ingest_plan(p); ex->set_config(cfg.describe()); } if (p.slot_bytes_all().empty() && cfg.kind != net::transport::memory_sink) { DPF_LOG(warning, "trials.skipped") .kv("experiment", ex != nullptr ? ex->name() : std::string("none")) .kv("detail", "the plan has no exchange rounds; nothing was timed and " "wall_ns stays 0"); return out; } const std::size_t total = cfg.warmup + std::max(1, cfg.trials); std::vector walls; std::vector slowest; detail::once last; for (std::size_t t = 0; t < total; ++t) { const bool timed = t >= cfg.warmup; const bool record = t + 1 == total; last = detail::run_once(plans, inputs, kernels, cfg, record ? ex : nullptr, ex); DPF_LOG(debug, "trial").kv("index", t).kv("timed", timed) .kv("instrumented", record && ex != nullptr).kv("p0_wall_ns", last.wall_ns) .kv("wire_out", last.wire.bytes_out).kv("wire_in", last.wire.bytes_in); if (timed) { walls.push_back(last.wall_ns); std::uint64_t w = last.wall_ns; for (auto p : last.party_walls) w = std::max(w, p); slowest.push_back(w); if (ex != nullptr) ex->add_trial(last.wall_ns, last.party_walls); } } std::sort(walls.begin(), walls.end()); std::sort(slowest.begin(), slowest.end()); out.wall_ns = walls.empty() ? 0 : walls[walls.size() / 2]; if (!walls.empty() && log::enabled(log::level::info)) { double mean = 0; for (auto w : walls) mean += static_cast(w); mean /= static_cast(walls.size()); double var = 0; for (auto w : walls) var += (static_cast(w) - mean) * (static_cast(w) - mean); const double sd = walls.size() > 1 ? std::sqrt(var / static_cast(walls.size() - 1)) : 0.0; DPF_LOG(info, "trials") .kv("experiment", ex != nullptr ? ex->name() : std::string("none")) .kv("transport", net::transport_name(cfg.kind)).kv("n", walls.size()) .kv("warmup", cfg.warmup).kv("median_ns", out.wall_ns) .kv("min_ns", walls.front()).kv("max_ns", walls.back()) .kv("mean_ns", mean).kv("sd_ns", sd) .kv("slowest_median_ns", slowest[slowest.size() / 2]) .kv("instrumented_trial", "last") .kv("links", "rebuilt per trial"); } out.wire_out = last.wire.bytes_out; out.wire_in = last.wire.bytes_in; out.frames_out = last.wire.frames_out; out.frames_in = last.wire.frames_in; out.write_calls = last.wire.write_calls; if (ex != nullptr) { experiment::wire_counts w; w.bytes_out = last.wire.bytes_out; w.bytes_in = last.wire.bytes_in; w.payload_out = last.wire.payload_out; w.payload_in = last.wire.payload_in; w.frames_out = last.wire.frames_out; w.frames_in = last.wire.frames_in; w.write_calls = last.wire.write_calls; ex->set_wire(w); } return out; } /// @brief Drive `p` as both parties' plan over `cfg` and return its cost. inline cost exercise_plan(const protocol::plan & p, experiment * ex, const run_config & cfg, const std::map & kernels = {}) { return exercise_parties({p, p}, {}, kernels, ex, cfg); } /// @brief `exercise_plan` on `run_config::from_env()`. inline cost exercise_plan(const protocol::plan & p, experiment * ex = nullptr) { return exercise_plan(p, ex, run_config::from_env()); } inline experiment measure_plan(const char * name, const protocol::plan & p, const run_config & cfg) { experiment ex(name, "p0"); (void)exercise_plan(p, &ex, cfg); return ex; } inline experiment measure_plan(const char * name, const protocol::plan & p) { return measure_plan(name, p, run_config::from_env()); } /// @brief Schedule `c`, then `exercise_plan`. inline cost exercise(protocol::composer & c) { return exercise_plan(c.default_plan()); } /// @brief Exercise `p` and require `expect_rounds`. Prints `name rounds= bytes=`. inline int run_plan(const char * name, const protocol::plan & p, std::size_t expect_rounds, const run_config & cfg = run_config::from_env()) { try { start_logging(cfg); const cost got = exercise_plan(p, nullptr, cfg); std::cout << name << " rounds=" << got.rounds << " bytes=" << got.bytes << " wire_out=" << got.wire_out << " wire_in=" << got.wire_in << "\n"; if (got.rounds != expect_rounds) { std::cerr << name << " expected " << expect_rounds << " rounds\n"; return 1; } } catch (const std::exception & ex) { DPF_LOG(error, "run.failed").kv("name", name).kv("what", ex.what()); std::cerr << name << " flow: " << ex.what() << "\n"; return 1; } return 0; } /// @brief Measure `p`, print cost, and write CSVs under `DPF_EXPERIMENT_DIR`. /// @details When `prep` is set, the prep for that demand is dealt and shipped /// over `cfg`'s transport and its size is printed. inline int run_measured(const char * name, const protocol::plan & p, std::size_t expect_rounds, const run_config & cfg = run_config::from_env(), const prep::demand * prep = nullptr) { try { start_logging(cfg); auto ex = measure_plan(name, p, cfg); std::cout << name << " transport=" << net::transport_name(cfg.kind) << " rounds=" << ex.interactive_rounds() << " bytes=" << ex.plan_bytes_out() << " wire_out=" << ex.wire().bytes_out << " wire_in=" << ex.wire().bytes_in << " wall_ns=" << ex.wall_ns() << " median_ns=" << ex.median_trial_ns() << " cpu_ns=" << ex.cpu_ns() << " prg_evals=" << ex.prg_evals() << " random_bytes=" << ex.random_bytes() << " seed=" << ex.seed_hex(); if (prep != nullptr) { const auto shipped = session::ship_prep(*prep, cfg); std::cout << " prep_bytes=" << shipped.bytes0; } std::cout << "\n"; if (ex.interactive_rounds() != expect_rounds) { std::cerr << name << " expected " << expect_rounds << " rounds\n"; return 1; } if (const char * dir = std::getenv("DPF_EXPERIMENT_DIR")) { if (dir[0] != '\0') ex.write_csv(dir); } } catch (const std::exception & ex) { DPF_LOG(error, "run.failed").kv("name", name).kv("what", ex.what()); std::cerr << name << " measure: " << ex.what() << "\n"; return 1; } return 0; } /// @brief Exercise `c` and require `expect_rounds`. inline int run(const char * name, protocol::composer & c, std::size_t expect_rounds) { return run_plan(name, c.default_plan(), expect_rounds); } /// @brief `run_measured` on `c.default_plan()`. inline int run_measured(const char * name, protocol::composer & c, std::size_t expect_rounds) { return run_measured(name, c.default_plan(), expect_rounds); } /// @brief Many instances, both parties, delays that do not line up. /// @details A worker never sits on a side whose peer is the one that still /// has to submit. Parked receives yield the thread to whichever /// side is behind, so a slow step does not stall the rest and a /// round-robin that always resumes the waiter cannot deadlock. inline void run_fleet(protocol::composer & c, std::size_t instances, std::uint32_t chaos_seed) { if (instances == 0) throw std::invalid_argument("run_fleet needs instances"); auto plan = c.default_plan(); auto slots = plan.slot_bytes_all(); if (slots.empty()) slots.push_back(0); struct side { protocol::drive_cursor cursor{}; std::vector> values; net::memory_sink * sink = nullptr; std::chrono::steady_clock::time_point ready_at{}; bool busy = false; }; struct inst { std::pair sinks; side party[2]{}; }; std::vector all; all.reserve(instances); auto chaos_us = [&](std::size_t id, int party, std::size_t step) { std::uint32_t x = chaos_seed ^ static_cast(id * 0x9E3779B9u) ^ static_cast(party * 0x85EBCA6Bu) ^ static_cast(step * 0xC2B2AE35u); x ^= x << 13; x ^= x >> 17; // Mostly short, with occasional long stalls. Not sorted by instance. const std::uint32_t bucket = x % 17u; if (party == 0 && step == 0 && (id % 5u) == 0) return 12000; if (bucket == 0) return 1500; if (bucket < 4) return static_cast(x % 400u); return static_cast(x % 40u); }; for (std::size_t i = 0; i < instances; ++i) { all.push_back(inst{net::make_memory_sink_pair(1, slots), {}}); auto & created = all.back(); created.party[0].sink = &created.sinks.first; created.party[1].sink = &created.sinks.second; const auto now = std::chrono::steady_clock::now(); created.party[0].ready_at = now + std::chrono::microseconds( chaos_us(i, 0, 0)); created.party[1].ready_at = now + std::chrono::microseconds( chaos_us(i, 1, 0)); } std::mutex mu; std::condition_variable cv; std::map kernels; std::size_t finished = 0; const std::size_t workers = std::min( std::thread::hardware_concurrency() == 0 ? 4 : std::thread::hardware_concurrency(), instances); auto pick = [&](std::unique_lock & lock) -> side * { for (;;) { const auto now = std::chrono::steady_clock::now(); side * best = nullptr; int best_score = 0x7fffffff; std::chrono::steady_clock::time_point soonest = now + std::chrono::hours(1); bool any_left = false; for (auto & item : all) { for (int p = 0; p < 2; ++p) { side & s = item.party[p]; if (s.cursor.done) continue; if (s.busy) { any_left = true; continue; } any_left = true; if (s.ready_at > now) { if (s.ready_at < soonest) soonest = s.ready_at; continue; } // A side still submitting outranks one parked on a receive. // Among those, the one further behind runs first. const int score = static_cast(s.cursor.exchange_i) + (s.cursor.awaiting_peer ? 100000 : 0); if (score < best_score) { best_score = score; best = &s; } } } if (best != nullptr) { best->busy = true; return best; } if (!any_left) return nullptr; cv.wait_until(lock, soonest); } }; auto worker = [&] { std::unique_lock lock(mu); for (;;) { side * s = pick(lock); if (s == nullptr) return; std::size_t inst_i = 0; int party = 0; for (; inst_i < all.size(); ++inst_i) { if (&all[inst_i].party[0] == s) { party = 0; break; } if (&all[inst_i].party[1] == s) { party = 1; break; } } protocol::drive_options opt; opt.park_if_waiting = true; opt.one_exchange = true; opt.cursor = &s->cursor; auto * sink = s->sink; auto * values = &s->values; const std::size_t step = s->cursor.exchange_i; lock.unlock(); protocol::drive(plan, *sink, *values, kernels, static_cast(party), opt); lock.lock(); s->busy = false; if (s->cursor.done) ++finished; else { s->ready_at = std::chrono::steady_clock::now() + std::chrono::microseconds(chaos_us(inst_i, party, step)); } cv.notify_all(); } }; std::vector pool; pool.reserve(workers); for (std::size_t i = 0; i < workers; ++i) pool.emplace_back(worker); for (auto & th : pool) th.join(); if (finished != instances * 2) throw std::runtime_error("run_fleet: not every side finished"); } } // namespace app } // namespace dpf #endif