libdpf/include/dpf/app_flow.hpp

634 lines
21 KiB
C++
Raw Normal View History

/// @file dpf/app_flow.hpp
/// @brief Measure an application plan on an explicit `run_config`.
/// @details `exercise_plan(p, ex, cfg)` drives both parties of `p` over
/// `cfg.kind` (in-process memory, sync streams, async memory, unix
/// sockets, TCP mux, parallel TCP, or SCTP) with the configured lanes,
/// framing, instances, window, chunking, pipelining, compute threads,
/// warmup, and trials. Link setup is outside the timed region. The
/// overloads without a config read `run_config::from_env()` once, for
/// the example binaries.
#ifndef LIBDPF_INCLUDE_DPF_APP_FLOW_HPP__
#define LIBDPF_INCLUDE_DPF_APP_FLOW_HPP__
#include <algorithm>
#include <atomic>
#include <chrono>
#include <cmath>
#include <condition_variable>
#include <cstddef>
#include <cstdint>
#include <cstdlib>
#include <cstring>
#include <exception>
#include <iostream>
#include <map>
#include <mutex>
#include <optional>
#include <string>
#include <thread>
#include <utility>
#include <vector>
#include "dpf/net/asio_ns.hpp"
#include "dpf/app_runtime.hpp"
#include "dpf/compose.hpp"
#include "dpf/compose_async.hpp"
#include "dpf/experiment.hpp"
#include "dpf/net/async_round_sink.hpp"
#include "dpf/net/memory_sink.hpp"
#include "dpf/net/round_lane.hpp"
#include "dpf/net/stream_array.hpp"
#include "dpf/online_session.hpp"
#include "dpf/party_runner.hpp"
#include "dpf/run_config.hpp"
#include "dpf/run_log.hpp"
namespace dpf
{
namespace app
{
using transport_kind = net::transport;
inline const char * transport_name(transport_kind t) noexcept
{
return net::transport_name(t);
}
/// @brief `DPF_TRANSPORT` (default `async`). Unknown names throw.
inline transport_kind transport_from_env()
{
return run_config::from_env().kind;
}
/// @brief Drive a plan through `schedule_session` on an `async_round_sink`.
inline void drive_via_schedule_async(const protocol::plan & p,
net::async_round_sink & sink,
std::vector<std::vector<std::uint8_t>> & values,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels,
std::size_t party, const protocol::drive_options & opt = {})
{
protocol::drive_via_schedule(p, sink, values, kernels, party, opt);
}
/// @brief What one `exercise_plan` measured.
/// @details `bytes` is the plan's slot bytes (one instance). `wire_*` are
/// party 0's link counters including every header. `wall_ns` is the
/// median party-0 drive time over the timed trials.
struct cost
{
std::size_t rounds = 0;
std::size_t bytes = 0;
std::uint64_t wire_out = 0;
std::uint64_t wire_in = 0;
std::uint64_t frames_out = 0;
std::uint64_t frames_in = 0;
std::uint64_t write_calls = 0;
std::uint64_t wall_ns = 0;
};
inline cost plan_cost(const protocol::plan & p)
{
cost out;
out.rounds = p.rounds();
for (auto n : p.slot_bytes_all())
out.bytes += n;
return out;
}
namespace detail
{
struct once
{
std::uint64_t wall_ns = 0;
net::stream_stats wire;
/// Every party's drive time (`party_walls[0] == wall_ns`).
std::vector<std::uint64_t> party_walls;
};
/// @brief Parties 0 and 1 on two threads over a paired in-process sink. Party
/// `p` draws from `seeds->derive_party(p)` when `seeds` is set; `ex`
/// records party 0 and receives both parties' noted seeds.
template <typename Drive>
inline once two_threads(experiment * ex, const experiment * seeds, Drive drive)
{
once out;
out.party_walls.assign(2, 0);
std::exception_ptr err[2];
std::vector<experiment::noted_seed> noted[2];
start_gate gate(2);
auto side = [&](unsigned party) {
try
{
std::optional<experiment> stream;
if (seeds != nullptr)
stream.emplace(seeds->derive_party(party));
if (!gate.arrive_and_wait())
throw gate_broken();
experiment * mine = party == 0 ? ex : nullptr;
protocol::round_probe probe{};
if (mine != nullptr)
{
probe = mine->probe();
mine->begin_timing();
}
const auto a = std::chrono::steady_clock::now();
drive(party, mine != nullptr ? &probe : nullptr);
out.party_walls[party] = static_cast<std::uint64_t>(
std::chrono::duration_cast<std::chrono::nanoseconds>(
std::chrono::steady_clock::now() - a)
.count());
if (mine != nullptr)
mine->end_timing();
if (stream)
noted[party] = stream->seeds();
}
catch (...)
{
err[party] = std::current_exception();
gate.fail();
}
};
std::thread t0(side, 0u);
std::thread t1(side, 1u);
t0.join();
t1.join();
for (auto & e : err)
{
if (!e)
continue;
try
{
std::rethrow_exception(e);
}
catch (const gate_broken &)
{
continue;
}
catch (...)
{
}
std::rethrow_exception(e);
}
out.wall_ns = out.party_walls[0];
if (ex != nullptr && seeds != nullptr)
for (unsigned p = 0; p < 2; ++p)
ex->fold_seeds("p" + std::to_string(p), noted[p]);
return out;
}
/// @brief One run of `plans` on `cfg.kind` from fresh copies of `inputs`;
/// `ex` records party 0 when set, and parties draw from `seeds`'s
/// derived streams when it is set.
inline once run_once(const std::vector<protocol::plan> & plans,
const std::vector<party_values> & inputs,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels,
const run_config & cfg, experiment * ex, const experiment * seeds = nullptr)
{
std::vector<party_values> values = inputs;
values.resize(plans.size());
const auto slots = plans[0].slot_bytes_all();
auto opt = cfg.drive();
const bool in_memory = cfg.kind == net::transport::memory_sink
|| cfg.kind == net::transport::memory_stream;
if (in_memory && plans.size() != 2)
throw std::invalid_argument(std::string("transport ")
+ net::transport_name(cfg.kind) + " runs two parties");
switch (cfg.kind)
{
case net::transport::memory_sink:
{
auto sl = slots.empty() ? std::vector<std::size_t>{0} : slots;
auto sinks = net::make_memory_sink_pair(cfg.instances, sl);
return two_threads(ex, seeds, [&](unsigned party, const protocol::round_probe * pr) {
auto o = opt;
o.probe = pr;
protocol::drive_via_schedule(plans[party],
party == 0 ? sinks.first : sinks.second, values[party], kernels,
party, o);
});
}
case net::transport::memory_stream:
{
const std::size_t nstreams =
slots.empty() ? 1 : protocol::lanes_for_plan(slots.size(), opt);
auto ends = net::make_memory_stream_pair(nstreams);
return two_threads(ex, seeds, [&](unsigned party, const protocol::round_probe * pr) {
auto o = opt;
o.probe = pr;
protocol::drive_plan_on_streams(plans[party],
party == 0 ? ends.first : ends.second, values[party], kernels,
party, cfg.instances, o);
});
}
default:
{
const auto r = run_parties(plans, values, kernels, cfg, ex, nullptr, seeds);
once out;
out.wall_ns = r.party0_wall_ns;
out.wire = r.wire[0];
out.party_walls = r.party_wall_ns;
return out;
}
}
}
} // namespace detail
/// @brief Drive every party's plan over `cfg` and return party 0's cost.
/// @details `plans[i]` is party `i`'s plan (2 or 3 parties); each trial starts
/// from a fresh copy of `inputs` (missing entries start empty).
/// Warmup runs are not timed; `wall_ns` is the median timed trial.
inline cost exercise_parties(const std::vector<protocol::plan> & plans,
const std::vector<party_values> & inputs,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels, experiment * ex,
const run_config & cfg)
{
if (plans.size() < 2 || plans.size() > 3)
throw std::invalid_argument("exercise_parties needs 2 or 3 plans");
const protocol::plan & p = plans[0];
cost out = plan_cost(p);
if (ex != nullptr)
{
ex->ingest_plan(p);
ex->set_config(cfg.describe());
}
if (p.slot_bytes_all().empty() && cfg.kind != net::transport::memory_sink)
{
DPF_LOG(warning, "trials.skipped")
.kv("experiment", ex != nullptr ? ex->name() : std::string("none"))
.kv("detail", "the plan has no exchange rounds; nothing was timed and "
"wall_ns stays 0");
return out;
}
const std::size_t total = cfg.warmup + std::max<std::size_t>(1, cfg.trials);
std::vector<std::uint64_t> walls;
std::vector<std::uint64_t> slowest;
detail::once last;
for (std::size_t t = 0; t < total; ++t)
{
const bool timed = t >= cfg.warmup;
const bool record = t + 1 == total;
last = detail::run_once(plans, inputs, kernels, cfg, record ? ex : nullptr, ex);
DPF_LOG(debug, "trial").kv("index", t).kv("timed", timed)
.kv("instrumented", record && ex != nullptr).kv("p0_wall_ns", last.wall_ns)
.kv("wire_out", last.wire.bytes_out).kv("wire_in", last.wire.bytes_in);
if (timed)
{
walls.push_back(last.wall_ns);
std::uint64_t w = last.wall_ns;
for (auto p : last.party_walls)
w = std::max(w, p);
slowest.push_back(w);
if (ex != nullptr)
ex->add_trial(last.wall_ns, last.party_walls);
}
}
std::sort(walls.begin(), walls.end());
std::sort(slowest.begin(), slowest.end());
out.wall_ns = walls.empty() ? 0 : walls[walls.size() / 2];
if (!walls.empty() && log::enabled(log::level::info))
{
double mean = 0;
for (auto w : walls)
mean += static_cast<double>(w);
mean /= static_cast<double>(walls.size());
double var = 0;
for (auto w : walls)
var += (static_cast<double>(w) - mean) * (static_cast<double>(w) - mean);
const double sd = walls.size() > 1
? std::sqrt(var / static_cast<double>(walls.size() - 1)) : 0.0;
DPF_LOG(info, "trials")
.kv("experiment", ex != nullptr ? ex->name() : std::string("none"))
.kv("transport", net::transport_name(cfg.kind)).kv("n", walls.size())
.kv("warmup", cfg.warmup).kv("median_ns", out.wall_ns)
.kv("min_ns", walls.front()).kv("max_ns", walls.back())
.kv("mean_ns", mean).kv("sd_ns", sd)
.kv("slowest_median_ns", slowest[slowest.size() / 2])
.kv("instrumented_trial", "last")
.kv("links", "rebuilt per trial");
}
out.wire_out = last.wire.bytes_out;
out.wire_in = last.wire.bytes_in;
out.frames_out = last.wire.frames_out;
out.frames_in = last.wire.frames_in;
out.write_calls = last.wire.write_calls;
if (ex != nullptr)
{
experiment::wire_counts w;
w.bytes_out = last.wire.bytes_out;
w.bytes_in = last.wire.bytes_in;
w.payload_out = last.wire.payload_out;
w.payload_in = last.wire.payload_in;
w.frames_out = last.wire.frames_out;
w.frames_in = last.wire.frames_in;
w.write_calls = last.wire.write_calls;
ex->set_wire(w);
}
return out;
}
/// @brief Drive `p` as both parties' plan over `cfg` and return its cost.
inline cost exercise_plan(const protocol::plan & p, experiment * ex,
const run_config & cfg,
const std::map<std::uint32_t, protocol::kernel_fn> & kernels = {})
{
return exercise_parties({p, p}, {}, kernels, ex, cfg);
}
/// @brief `exercise_plan` on `run_config::from_env()`.
inline cost exercise_plan(const protocol::plan & p, experiment * ex = nullptr)
{
return exercise_plan(p, ex, run_config::from_env());
}
inline experiment measure_plan(const char * name, const protocol::plan & p,
const run_config & cfg)
{
experiment ex(name, "p0");
(void)exercise_plan(p, &ex, cfg);
return ex;
}
inline experiment measure_plan(const char * name, const protocol::plan & p)
{
return measure_plan(name, p, run_config::from_env());
}
/// @brief Schedule `c`, then `exercise_plan`.
inline cost exercise(protocol::composer & c)
{
return exercise_plan(c.default_plan());
}
/// @brief Exercise `p` and require `expect_rounds`. Prints `name rounds= bytes=`.
inline int run_plan(const char * name, const protocol::plan & p,
std::size_t expect_rounds, const run_config & cfg = run_config::from_env())
{
try
{
start_logging(cfg);
const cost got = exercise_plan(p, nullptr, cfg);
std::cout << name << " rounds=" << got.rounds << " bytes=" << got.bytes
<< " wire_out=" << got.wire_out << " wire_in=" << got.wire_in
<< "\n";
if (got.rounds != expect_rounds)
{
std::cerr << name << " expected " << expect_rounds << " rounds\n";
return 1;
}
}
catch (const std::exception & ex)
{
DPF_LOG(error, "run.failed").kv("name", name).kv("what", ex.what());
std::cerr << name << " flow: " << ex.what() << "\n";
return 1;
}
return 0;
}
/// @brief Measure `p`, print cost, and write CSVs under `DPF_EXPERIMENT_DIR`.
/// @details When `prep` is set, the prep for that demand is dealt and shipped
/// over `cfg`'s transport and its size is printed.
inline int run_measured(const char * name, const protocol::plan & p,
std::size_t expect_rounds, const run_config & cfg = run_config::from_env(),
const prep::demand * prep = nullptr)
{
try
{
start_logging(cfg);
auto ex = measure_plan(name, p, cfg);
std::cout << name << " transport=" << net::transport_name(cfg.kind)
<< " rounds=" << ex.interactive_rounds()
<< " bytes=" << ex.plan_bytes_out()
<< " wire_out=" << ex.wire().bytes_out
<< " wire_in=" << ex.wire().bytes_in
<< " wall_ns=" << ex.wall_ns()
<< " median_ns=" << ex.median_trial_ns()
<< " cpu_ns=" << ex.cpu_ns()
<< " prg_evals=" << ex.prg_evals()
<< " random_bytes=" << ex.random_bytes()
<< " seed=" << ex.seed_hex();
if (prep != nullptr)
{
const auto shipped = session::ship_prep(*prep, cfg);
std::cout << " prep_bytes=" << shipped.bytes0;
}
std::cout << "\n";
if (ex.interactive_rounds() != expect_rounds)
{
std::cerr << name << " expected " << expect_rounds << " rounds\n";
return 1;
}
if (const char * dir = std::getenv("DPF_EXPERIMENT_DIR"))
{
if (dir[0] != '\0')
ex.write_csv(dir);
}
}
catch (const std::exception & ex)
{
DPF_LOG(error, "run.failed").kv("name", name).kv("what", ex.what());
std::cerr << name << " measure: " << ex.what() << "\n";
return 1;
}
return 0;
}
/// @brief Exercise `c` and require `expect_rounds`.
inline int run(const char * name, protocol::composer & c, std::size_t expect_rounds)
{
return run_plan(name, c.default_plan(), expect_rounds);
}
/// @brief `run_measured` on `c.default_plan()`.
inline int run_measured(const char * name, protocol::composer & c,
std::size_t expect_rounds)
{
return run_measured(name, c.default_plan(), expect_rounds);
}
/// @brief Many instances, both parties, delays that do not line up.
/// @details A worker never sits on a side whose peer is the one that still
/// has to submit. Parked receives yield the thread to whichever
/// side is behind, so a slow step does not stall the rest and a
/// round-robin that always resumes the waiter cannot deadlock.
inline void run_fleet(protocol::composer & c, std::size_t instances,
std::uint32_t chaos_seed)
{
if (instances == 0)
throw std::invalid_argument("run_fleet needs instances");
auto plan = c.default_plan();
auto slots = plan.slot_bytes_all();
if (slots.empty())
slots.push_back(0);
struct side
{
protocol::drive_cursor cursor{};
std::vector<std::vector<std::uint8_t>> values;
net::memory_sink * sink = nullptr;
std::chrono::steady_clock::time_point ready_at{};
bool busy = false;
};
struct inst
{
std::pair<net::memory_sink, net::memory_sink> sinks;
side party[2]{};
};
std::vector<inst> all;
all.reserve(instances);
auto chaos_us = [&](std::size_t id, int party, std::size_t step) {
std::uint32_t x = chaos_seed
^ static_cast<std::uint32_t>(id * 0x9E3779B9u)
^ static_cast<std::uint32_t>(party * 0x85EBCA6Bu)
^ static_cast<std::uint32_t>(step * 0xC2B2AE35u);
x ^= x << 13;
x ^= x >> 17;
// Mostly short, with occasional long stalls. Not sorted by instance.
const std::uint32_t bucket = x % 17u;
if (party == 0 && step == 0 && (id % 5u) == 0)
return 12000;
if (bucket == 0)
return 1500;
if (bucket < 4)
return static_cast<int>(x % 400u);
return static_cast<int>(x % 40u);
};
for (std::size_t i = 0; i < instances; ++i)
{
all.push_back(inst{net::make_memory_sink_pair(1, slots), {}});
auto & created = all.back();
created.party[0].sink = &created.sinks.first;
created.party[1].sink = &created.sinks.second;
const auto now = std::chrono::steady_clock::now();
created.party[0].ready_at = now + std::chrono::microseconds(
chaos_us(i, 0, 0));
created.party[1].ready_at = now + std::chrono::microseconds(
chaos_us(i, 1, 0));
}
std::mutex mu;
std::condition_variable cv;
std::map<std::uint32_t, protocol::kernel_fn> kernels;
std::size_t finished = 0;
const std::size_t workers = std::min<std::size_t>(
std::thread::hardware_concurrency() == 0
? 4
: std::thread::hardware_concurrency(),
instances);
auto pick = [&](std::unique_lock<std::mutex> & lock) -> side * {
for (;;)
{
const auto now = std::chrono::steady_clock::now();
side * best = nullptr;
int best_score = 0x7fffffff;
std::chrono::steady_clock::time_point soonest =
now + std::chrono::hours(1);
bool any_left = false;
for (auto & item : all)
{
for (int p = 0; p < 2; ++p)
{
side & s = item.party[p];
if (s.cursor.done)
continue;
if (s.busy)
{
any_left = true;
continue;
}
any_left = true;
if (s.ready_at > now)
{
if (s.ready_at < soonest)
soonest = s.ready_at;
continue;
}
// A side still submitting outranks one parked on a receive.
// Among those, the one further behind runs first.
const int score = static_cast<int>(s.cursor.exchange_i)
+ (s.cursor.awaiting_peer ? 100000 : 0);
if (score < best_score)
{
best_score = score;
best = &s;
}
}
}
if (best != nullptr)
{
best->busy = true;
return best;
}
if (!any_left)
return nullptr;
cv.wait_until(lock, soonest);
}
};
auto worker = [&] {
std::unique_lock<std::mutex> lock(mu);
for (;;)
{
side * s = pick(lock);
if (s == nullptr)
return;
std::size_t inst_i = 0;
int party = 0;
for (; inst_i < all.size(); ++inst_i)
{
if (&all[inst_i].party[0] == s)
{
party = 0;
break;
}
if (&all[inst_i].party[1] == s)
{
party = 1;
break;
}
}
protocol::drive_options opt;
opt.park_if_waiting = true;
opt.one_exchange = true;
opt.cursor = &s->cursor;
auto * sink = s->sink;
auto * values = &s->values;
const std::size_t step = s->cursor.exchange_i;
lock.unlock();
protocol::drive(plan, *sink, *values, kernels,
static_cast<std::size_t>(party), opt);
lock.lock();
s->busy = false;
if (s->cursor.done)
++finished;
else
{
s->ready_at = std::chrono::steady_clock::now()
+ std::chrono::microseconds(chaos_us(inst_i, party, step));
}
cv.notify_all();
}
};
std::vector<std::thread> pool;
pool.reserve(workers);
for (std::size_t i = 0; i < workers; ++i)
pool.emplace_back(worker);
for (auto & th : pool)
th.join();
if (finished != instances * 2)
throw std::runtime_error("run_fleet: not every side finished");
}
} // namespace app
} // namespace dpf
#endif