Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions bench/bench_block.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -155,10 +155,10 @@ static void bench(double sec) {
cout << "\n=== bools_to_bits / bits_to_bools (sweep N bits) ===\n";
for (int len : {32, 128, 1024, 8192, 65536}) {
vector<uint8_t> bools(len);
prg.random_bool(reinterpret_cast<bool *>(bools.data()), len);
prg.random_bool(bools.data(), len);
vector<uint8_t> packed((len + 7) / 8);
double calls = run_for(sec, [&]() {
bools_to_bits(packed.data(), reinterpret_cast<const bool *>(bools.data()), len);
bools_to_bits(packed.data(), bools.data(), len);
}, packed.data());
ostringstream lbl; lbl << "bools_to_bits(N=" << len << ")";
// Bandwidth: bytes of bool input read.
Expand All @@ -169,7 +169,7 @@ static void bench(double sec) {
prg.random_data_unaligned(packed.data(), (int)packed.size());
vector<uint8_t> bools(len);
double calls = run_for(sec, [&]() {
bits_to_bools(reinterpret_cast<bool *>(bools.data()), packed.data(), len);
bits_to_bools(bools.data(), packed.data(), len);
}, bools.data());
ostringstream lbl; lbl << "bits_to_bools(N=" << len << ")";
print_vec_bytes(lbl.str(), calls, (size_t)len);
Expand Down
13 changes: 6 additions & 7 deletions bench/bench_f2k.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -117,13 +117,12 @@ static void bench(double sec) {
vector<block> a(n);
vector<uint8_t> bs(n);
prg.random_block(a.data(), n);
prg.random_bool(reinterpret_cast<bool *>(bs.data()), n);
prg.random_bool(bs.data(), n);
block r;
double calls = run_for(sec, [&]() {
vector_inn_prdt_sum_red(&r, a.data(),
reinterpret_cast<const bool *>(bs.data()), n);
vector_inn_prdt_sum_red(&r, a.data(), bs.data(), n);
}, &r);
ostringstream lbl; lbl << "vec_inn_prdt_red(bool, N=" << n << ")";
ostringstream lbl; lbl << "vec_inn_prdt_red(byte-bool, N=" << n << ")";
print_vec(lbl.str(), calls, n);
}

Expand Down Expand Up @@ -174,12 +173,12 @@ static void bench(double sec) {
{
GaloisFieldPacking pkr;
uint8_t bits[128];
prg.random_bool(reinterpret_cast<bool *>(bits), 128);
prg.random_bool(bits, 128);
block r;
double calls = run_for(sec, [&]() {
pkr.packing(&r, reinterpret_cast<const bool *>(bits));
pkr.packing(&r, bits);
}, &r);
print_op("packing(bool*, 128)", calls);
print_op("packing(byte-bool*, 128)", calls);
}
}

Expand Down
1 change: 0 additions & 1 deletion bench/bench_netio.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,6 @@
// send_block / recv_block block-typed wrapper
// send_bool / recv_bool packed via bools_to_bits
// flush() drain outbound only (no peer coupling)
// sync() 1-byte ping/pong handshake
//
// Benchmark below runs the loopback throughput sweep only. Correctness and
// regression coverage lives in test/test_netio.cpp.
Expand Down
2 changes: 1 addition & 1 deletion bench/bench_prg.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -89,7 +89,7 @@ static void bench(double sec) {
for (int nb : {32, 128, 512, 2048, 8192, 32768, 131072}) {
vector<uint8_t> buf(nb);
double calls = run_for(sec, [&]() {
prg.random_bool(reinterpret_cast<bool *>(buf.data()), nb);
prg.random_bool(buf.data(), nb);
}, buf.data());
ostringstream lbl; lbl << "random_bool(N=" << nb << ")";
print_vec(lbl.str(), calls, (size_t)nb);
Expand Down
1 change: 0 additions & 1 deletion bench/bench_tlsio.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,6 @@
// send_block / recv_block block-typed wrapper (inherited)
// send_bool / recv_bool packed via bools_to_bits (inherited)
// flush() drain outbound coalescing buffer
// sync() 1-byte ping/pong handshake
//
// Same flush contract and thread-safety rules as NetIO. This benchmark runs
// the TLSIO loopback throughput sweep with the same PKI setup as the test.
Expand Down
10 changes: 7 additions & 3 deletions docs/api_conventions.md
Original file line number Diff line number Diff line change
Expand Up @@ -102,12 +102,16 @@ std::vector<uint8_t> // owning; the byte-bool codec returns this
const uint8_t* // + length; the byte-bool codec reads this
```

Each byte represents one bit and must be normalized to `0` or `1`.
Each input byte represents one bit: zero is false and any nonzero value
is true. APIs that produce byte-bools normalize their output to `0` or
`1`. The bit packing, packed-bool I/O, PRG, and GF bit-vector helpers
accept `uint8_t*` directly alongside their existing `bool*` overloads.

Do not use `std::vector<bool>` in emp-tool library/protocol code. It is
bit-packed, has proxy references, has no real `bool*`, and forces
hidden copies. Do not reinterpret byte-bool storage as `bool*`; convert
explicitly if an API requires real `bool` storage.
hidden copies. Do not reinterpret byte-bool storage as `bool*`; pass it
to a byte-bool overload, or convert explicitly if an API requires real
`bool` storage.

## Length and count parameters

Expand Down
50 changes: 40 additions & 10 deletions docs/io_channel.md
Original file line number Diff line number Diff line change
Expand Up @@ -69,9 +69,6 @@ are per-direction snapshots for diagnostics. All three assert that

## Other base surface

- **`sync()`**: optional 1-byte ping/pong handshake to confirm both
directions are alive. NetIO implements it; the base default is a
no-op.
- **Telemetry**: the base tracks `send_counter` / `recv_counter` /
`rounds` / `flushes_count`; `get_statistics_string()` renders them
for logging (`~NetIO` prints it unless constructed `quiet`).
Expand All @@ -87,17 +84,50 @@ are per-direction snapshots for diagnostics. All three assert that
`make_sibling()`, calling it **serially and in the same order on both
parties** (its accept/connect pairing is FIFO on the shared listener —
concurrent `make_sibling()` from multiple threads is not deterministic).
Do *not* rely on closing every channel and reopening a new one on the
same port as the coordination mechanism. That reopen path is supported
and race-free — each connection is only considered established once the
peer has actually `accept()`ed it (a one-byte accept acknowledgement in
`tcp_socket.h` protects against a `connect()` landing on the previous,
now-stale listener) — but the anchor + `make_sibling` pattern is simpler
and avoids the reconnect entirely.
Closing all channels and reopening the same port is supported, but sibling
channels avoid reconnecting.
- **`TraceIO`** (`trace_io.h`): an `IOChannel` that tees every wire
byte to `<prefix>.send` / `<prefix>.recv` files for diff-based
wire-equivalence checks; see `test_mode.md`.

## TCP socket buffers

`tcp::SocketOptions` sets the send and receive buffer sizes before
`listen()` or `connect()`:

```cpp
tcp::SocketOptions options;
options.send_buffer_size = 16 * 1024 * 1024;
options.receive_buffer_size = 16 * 1024 * 1024;

auto io = NetIO::connect(peer, port, options);
```

When the path capacity and round-trip time are known, the helper sizes both
directions to the first power-of-two tier at or above the bandwidth-delay
product, with a 256 KiB minimum:

```cpp
using namespace std::chrono_literals;

auto options = tcp::SocketOptions::for_bandwidth_and_rtt(
400'000'000, 100ms); // 8 MiB
auto io = NetIO::connect(peer, port, options);
```

Use the direct byte fields when the two directions need different sizes.

Pass the same options to `NetIO::listen`; `make_sibling()` propagates them to
each new connection. `TLSIO` takes them through `TLSConfig::socket_options`.
A listening endpoint applies the options before `listen()` and verifies them
on each accepted socket. TLSIO applies the options before the TLS handshake.
The adopted-socket TLS constructor requires default socket options.
A zero size keeps the operating-system default. An explicit request fails if
the kernel caps the buffer below the requested size. Larger requests may require
raising `net.core.wmem_max` / `net.core.rmem_max` on Linux or
`kern.ipc.maxsockbuf` on macOS. On Linux, setting `SO_RCVBUF` disables TCP
receive-buffer autotuning for that socket.

## TLS variant

`TLSIO` (in `emp-tool/runtime/io/tls_io_channel.h`) is another `IOChannel`
Expand Down
25 changes: 14 additions & 11 deletions docs/test_mode.md
Original file line number Diff line number Diff line change
Expand Up @@ -56,17 +56,18 @@ emp::set_test_mode(true); // before any PRG() default-construction
```

The env var is read once at first call to `is_test_mode()` and
cached. `set_test_mode()` overrides it programmatically.
cached. Only the exact value `1` enables test mode; other values leave it off.
`set_test_mode()` overrides it programmatically.

The first activation by either mechanism prints a prominent warning to
`stderr`, once per process, that default PRG seeds and EC scalar randomness are
deterministic and insecure. The warning happens at activation rather than on
each random draw, so it adds no work to the randomness hot path. Never process
real secrets in a process running in test mode.

`reset_test_seed_counter()` rewinds every lane's ordinal and
releases lane 0 — call it between independent test iterations to
get reproducible PRG sequences within one process.
`reset_test_seed_counter()` rewinds every lane's ordinal and releases lane 0.
Call it between independent test iterations, after joining threads and draining
pool futures, to get reproducible PRG sequences within one process.

## Multi-threading: lanes

Expand All @@ -88,9 +89,12 @@ order is deterministic), never discovered by the worker itself.
});
```

- **Forgetting is loud.** A second thread drawing from lane 0 would
replay the main thread's streams byte-for-byte — silently wrong —
so test mode aborts with a pointer to this document instead.
Manually assigned lane ids must be nonzero and unique among concurrently
active work. Lane 0 is reserved for the unscoped main thread.

- **Forgetting is loud.** A second thread drawing from lane 0 or deriving a
child lane from it would replay deterministic streams byte-for-byte, so test
mode aborts with a pointer to this document instead.

Lanes make the *randomness* reproducible. Byte-identical *traces*
additionally require that each traced channel has a single writer
Expand Down Expand Up @@ -124,10 +128,9 @@ in production paths.

`TraceIO` wraps any `IOChannel*` and writes a copy of every wire
byte to two files: `<prefix>.send` and `<prefix>.recv`. Bytes are
delivered to the underlying channel either before (recv) or
synchronously (send) with the file write, so a crash mid-write
leaves a trace prefix that still matches what the peer didn't yet
see.
copied before outbound delivery and after inbound delivery. Trace
files are created with mode `0600`; `TraceIO::flush()` flushes both
files before flushing the wrapped channel.

```cpp
NetIO* under = new NetIO(...);
Expand Down
2 changes: 2 additions & 0 deletions emp-tool/runtime/core/block.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,8 @@ inline block set_bit(const block & a, int i) {
}

inline std::string to_hex(const void* data, size_t n) {
expecting(n <= std::string{}.max_size() / 2,
"to_hex: input too large");
static const char digits[] = "0123456789abcdef";
const unsigned char* b = static_cast<const unsigned char*>(data);
std::string s(2 * n, '0');
Expand Down
1 change: 1 addition & 0 deletions emp-tool/runtime/core/error.h
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@ inline void error(const char *s,
int line = __builtin_LINE(),
const char *file = __builtin_FILE()) {
std::fprintf(stderr, "%s at %s:%d\n", s, file, line);
std::fflush(stderr);
// _Exit, not exit(): error() can fire from a worker thread while sibling
// workers still own heap state. Running destructors/atexit handlers in that
// situation races their live work; terminate the process immediately.
Expand Down
3 changes: 2 additions & 1 deletion emp-tool/runtime/core/simd_tier.h
Original file line number Diff line number Diff line change
@@ -1,7 +1,6 @@
#ifndef EMP_SIMD_TIER_H__
#define EMP_SIMD_TIER_H__

#include "emp-tool/runtime/core/block.h"
#include <cstdint>

// Centralized SIMD tier detection + Lane abstractions used by emp-tool's
Expand Down Expand Up @@ -101,6 +100,8 @@
//
// AesLane<N> lives in emp:: (public). ClmulLane<N> lives in emp::detail.

#include "emp-tool/runtime/core/block.h"

#ifdef __x86_64__

namespace emp {
Expand Down
59 changes: 34 additions & 25 deletions emp-tool/runtime/core/test_mode.h
Original file line number Diff line number Diff line change
Expand Up @@ -18,16 +18,18 @@
// depend only on per-lane program order, never on cross-thread
// scheduling, so multi-threaded runs reproduce. ThreadPool::enqueue
// derives and installs a lane per task automatically; hand-spawned
// threads wrap their body in test_lane_scope. A second thread drawing
// from lane 0 aborts: two threads sharing a lane would replay identical
// "random" streams — silently wrong rather than merely nondeterministic.
// threads wrap their body in test_lane_scope. A second thread using lane
// 0 to draw a seed or derive a child lane aborts: two threads sharing a
// lane would replay identical "random" streams — silently wrong rather
// than merely nondeterministic.

#include "emp-tool/runtime/core/error.h"

#include <atomic>
#include <cstdint>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <functional>
#include <thread>

Expand Down Expand Up @@ -58,7 +60,7 @@ inline std::atomic<bool>& test_mode_flag() {
static std::atomic<bool> flag(
[]() {
const char* v = std::getenv("EMP_TEST_MODE");
const bool enabled = v != nullptr && v[0] == '1';
const bool enabled = v != nullptr && std::strcmp(v, "1") == 0;
if (enabled) warn_insecure_test_mode_once();
return enabled;
}());
Expand Down Expand Up @@ -99,16 +101,31 @@ inline void sync_test_epoch(TestSeedTls& s) {
}
}

// Owner token of lane 0: the one thread allowed to draw main-lane
// seeds. Cleared by reset_test_seed_counter(), so sequential
// Owner token of lane 0: the one thread allowed to draw main-lane seeds or
// derive child lanes. Cleared by reset_test_seed_counter(), so sequential
// independent units may run on different threads.
inline std::atomic<uint64_t>& lane0_owner() {
static std::atomic<uint64_t> owner(0);
return owner;
}
inline std::atomic<uint64_t>& next_thread_token() {
static std::atomic<uint64_t> next(1);
return next;
}
inline uint64_t this_thread_token() {
// Nonzero hash of the thread id; 0 is the "unowned" sentinel.
return (uint64_t)std::hash<std::thread::id>()(std::this_thread::get_id()) | 1ULL;
thread_local const uint64_t token =
next_thread_token().fetch_add(1, std::memory_order_relaxed);
expecting(token != 0, "test mode: thread token space exhausted");
return token;
}
inline void claim_lane0() {
auto& owner = lane0_owner();
const uint64_t token = this_thread_token();
uint64_t expected = 0;
expecting(owner.compare_exchange_strong(expected, token) ||
expected == token,
"test mode: a second thread used lane 0; run spawned work "
"under emp::test_lane_scope (see docs/test_mode.md)");
}

// splitmix64 finalizer: full-avalanche 64-bit mix for deriving child
Expand Down Expand Up @@ -149,19 +166,7 @@ struct TestSeed {
inline TestSeed next_test_seed() {
auto& s = detail::test_seed_tls();
detail::sync_test_epoch(s);
if (s.lane == 0) {
// Only one thread may consume main-lane seeds; a second one
// would replay the same streams. Always-on: test mode usually
// runs under Release/NDEBUG builds.
auto& owner = detail::lane0_owner();
const uint64_t token = detail::this_thread_token();
uint64_t expected = 0;
expecting(owner.compare_exchange_strong(expected, token) ||
expected == token,
"test mode: a second thread drew lane-0 randomness; run "
"spawned work under emp::test_lane_scope (see "
"docs/test_mode.md)");
}
if (s.lane == 0) detail::claim_lane0();
return {s.lane, s.ctr++};
}

Expand All @@ -174,6 +179,7 @@ inline TestSeed next_test_seed() {
inline uint64_t next_test_child_lane() {
auto& s = detail::test_seed_tls();
detail::sync_test_epoch(s);
if (s.lane == 0) detail::claim_lane0();
uint64_t lane = detail::mix64(detail::mix64(s.lane) ^ s.child_ctr++);
if (lane == 0) lane = 1; // 0 is reserved for the main thread
return lane;
Expand All @@ -186,6 +192,8 @@ inline uint64_t next_test_child_lane() {
class test_lane_scope {
public:
explicit test_lane_scope(uint64_t lane) : saved_(detail::test_seed_tls()) {
expecting(lane != 0,
"test_lane_scope: lane 0 is reserved for the main thread");
auto& s = detail::test_seed_tls();
s.lane = lane;
s.ctr = 0;
Expand All @@ -207,10 +215,11 @@ inline uint64_t current_test_seed_epoch() {
return detail::test_seed_epoch().load();
}

// Rewind every lane's draw ordinal (lazily, when each thread next
// draws) and release lane 0. Use before each independent unit (e.g.
// each protocol in a trace) to make that unit's randomness -- and thus
// its wire bytes -- independent of whatever consumed seeds before it.
// Rewind every lane's draw ordinal (lazily, when each thread next draws) and
// release lane 0. All work that can draw seeds or derive child lanes must be
// quiescent first. Use before each independent unit (e.g. each protocol in a
// trace) to make that unit's randomness -- and thus its wire bytes --
// independent of whatever consumed seeds before it.
inline void reset_test_seed_counter() {
detail::test_seed_epoch().fetch_add(1);
detail::lane0_owner().store(0);
Expand Down
Loading