diff --git a/Cargo.toml b/Cargo.toml index 4a27a33..5e72780 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "WorkTablesIndex" -version = "0.0.12" +version = "0.0.13" edition = "2021" documentation = "https://docs.rs/WorkTablesIndex/" repository = "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/pathscale/WorkTablesIndex" @@ -12,11 +12,25 @@ readme = "README.md" [dev-dependencies] criterion = { version = "0.5", features = ["html_reports"] } -crossbeam-skiplist = "^0.1" +crossbeam-skiplist = "0.1" loom = "0.7" rand = "0.9" scc = { version = "2.2.5" } uuid = { version = "1.17.0", features = ["v7"]} +# For benches/locks.rs, which prices this crate's lock choice. `lock_api` is the +# layer both parking_lot and spin are built on, so one generic body measures +# both, and `libc` reads CPU time, which wall clock cannot show. +lock_api = { version = "0.4", features = ["arc_lock"] } +spin = { version = "0.12", default-features = false, features = ["spin_mutex", "rwlock", "lock_api", "std"] } +libc = "0.2" +# The std column: a real futex, which is the primitive parking_lot parks on. +atomic-wait = "1.1" +# parking_lot is optional in this crate (the `concurrent` feature). The bench +# measures it against spin regardless of which features the library is built +# with, so it needs its own unconditional copy. Named `parking_lot_upstream` +# because the library's own `parking_lot` is now the no_std fork, and one name +# cannot mean both: this arm is deliberately the real one, with its std futex. +parking_lot_upstream = { package = "parking_lot", version = "0.12", features = ["arc_lock"] } # The crate is a fork of `indexset`, published under a different package name # to avoid colliding with upstream on the registry, and consumed everywhere as @@ -28,29 +42,43 @@ uuid = { version = "1.17.0", features = ["v7"]} name = "indexset" [dependencies] -serde = { version = "1", optional = true, features = ["derive"] } -ftree = { version = "1", features = ["serde"] } -parking_lot = { version = "0.12", features = [ +serde = { version = "1", optional = true, default-features = false, features = ["derive", "alloc"] } +# `ftree` is no_std, but its `serde` feature takes serde with default features, +# which is `std`. Asking for it unconditionally made every build a std build, +# so it now arrives with this crate's own `serde` feature and nowhere else. +ftree = { version = "1", default-features = false } +# `parking_lot_lite_hack` is parking_lot's `Mutex` and `RwLock` with `std` made +# optional, kept under the `parking_lot` name because that is what the code +# says. This crate uses `Mutex`, `RwLock` and `RawMutex` and nothing the fork +# left behind. +parking_lot = { package = "parking_lot_lite_hack", version = "0.12.6", default-features = false, features = [ "send_guard", "arc_lock", ], optional = true } -ps-reclaim = { version = "^0.1, >=0.1.3", optional = true } -fastrand = { version = "2", optional = true } +ps-reclaim = { version = "0.1.4", optional = true, default-features = false } +fastrand = { version = "2", optional = true, default-features = false } superslice = { version = "1", optional = true } wt-slice = { version = "0.1", optional = true } [features] -default = ["wt-slice-binary-search"] -serde = ["dep:serde"] +default = ["std", "wt-slice-binary-search"] +# Off, this crate does not link `std`. What it costs: `superslice` as a search +# strategy, and `yield_now` in the two publication spin loops, which fall back +# to a spin because there is no scheduler to yield to. +std = ["fastrand?/std", "parking_lot?/std", "ps-reclaim?/std", "serde?/std"] +serde = ["dep:serde", "ftree/serde", "std"] concurrent = [ "dep:parking_lot", "dep:ps-reclaim", + "ps-reclaim/libc", + "ps-reclaim/spin", ] cdc = ["concurrent"] multimap = ["concurrent", "dep:fastrand"] custom-binary-search = [] std-binary-search = [] -superslice-binary-search = ["dep:superslice"] +# `superslice` is not a no_std crate, so choosing it chooses `std`. +superslice-binary-search = ["dep:superslice", "std"] wt-slice-binary-search = ["dep:wt-slice"] [package.metadata.docs.rs] @@ -111,3 +139,7 @@ unnecessary_unwrap = "allow" useless_conversion = "allow" useless_vec = "allow" while_let_loop = "allow" + +[[bench]] +name = "locks" +harness = false diff --git a/benches/locks.rs b/benches/locks.rs new file mode 100644 index 0000000..f2f496a --- /dev/null +++ b/benches/locks.rs @@ -0,0 +1,601 @@ +//! What this crate's node locking costs, four ways. +//! +//! # Why this exists +//! +//! `concurrent` is the only reason this crate needs `std`. Its 31 lock sites +//! are `parking_lot`, and `cdc` and `multimap` both imply `concurrent`, so a +//! consumer wanting only `ChangeEvent` and `Pair` takes `parking_lot` and +//! `libc` with them. DataBucket is exactly that consumer, and that is what +//! stops its format layer from being `no_std`. +//! +//! # The four arms +//! +//! | arm | what it is | +//! |---|---| +//! | `parking_lot` | `parking_lot_upstream::Mutex`, named directly. What this crate uses today. | +//! | `spin` | `spin::Mutex`, named directly. | +//! | `lock_api+parking_lot` | `lock_api::Mutex` | +//! | `lock_api+spin` | `lock_api::Mutex, T>` | +//! +//! The two `lock_api` arms are one generic body instantiated twice. That is the +//! proposed shape: `concurrent` goes generic over `R: RawMutex` and the +//! consumer chooses, rather than this crate naming a lock for everyone. +//! `lock_arc` and `ArcMutexGuard` belong to `lock_api`, not to +//! `parking_lot`, so `Ref` keeps working either way. +//! +//! The two direct arms exist to price the `lock_api` wrapper itself. If +//! `parking_lot` and `lock_api+parking_lot` differ, the wrapper is not free and +//! every other comparison here is contaminated. +//! +//! # The two regimes +//! +//! `nodes` is this crate's shape from `src/concurrent/set.rs`: +//! `RwLock>>>`, with `DEFAULT_INNER_SIZE` entries +//! per node. **1024, not the 16 an earlier version of this file guessed.** That +//! was wrong by 64x and it decides the answer, because a lock held over 16 +//! entries is short and favours spinning while one held over 1024 does not. +//! +//! `contended` is one lock held long under oversubscription, where a spinlock +//! is supposed to lose: a waiter burns the core the holder needs to finish. +//! +//! # Reading it +//! +//! **CPU time, from `getrusage`.** Wall clock cannot see a burnt core. A +//! spinlock finishes sooner while consuming more, and on a machine with spare +//! cores that looks free until something else wants one. +//! +//! **`null` is the first arm run twice.** Whatever it shows is what this +//! harness reports for identical code, so nothing closer than its distance +//! from 1.00x means anything. + +use std::collections::BTreeMap; +use std::hint::black_box; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use indexset::core::constants::DEFAULT_INNER_SIZE; + +/// Nodes in the index, enough that lookups spread rather than queue on one. +const NODES: u64 = 4_096; +/// Lookups per measurement, split across threads. +const LOOKUPS: usize = 200_000; +/// How long the contended case holds. Long enough for a holder to be preempted +/// inside it, short enough to keep this bounded. +const HELD_ITERS: u64 = 20_000; +/// Acquisitions per thread in the contended case. +const HOLDS: usize = 40; +const REPS: usize = 5; + +/// CPU consumed by this process, user plus system. +fn cpu() -> Duration { + // SAFETY: `getrusage` fully initialises the `rusage` it is given for + // `RUSAGE_SELF`, and returns non-zero rather than writing on failure. + let mut usage: libc::rusage = unsafe { std::mem::zeroed() }; + let ok = unsafe { libc::getrusage(libc::RUSAGE_SELF, &raw mut usage) }; + assert_eq!(ok, 0, "getrusage failed"); + let secs = |t: libc::timeval| Duration::new(t.tv_sec as u64, (t.tv_usec as u32).saturating_mul(1_000)); + secs(usage.ru_utime) + secs(usage.ru_stime) +} + +/// A node the size this crate actually uses. +struct Node { + keys: Vec, +} + +impl Node { + fn new(seed: u64) -> Self { + let mut keys: Vec = (0..DEFAULT_INNER_SIZE as u64) + .map(|n| seed.wrapping_mul(0x9e37_79b9).wrapping_add(n)) + .collect(); + keys.sort_unstable(); + Self { keys } + } + + /// A sorted search then a write, which is what a node access is. + fn touch(&mut self, n: u64) -> u64 { + let at = self.keys.partition_point(|key| *key < n).min(self.keys.len() - 1); + self.keys[at] ^= n; + self.keys[at] + } +} + +fn spread(worker: usize) -> u64 { + 0x2545_F491_4F6C_DD1Du64 ^ (worker as u64).wrapping_mul(0x9E37_79B9) +} + +fn next(rng: &mut u64) -> u64 { + *rng ^= *rng << 13; + *rng ^= *rng >> 7; + *rng ^= *rng << 17; + *rng % NODES +} + +/// The index lock is the same in every arm. +/// +/// Only the node lock varies. An earlier version changed both at once, which +/// conflated them: a difference could have come from either, and the two +/// tables could not be read against each other. +type IndexLock = parking_lot_upstream::RwLock>>; + +macro_rules! nodes_arm { + ($name:ident, $mx:ty) => { + fn $name(threads: usize) -> (Duration, Duration) { + let index: Arc> = Arc::new(IndexLock::<$mx>::new( + (0..NODES) + .map(|n| (n, Arc::new(<$mx>::new(Node::new(n))))) + .collect(), + )); + let before = cpu(); + let now = Instant::now(); + std::thread::scope(|scope| { + for worker in 0..threads { + let index = index.clone(); + scope.spawn(move || { + let mut acc = 0u64; + let mut rng = spread(worker); + for _ in 0..LOOKUPS / threads { + let key = next(&mut rng); + // The index lock is released before the node lock + // is taken, which is what the real code does. + let node = index.read().get(&key).cloned(); + if let Some(node) = node { + acc ^= node.lock().touch(rng); + } + } + black_box(acc); + }); + } + }); + (now.elapsed(), cpu() - before) + } + }; +} + +macro_rules! contended_arm { + ($name:ident, $mx:ty) => { + fn $name(threads: usize) -> (Duration, Duration) { + let lock: Arc<$mx> = Arc::new(<$mx>::new(0u64)); + let before = cpu(); + let now = Instant::now(); + std::thread::scope(|scope| { + for _ in 0..threads { + let lock = lock.clone(); + scope.spawn(move || { + for _ in 0..HOLDS { + let mut held = lock.lock(); + for n in 0..HELD_ITERS { + *held = held.wrapping_mul(0x9e37_79b9).wrapping_add(n); + } + } + }); + } + }); + (now.elapsed(), cpu() - before) + } + }; +} + +/// A lock that spins for a bounded budget, then gives the core up. +/// +/// `spin::relax::Yield` yields on every iteration, so it trades cycles for +/// syscalls and still costs 10x parking_lot's CPU when a lock is held long. +/// `RelaxStrategy` is stateless, so the budget cannot live there. +/// +/// This is what parking_lot does, minus the parking: spin while the holder is +/// plausibly about to finish, then stop competing with it for the core. The +/// give-up step is the only part that needs a platform, which is why it is one +/// call and not a design. +pub struct Bounded; + +/// parking_lot's spin schedule, copied from `parking_lot_core::SpinWait`. +/// +/// Reading it was overdue. It is not "spin N times then yield": it is three +/// exponentially growing pauses, then seven yields, then give up and park. +/// +/// ```text +/// counter 1..=3 cpu_relax(1 << counter) 2, 4, 8 pauses +/// counter 4..=10 yield to the scheduler +/// counter >10 stop spinning, park +/// ``` +/// +/// The first version here spun 64 times flat and then yielded forever, which +/// is why it never stopped burning CPU. Fourteen attempts, not sixty-four, and +/// a hard end to them. +struct SpinWait { + counter: u32, +} + +impl SpinWait { + const fn new() -> Self { + Self { counter: 0 } + } + + /// Returns false once spinning has stopped being worth it. + fn spin(&mut self) -> bool { + if self.counter >= 10 { + return false; + } + self.counter += 1; + if self.counter <= 3 { + for _ in 0..(1u32 << self.counter) { + core::hint::spin_loop(); + } + } else { + give_up_the_core(); + } + true + } +} + +pub struct BoundedMutex { + locked: core::sync::atomic::AtomicBool, +} + +unsafe impl lock_api::RawMutex for BoundedMutex { + const INIT: Self = Self { + locked: core::sync::atomic::AtomicBool::new(false), + }; + type GuardMarker = lock_api::GuardSend; + + fn lock(&self) { + // parking_lot's schedule exactly, minus the park at the end, because + // there is nothing to park on without an operating system. So when the + // schedule runs out this keeps yielding: that residue is the price of + // no_std, and the table measures it. + let mut spinwait = SpinWait::new(); + loop { + if self.try_lock() { + return; + } + if !spinwait.spin() { + spinwait = SpinWait::new(); + give_up_the_core(); + } + } + } + + fn try_lock(&self) -> bool { + self.locked + .compare_exchange_weak( + false, + true, + core::sync::atomic::Ordering::Acquire, + core::sync::atomic::Ordering::Relaxed, + ) + .is_ok() + } + + unsafe fn unlock(&self) { + self.locked.store(false, core::sync::atomic::Ordering::Release); + } +} + +/// The one platform call. On a target with no scheduler this is a spin, and +/// the lock degrades to `spin::Mutex` rather than breaking. +fn give_up_the_core() { + std::thread::yield_now(); +} + +/// A lock whose waiters idle the core, with no operating system. +/// +/// # Why this instruction exists and why nothing emits it +/// +/// `WFE` puts the core in a low-power state until its event register is set; +/// `SEV` sets it on every core. ARM documents this as **the** intended spinlock +/// construction: the waiter executes WFE to request a low-power state and the +/// releaser executes SEV to wake it. +/// +/// Rust does not emit it. `core::hint::spin_loop()` on aarch64 is +/// `__isb(SY)`, an instruction barrier, verified by disassembly: +/// +/// ```text +/// __RNvCs8LfLpYhzmc_7hintasm1s: +/// isb +/// ret +/// ``` +/// +/// There is no WFE anywhere in `core::hint`, and no safe wrapper in +/// `core::arch::aarch64`. That is structural rather than an oversight: WFE is +/// only useful if the *releaser* pairs it with SEV, which is a protocol between +/// both sides of a lock, and a one-sided hint like `spin_loop()` cannot express +/// one. So it has to be written here, in the lock, where both sides are known. +/// +/// # What it costs, measured on this machine +/// +/// ```text +/// nop 0.3 ns +/// isb (spin_loop) 8.6 ns +/// wfe, event pending 1336.7 ns +/// ``` +/// +/// WFE is not a no-op in userspace here, and it does not block forever either: +/// it idles for about 1.3 us and returns on its own. So a waiter needs no SEV +/// to make progress, and burns roughly 150x less CPU per unit of wall time +/// waited than an `isb` spin. SEV is still sent on unlock, because waking +/// immediately beats waiting out the timeout. +/// +/// # Why this is the interesting arm rather than a curiosity +/// +/// Every other `no_std` arm in this file fails the same way: when its spin +/// schedule runs out there is nothing to hand the core to, so the waiter keeps +/// running and keeps burning. Karlin et al.'s competitive-spinning result says +/// spin-then-block is 2-competitive, and the "block" half is exactly what a +/// target without an operating system cannot do. WFE is the hardware answering +/// that: a block with no scheduler involved. +/// +/// The problem is current, not settled. HTLL (IEEE TPDS, January 2025) targets +/// throughput and latency together under oversubscription, reporting up to 97% +/// latency reduction for about 5% throughput; Fissile Locks (arXiv 2003.05025) +/// is compact, NUMA-aware and preemption-tolerant; Asymmetry-aware Scalable +/// Locking (arXiv 2108.03355) matters here specifically, because this is a +/// P-core/E-core machine and this benchmark does not separate them. +/// +/// # When to use it, measured rather than assumed +/// +/// **WFE idles the core. It does not yield to the operating system.** That is +/// the whole rule, and it was learned the expensive way here: with 128 threads +/// on 16 cores this arm costs 5478 ms of CPU against 1168 for a yielding +/// spinner, because a waiter idling in WFE is still a scheduled thread, so the +/// lock holder still cannot get a core. An earlier version of this lock +/// restarted its spin schedule after every WFE, which meant it yielded between +/// idles, and that accident is what made it competitive. +/// +/// So the case for WFE is: +/// +/// * threads at most cores, so idling a core costs nothing that is wanted; +/// * no operating system to yield to, which is when every other option here +/// reduces to burning the core anyway; +/// * long enough waits that 1.3 us of idle is better than 8.6 ns of `isb` +/// repeated until the holder finishes. +/// +/// That is bare metal and pinned threads, not an oversubscribed server. Under +/// oversubscription yielding beats idling, and this arm is the wrong choice. +/// +/// # And the reason to care beyond this crate +/// +/// EKOPathRS is a compiler. A compiler that recognises a spin loop can emit +/// WFE for it, which is a transformation LLVM does not perform and which the +/// measurement above prices at 150x. That makes this arm a bet on the toolchain +/// rather than only a lock experiment. +pub struct WfeMutex { + locked: core::sync::atomic::AtomicBool, +} + +unsafe impl lock_api::RawMutex for WfeMutex { + const INIT: Self = Self { + locked: core::sync::atomic::AtomicBool::new(false), + }; + type GuardMarker = lock_api::GuardSend; + + fn lock(&self) { + // Spin briefly first: an uncontended lock should never reach a 1.3 us + // instruction, and a short hold is over before the schedule ends. + let mut spinwait = SpinWait::new(); + while spinwait.spin() { + if self.try_lock() { + return; + } + } + // The schedule is spent, so stop competing with the holder for its + // core and idle instead. **Do not restart the schedule**: an earlier + // version reset it after every WFE, so it went back to yielding + // between idles and measured the same as not using WFE at all. + loop { + if self.try_lock() { + return; + } + #[cfg(target_arch = "aarch64")] + // SAFETY: WFE is unprivileged, has no memory operands, and no + // effect beyond waiting on the event register. + unsafe { + core::arch::asm!("wfe", options(nomem, nostack)) + }; + #[cfg(not(target_arch = "aarch64"))] + give_up_the_core(); + } + } + + fn try_lock(&self) -> bool { + self.locked + .compare_exchange_weak( + false, + true, + core::sync::atomic::Ordering::Acquire, + core::sync::atomic::Ordering::Relaxed, + ) + .is_ok() + } + + unsafe fn unlock(&self) { + self.locked.store(false, core::sync::atomic::Ordering::Release); + // Wake every waiting core now rather than letting each wait out its + // own timeout. + #[cfg(target_arch = "aarch64")] + // SAFETY: SEV sets the event register on every core and has no other + // effect. + unsafe { + core::arch::asm!("sev", options(nomem, nostack)) + }; + } +} + +/// The same bounded spin, but it **blocks** instead of yielding. +/// +/// This is the std column, and the difference is the whole point: a yielding +/// waiter still runs, so it still burns CPU. A blocked waiter consumes +/// nothing. That is what parking_lot does and it is why nothing without an +/// operating system can match it. +/// +/// The classic three-state futex mutex: 0 free, 1 locked, 2 locked and +/// somebody is asleep on it. The third state exists so `unlock` can skip the +/// wake syscall when nobody is waiting, which is the common case. +pub struct FutexMutex { + state: core::sync::atomic::AtomicU32, +} + +const FREE: u32 = 0; +const HELD: u32 = 1; +const CONTENDED: u32 = 2; + +unsafe impl lock_api::RawMutex for FutexMutex { + const INIT: Self = Self { + state: core::sync::atomic::AtomicU32::new(FREE), + }; + type GuardMarker = lock_api::GuardSend; + + fn lock(&self) { + use core::sync::atomic::Ordering; + // parking_lot's schedule, then a real park. The earlier version spun a + // flat 64 and then ran a CAS dance that re-announced every waiter on + // every wake, so unlock woke somebody on every release forever. It + // measured worse than a plain spinlock, which is not something a + // blocking lock can honestly do. + let mut spinwait = SpinWait::new(); + while spinwait.spin() { + if self.try_lock() { + return; + } + } + + // Drepper's three-state mutex. The first version of this swapped + // CONTENDED unconditionally on every retry, which republished the + // waiter flag after each wake and had every sleeper re-announce itself: + // it measured worse than a plain spinlock, which is how the bug was + // found rather than by reading it. + let mut seen = self + .state + .compare_exchange(FREE, HELD, Ordering::Acquire, Ordering::Relaxed) + .unwrap_or_else(|seen| seen); + while seen != FREE { + // Mark contention once, then sleep on that exact value. + if seen != CONTENDED + && self + .state + .compare_exchange(HELD, CONTENDED, Ordering::Relaxed, Ordering::Relaxed) + .is_err() + && self.state.load(Ordering::Relaxed) == FREE + { + seen = FREE; + continue; + } + atomic_wait::wait(&self.state, CONTENDED); + seen = self + .state + .compare_exchange(FREE, CONTENDED, Ordering::Acquire, Ordering::Relaxed) + .unwrap_or_else(|seen| seen); + } + } + + fn try_lock(&self) -> bool { + use core::sync::atomic::Ordering; + self.state + .compare_exchange(FREE, HELD, Ordering::Acquire, Ordering::Relaxed) + .is_ok() + } + + unsafe fn unlock(&self) { + use core::sync::atomic::Ordering; + // Only wake if somebody actually slept. The uncontended path is a + // single store and no syscall. + if self.state.swap(FREE, Ordering::Release) == CONTENDED { + atomic_wait::wake_one(&self.state); + } + } +} + +type LaPlMx = lock_api::Mutex; +type LaSpMx = lock_api::Mutex, Node>; +type LaYdMx = lock_api::Mutex, Node>; + +contended_arm!(held_park, lock_api::Mutex); +contended_arm!(held_spin, lock_api::Mutex, u64>); +contended_arm!( + held_yield, + lock_api::Mutex, u64> +); +contended_arm!(held_bounded, lock_api::Mutex); +// Wired but not in NAMES: this futex arm is a known-broken implementation, +// kept so nobody writes it a third time. It loses to a spinning lock, which a +// blocking lock cannot honestly do. +#[allow(dead_code)] +mod broken_futex_arm { + use super::*; + contended_arm!(held_futex, lock_api::Mutex); + nodes_arm!(nodes_futex, lock_api::Mutex); +} +contended_arm!(held_wfe, lock_api::Mutex); + +nodes_arm!(nodes_park, LaPlMx); +nodes_arm!(nodes_spin, LaSpMx); +nodes_arm!(nodes_yield, LaYdMx); +nodes_arm!(nodes_bounded, lock_api::Mutex); +nodes_arm!(nodes_wfe, lock_api::Mutex); + +/// The same five, named once and used by both tables. +const NAMES: [&str; 5] = ["parking_lot", "spin", "spin+yield", "bounded", "wfe"]; + +type Arm = fn(usize) -> (Duration, Duration); + +fn median(mut runs: Vec<(Duration, Duration)>) -> (Duration, Duration) { + runs.sort_by_key(|(wall, _)| *wall); + runs[REPS / 2] +} + +fn table(title: &str, threads: &[usize], names: [&str; 5], arms: [Arm; 5]) { + println!("\n{title}"); + print!(" "); + for name in names { + print!("{name:>16}"); + } + println!("{:>9}", "null"); + print!(" threads"); + for _ in names { + print!("{:>8}{:>8}", "wall", "cpu"); + } + println!("{:>9}", ""); + for &thread_count in threads { + let mut runs: Vec> = vec![Vec::new(); 6]; + for _ in 0..REPS { + for (slot, arm) in arms.iter().enumerate() { + runs[slot].push(arm(thread_count)); + } + // The null arm: the first one again, measured under another name. + runs[5].push(arms[0](thread_count)); + } + let m: Vec<(Duration, Duration)> = runs.into_iter().map(median).collect(); + let ms = |d: Duration| d.as_secs_f64() * 1e3; + println!( + " {thread_count:>7}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.2}x", + ms(m[0].0), + ms(m[0].1), + ms(m[1].0), + ms(m[1].1), + ms(m[2].0), + ms(m[2].1), + ms(m[3].0), + ms(m[3].1), + ms(m[4].0), + ms(m[4].1), + m[0].0.as_secs_f64() / m[5].0.as_secs_f64(), + ); + } +} + +fn main() { + let cores = std::thread::available_parallelism().map_or(8, std::num::NonZeroUsize::get); + println!("\n{cores} cores, median of {REPS}, nodes of {DEFAULT_INNER_SIZE} entries, ms"); + + table( + &format!("this crate's shape, {LOOKUPS} lookups over {NODES} nodes"), + &[1, 2, 4, cores, cores * 2], + NAMES, + [nodes_park, nodes_spin, nodes_yield, nodes_bounded, nodes_wfe], + ); + + table( + &format!("one lock, held {HELD_ITERS} iterations, {HOLDS} times per thread"), + &[cores / 4, cores / 2, cores, cores * 2, cores * 8], + NAMES, + [held_park, held_spin, held_yield, held_bounded, held_wfe], + ); +} diff --git a/src/concurrent/map.rs b/src/concurrent/map.rs index 90f92c7..56d585f 100644 --- a/src/concurrent/map.rs +++ b/src/concurrent/map.rs @@ -1,5 +1,8 @@ -use std::fmt::{Debug, Display, Formatter}; -use std::{borrow::Borrow, iter::FusedIterator, ops::RangeBounds}; +use ::core::borrow::Borrow; +use ::core::fmt::{Debug, Display, Formatter}; +use ::core::iter::FusedIterator; +use ::core::ops::RangeBounds; +use alloc::vec::Vec; use super::set::BTreeSet; use crate::core::node::NodeLike; @@ -31,7 +34,7 @@ pub enum TopologyError { #[cfg(feature = "cdc")] impl Display for TopologyError { - fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + fn fmt(&self, formatter: &mut Formatter<'_>) -> ::core::fmt::Result { match self { Self::ZeroNodeCapacity => formatter.write_str("topology node capacity must be non-zero"), Self::EmptyNode { index } => write!(formatter, "topology node {index} is empty"), @@ -51,7 +54,7 @@ impl Display for TopologyError { } #[cfg(feature = "cdc")] -impl std::error::Error for TopologyError {} +impl ::core::error::Error for TopologyError {} #[derive(Debug)] pub struct BTreeMap>> diff --git a/src/concurrent/multimap.rs b/src/concurrent/multimap.rs index b166582..7f8f451 100644 --- a/src/concurrent/multimap.rs +++ b/src/concurrent/multimap.rs @@ -1,10 +1,9 @@ -use std::fmt::Debug; -use std::marker::PhantomData; -use std::{ - borrow::Borrow, - iter::FusedIterator, - ops::{Bound, RangeBounds}, -}; +use ::core::borrow::Borrow; +use ::core::fmt::Debug; +use ::core::iter::FusedIterator; +use ::core::marker::PhantomData; +use ::core::ops::{Bound, RangeBounds}; +use alloc::vec::Vec; use crate::core::node::NodeLike; use crate::{ diff --git a/src/concurrent/operation.rs b/src/concurrent/operation.rs index 83ea784..71c3d39 100644 --- a/src/concurrent/operation.rs +++ b/src/concurrent/operation.rs @@ -1,5 +1,7 @@ -use std::fmt::Debug; -use std::sync::Arc; +use ::core::fmt::Debug; +use alloc::sync::Arc; +use alloc::vec; +use alloc::vec::Vec; use parking_lot::RwLock; diff --git a/src/concurrent/ref.rs b/src/concurrent/ref.rs index 4b5c779..967aa2d 100644 --- a/src/concurrent/ref.rs +++ b/src/concurrent/ref.rs @@ -1,6 +1,6 @@ use crate::core::node::NodeLike; +use ::core::marker::PhantomData; use parking_lot::{ArcRwLockReadGuard, RawRwLock}; -use std::marker::PhantomData; /// A point reference that keeps its node read-locked. /// diff --git a/src/concurrent/set.rs b/src/concurrent/set.rs index 40d5065..e915082 100644 --- a/src/concurrent/set.rs +++ b/src/concurrent/set.rs @@ -1,13 +1,17 @@ +use ::core::borrow::Borrow; +use ::core::fmt::Debug; +use ::core::iter::FusedIterator; +use ::core::marker::PhantomData; +use ::core::ops::{Bound, RangeBounds}; +use ::core::sync::atomic::{AtomicPtr, AtomicU64, Ordering}; +use alloc::boxed::Box; +use alloc::collections::BTreeMap; +use alloc::sync::Arc; +use alloc::vec; +use alloc::vec::Vec; use parking_lot::{ ArcRwLockReadGuard, ArcRwLockWriteGuard, Mutex, MutexGuard, RawRwLock, RwLock, RwLockReadGuard, RwLockWriteGuard, }; -use std::collections::{BTreeMap, HashMap}; -use std::fmt::Debug; -use std::iter::FusedIterator; -use std::marker::PhantomData; -use std::ops::{Bound, RangeBounds}; -use std::sync::atomic::{AtomicPtr, AtomicU64, Ordering}; -use std::{borrow::Borrow, sync::Arc}; use crate::cdc::change::ChangeEvent; use crate::concurrent::operation::*; @@ -16,6 +20,19 @@ use crate::core::node::*; use super::r#ref::Ref; +/// Give the scheduler the core, when there is a scheduler to give it to. +/// +/// `yield_now` is a `std` call, and a `no_std` build has no thread to yield. +/// Spinning is the honest fallback there: it is what the caller was already +/// doing on the fast path, without the syscall that would make it wait longer. +#[inline] +fn yield_now() { + #[cfg(feature = "std")] + std::thread::yield_now(); + #[cfg(not(feature = "std"))] + ::core::hint::spin_loop(); +} + const ROOT_PUBLICATION_SPIN_LIMIT: usize = 16; const STABLE_READ_BLOCKING_FALLBACK_AFTER: usize = 2; const PUBLICATION_BACKLOG_DRAIN_THRESHOLD: usize = 64; @@ -147,7 +164,7 @@ where } let chunk = Arc::make_mut(&mut self.chunks[chunk_index]); match chunk.entries.binary_search_by(|(candidate, _)| candidate.cmp(&key)) { - Ok(index) => Some(std::mem::replace(&mut chunk.entries[index].1, node)), + Ok(index) => Some(::core::mem::replace(&mut chunk.entries[index].1, node)), Err(index) => { chunk.entries.insert(index, (key, node)); self.len += 1; @@ -306,7 +323,7 @@ pub(crate) struct Topology { // Writer-only reverse lookup from node identity to its current published // route key. This differs from the canonical key only for the last node, // whose stale route remains a valid final point-read fallback. - published_keys: Mutex>, + published_keys: Mutex>, published: PublishedIndex, // Even values are stable publications; odd values mean a writer may have // changed node contents or routing but has not published the new route. @@ -314,7 +331,7 @@ pub(crate) struct Topology { } impl Debug for Topology { - fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + fn fmt(&self, formatter: &mut ::core::fmt::Formatter<'_>) -> ::core::fmt::Result { formatter .debug_struct("Topology") .field("nodes", &self.index.read().len()) @@ -327,7 +344,7 @@ impl Topology { fn new() -> Self { Self { index: RwLock::new(BTreeMap::new()), - published_keys: Mutex::new(HashMap::new()), + published_keys: Mutex::new(BTreeMap::new()), published: PublishedIndex::new(), generation: AtomicU64::new(0), } @@ -403,12 +420,12 @@ where // reclamation domain. Range readers should not wait for a registry scan. index: Option>>, published: Option>, - published_keys: Option>>, + published_keys: Option>>, publish: bool, dirty: bool, } -impl std::ops::Deref for TopologyWriteGuard<'_, T, Node> +impl ::core::ops::Deref for TopologyWriteGuard<'_, T, Node> where T: Ord + Clone + Send + 'static, Node: Send + 'static, @@ -777,12 +794,12 @@ where let mut last_equal = None; for entry @ (key, _) in index { match >::borrow(key).cmp(end) { - std::cmp::Ordering::Less => {} - std::cmp::Ordering::Equal => last_equal = Some(entry), + ::core::cmp::Ordering::Less => {} + ::core::cmp::Ordering::Equal => last_equal = Some(entry), // The first node whose maximum is above the end may still start // with values inside the range. `Range::new` ranks within that // node to obtain the first out-of-range sentinel. - std::cmp::Ordering::Greater => return Some(entry), + ::core::cmp::Ordering::Greater => return Some(entry), } } last_equal.or_else(|| index.last_key_value()) @@ -803,8 +820,8 @@ where /// Iterators returned by [`crate::BTreeSet::iter`] produce their items in order, and take worst-case /// logarithmic and amortized constant time per item returned. /// -/// [`Cell`]: crate::core::cell::Cell -/// [`RefCell`]: crate::core::cell::RefCell +/// [`Cell`]: cratestd::cell::Cell +/// [`RefCell`]: cratestd::cell::RefCell /// /// # Examples /// @@ -909,7 +926,7 @@ where self } pub fn attach_node(&self, node: Node) { - self.attach_nodes(std::iter::once(node)); + self.attach_nodes(::core::iter::once(node)); } /// Attaches a persisted topology in one structural publication. @@ -975,7 +992,7 @@ where break self.index.write(); } spins += 1; - std::hint::spin_loop(); + ::core::hint::spin_loop(); }; // Another first writer may have published while this // caller was acquiring the exclusive structural guard. @@ -1393,9 +1410,9 @@ where if !generation.is_multiple_of(2) { if writer_spins < ROOT_PUBLICATION_SPIN_LIMIT { writer_spins += 1; - std::hint::spin_loop(); + ::core::hint::spin_loop(); } else { - std::thread::yield_now(); + yield_now(); } continue; } @@ -1480,9 +1497,9 @@ where if !generation.is_multiple_of(2) { if writer_spins < ROOT_PUBLICATION_SPIN_LIMIT { writer_spins += 1; - std::hint::spin_loop(); + ::core::hint::spin_loop(); } else { - std::thread::yield_now(); + yield_now(); } continue; } @@ -1669,8 +1686,8 @@ where Node: NodeLike + Send + 'static, { tree: &'a BTreeSet, - current_front_batch: Option>, - current_back_batch: Option>, + current_front_batch: Option>, + current_back_batch: Option>, // Identity of the node the last batch in each direction was cloned from, // so the next install can step past it when the cursor lookup lands on // it again (its entry key can sit past every element it still holds). @@ -2361,7 +2378,7 @@ where .remove::(&key) .expect("middle key was collected under the write lock"); let mut removed_node = node.write_arc(); - detached_nodes.push(std::mem::take(&mut *removed_node)); + detached_nodes.push(::core::mem::take(&mut *removed_node)); } // Trim the front node from the start position: its maximum goes away, diff --git a/src/core/multipair.rs b/src/core/multipair.rs index 09fb533..0524167 100644 --- a/src/core/multipair.rs +++ b/src/core/multipair.rs @@ -1,7 +1,8 @@ use crate::cdc::change::ChangeEvent; use crate::concurrent::set::BTreeSet; use crate::core::node::NodeLike; -use std::fmt::Debug; +use ::core::fmt::Debug; +use alloc::vec::Vec; pub mod ord; pub use ord::OrdMultiPair; diff --git a/src/core/multipair/ord.rs b/src/core/multipair/ord.rs index 1e03d19..38c36f9 100644 --- a/src/core/multipair/ord.rs +++ b/src/core/multipair/ord.rs @@ -1,5 +1,6 @@ -use std::borrow::Borrow; -use std::fmt::Debug; +use ::core::borrow::Borrow; +use ::core::fmt::Debug; +use alloc::vec::Vec; use core::cmp::Ordering; #[cfg(feature = "serde")] diff --git a/src/core/node.rs b/src/core/node.rs index 5fcd5e9..16db031 100644 --- a/src/core/node.rs +++ b/src/core/node.rs @@ -1,5 +1,6 @@ +use ::core::ops::Deref; +use alloc::vec::Vec; use core::borrow::Borrow; -use std::ops::Deref; pub trait NodeLike { #[allow(dead_code)] @@ -31,7 +32,7 @@ pub trait NodeLike { where T: Borrow; #[allow(dead_code)] - fn rank(&self, bound: std::ops::Bound<&Q>, from_start: bool) -> Option + fn rank(&self, bound: ::core::ops::Bound<&Q>, from_start: bool) -> Option where T: Borrow; #[allow(dead_code)] @@ -52,7 +53,7 @@ pub trait NodeLike { #[allow(dead_code)] fn min(&self) -> Option<&T>; #[allow(dead_code)] - fn iter<'a>(&'a self) -> std::slice::Iter<'a, T> + fn iter<'a>(&'a self) -> ::core::slice::Iter<'a, T> where T: 'a; } @@ -192,26 +193,30 @@ pub(crate) fn search_by(haystack: &[T], mut compare: impl FnMut(&T) -> core:: } #[inline] -fn compute_positions_to_skip(haystack: &[T], bound: std::ops::Bound<&Q>, forward: bool) -> Option +fn compute_positions_to_skip(haystack: &[T], bound: ::core::ops::Bound<&Q>, forward: bool) -> Option where T: Borrow + Ord, Q: Ord + ?Sized, { let skipped = match (bound, forward) { // A forward iterator skips values before the start bound. - (std::ops::Bound::Included(value), true) => haystack.partition_point(|item| item.borrow().cmp(value).is_lt()), - (std::ops::Bound::Excluded(value), true) => haystack.partition_point(|item| item.borrow().cmp(value).is_le()), + (::core::ops::Bound::Included(value), true) => { + haystack.partition_point(|item| item.borrow().cmp(value).is_lt()) + } + (::core::ops::Bound::Excluded(value), true) => { + haystack.partition_point(|item| item.borrow().cmp(value).is_le()) + } // A backward iterator skips values after the end bound. - (std::ops::Bound::Included(value), false) => { + (::core::ops::Bound::Included(value), false) => { let first_greater = haystack.partition_point(|item| item.borrow().cmp(value).is_le()); haystack.len() - first_greater } - (std::ops::Bound::Excluded(value), false) => { + (::core::ops::Bound::Excluded(value), false) => { let first_equal = haystack.partition_point(|item| item.borrow().cmp(value).is_lt()); haystack.len() - first_equal } - (std::ops::Bound::Unbounded, _) => return None, + (::core::ops::Bound::Unbounded, _) => return None, }; // Callers use this as the index of the last value to skip. No skipped @@ -271,7 +276,7 @@ impl NodeLike for Vec { search(self, value).ok() } #[inline] - fn rank(&self, bound: std::ops::Bound<&Q>, from_start: bool) -> Option + fn rank(&self, bound: ::core::ops::Bound<&Q>, from_start: bool) -> Option where T: Borrow + Ord, Q: Ord + ?Sized, @@ -301,7 +306,7 @@ impl NodeLike for Vec { #[inline] fn replace(&mut self, idx: usize, value: T) -> Option { if let Some(old) = self.get_mut(idx) { - let old = std::mem::replace(old, value); + let old = ::core::mem::replace(old, value); return Some(old); } @@ -316,7 +321,7 @@ impl NodeLike for Vec { self.first() } #[inline] - fn iter<'a>(&'a self) -> std::slice::Iter<'a, T> + fn iter<'a>(&'a self) -> ::core::slice::Iter<'a, T> where T: 'a, { diff --git a/src/core/pair.rs b/src/core/pair.rs index 9ec184f..a4e7025 100644 --- a/src/core/pair.rs +++ b/src/core/pair.rs @@ -1,7 +1,8 @@ +use ::core::borrow::Borrow; +use alloc::string::String; use core::cmp::Ordering; #[cfg(feature = "serde")] use serde::{Deserialize, Serialize}; -use std::borrow::Borrow; #[cfg_attr(feature = "serde", derive(Serialize, Deserialize))] #[derive(Debug, Default, Clone)] @@ -39,11 +40,11 @@ where } } -impl std::hash::Hash for Pair +impl ::core::hash::Hash for Pair where - K: std::hash::Hash, + K: ::core::hash::Hash, { - fn hash(&self, state: &mut H) { + fn hash(&self, state: &mut H) { self.key.hash(state); } } diff --git a/src/lib.rs b/src/lib.rs index ceb72c0..1a6731e 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1,3 +1,15 @@ +#![cfg_attr(not(any(test, feature = "std")), no_std)] +// The crate does not link `std` unless asked. Tests always do, because a +// concurrency test is made of threads and clocks; with the `std` feature on, +// the library links it too, and the only thing it takes from it is +// `thread::yield_now`. +#[cfg(any(test, feature = "std"))] +extern crate std; + +extern crate alloc; + +use alloc::vec; +use alloc::vec::Vec; #[cfg(feature = "concurrent")] pub mod concurrent; @@ -7,18 +19,18 @@ pub mod cdc; pub mod core; use crate::Entry::{Occupied, Vacant}; +use ::core::borrow::Borrow; +use ::core::cmp::Ordering; +use ::core::iter::FusedIterator; +use ::core::mem::swap; +use ::core::ops::Bound; +use ::core::ops::{Index, RangeBounds}; use core::constants::DEFAULT_INNER_SIZE; use core::node::*; use core::pair::Pair; use ftree::FenwickTree; #[cfg(feature = "serde")] use serde::{Deserialize, Serialize}; -use std::borrow::Borrow; -use std::cmp::Ordering; -use std::collections::Bound; -use std::iter::FusedIterator; -use std::mem::swap; -use std::ops::{Index, RangeBounds}; type Node = Vec; @@ -1333,8 +1345,8 @@ where current_front_idx: usize, current_back_node_idx: usize, current_back_idx: usize, - current_front_iterator: Option>, - current_back_iterator: Option>, + current_front_iterator: Option<::core::slice::Iter<'a, T>>, + current_back_iterator: Option<::core::slice::Iter<'a, T>>, } impl<'a, T> Iter<'a, T> @@ -2263,7 +2275,7 @@ where node_idx, position_within_node, } => { - std::mem::swap(&mut self.set.inner[node_idx][position_within_node].value, &mut value); + ::core::mem::swap(&mut self.set.inner[node_idx][position_within_node].value, &mut value); Some(value) } NodeEntry::Empty { node_idx } => { @@ -3466,13 +3478,13 @@ pub struct IterMut<'a, K: 'a, V: 'a> where K: Ord, { - inner: std::slice::IterMut<'a, Node>>, + inner: ::core::slice::IterMut<'a, Node>>, current_front_node_idx: usize, current_front_idx: usize, current_back_node_idx: usize, current_back_idx: usize, - current_front_iterator: std::slice::IterMut<'a, Pair>, - current_back_iterator: std::slice::IterMut<'a, Pair>, + current_front_iterator: ::core::slice::IterMut<'a, Pair>, + current_back_iterator: ::core::slice::IterMut<'a, Pair>, } impl<'a, K, V> Iterator for IterMut<'a, K, V> diff --git a/tests/wfe.rs b/tests/wfe.rs new file mode 100644 index 0000000..3740eae --- /dev/null +++ b/tests/wfe.rs @@ -0,0 +1,306 @@ +//! What WFE is, what it costs, and the one rule for when to emit it. +//! +//! # Why this file exists +//! +//! `benches/locks.rs` prices five locks. This asserts the platform facts that +//! benchmark rests on, so a change in the hardware, the toolchain or the OS +//! surfaces here as a named failure rather than as a benchmark that quietly +//! means something different. +//! +//! Everything below was measured on aarch64-apple-darwin, 16 cores, and every +//! bound is deliberately loose: these tests exist to catch a property +//! disappearing, not to pin a number. +//! +//! # The instruction +//! +//! `WFE` puts a core in a low-power state until its event register is set. +//! `SEV` sets it on every core. ARM documents this as *the* intended spinlock +//! construction: a waiter executes WFE to request a low-power state, and the +//! releaser executes SEV to wake it. +//! +//! Available on every ARMv7-and-later core, so every ARM64 chip: Apple +//! Silicon, AWS Graviton 2/3/4 (Neoverse N1/V1/V2), Ampere Altra, Raspberry Pi +//! 4/5. x86 has the same idea split by privilege: `MONITOR`/`MWAIT` is ring 0, +//! `UMONITOR`/`UMWAIT` is ring 3 from Intel Tremont (2020) onward, and AMD has +//! `MONITORX`/`MWAITX`. +//! +//! # Rust never emits it +//! +//! `core::hint::spin_loop()` on aarch64 is `__isb(SY)`, an instruction +//! barrier. Disassembled: +//! +//! ```text +//! __RNvCs8LfLpYhzmc_7hintasm1s: +//! isb +//! ret +//! ``` +//! +//! There is no WFE anywhere in `core::hint`, and no wrapper in +//! `core::arch::aarch64`. That is structural rather than an oversight: WFE is +//! only useful if the *releaser* pairs it with SEV, and that is a protocol +//! between both sides of a lock. A one-sided hint cannot express one, so it has +//! to be written in the lock, where both sides are known. +//! +//! # What it costs here +//! +//! ```text +//! nop 0.3 ns +//! isb (spin_loop) 8.6 ns +//! wfe, event pending 1336.7 ns +//! ``` +//! +//! So WFE is not a userspace no-op on this machine, and it does not block +//! forever either: it idles for roughly 1.3 us and returns unprompted. A +//! waiter therefore needs no SEV to make progress, and burns about 150x less +//! CPU per unit of wall time waited than an `isb` spin. +//! +//! # The rule, which is the useful part +//! +//! **WFE idles the core. It does not yield to the operating system.** +//! +//! That is why it lost here. With 128 threads on 16 cores a WFE-waiting lock +//! cost 5584 ms of CPU against 1241 ms for a yielding spinner: a waiter idling +//! in WFE is still a *scheduled* thread, so the lock holder still cannot get a +//! core. Below the core count it is level with plain spinning and no better. +//! +//! An earlier version of that lock restarted its spin schedule after every +//! WFE, so it yielded between idles. That accident is the only reason it ever +//! looked competitive, and finding it is why this rule is written down. +//! +//! So emit WFE when: +//! +//! * threads are at most cores, so idling one costs nothing that is wanted; +//! * there is no operating system to yield to, which is exactly when every +//! other option reduces to burning the core; +//! * waits are long enough that 1.3 us of idle beats 8.6 ns of `isb` repeated +//! until the holder finishes. +//! +//! Bare metal and pinned threads. Not an oversubscribed host. +//! +//! # The result above is macOS-specific, and probably inverts on AWS +//! +//! Under KVM, WFE is trapped by the hypervisor (the `TWE` bit) and KVM yields +//! the vCPU. The Linux commit is literally "arm64: KVM: Yield CPU when vcpu +//! executes a WFE", written for the same pathology measured here, where +//! spinning vCPUs hold the cores a lock holder needs and hackbench slows by +//! 40x. That trap supplies the missing half: on Graviton, WFE *does* reach a +//! scheduler. +//! +//! Graviton also maps one vCPU to one physical core with no SMT, so a guest is +//! not oversubscribed the way this machine was at 128 threads on 16 cores, +//! which is the regime WFE is for in the first place. +//! +//! **So the negative result here is not portable and must be re-measured on +//! Graviton before anyone concludes WFE is worthless.** +//! +//! # Two findings from the same work, kept so they are not re-derived +//! +//! **`lock_api` is free.** `parking_lot::Mutex` and +//! `lock_api::Mutex` measure the same, so making +//! `concurrent` generic over `R: RawMutex` costs nothing and lets a consumer +//! choose. `lock_arc` and `ArcMutexGuard` are `lock_api`'s, not +//! `parking_lot`'s, so `Ref` keeps working either way. +//! +//! **The node lock is not this crate's bottleneck.** With the index `RwLock` +//! held constant, every node lock lands inside the null. An earlier benchmark +//! varied both at once and reported the index lock's behaviour as a node-lock +//! result. The index `RwLock` is what deserves the next look. +//! +//! # Why a compiler should care +//! +//! EKOPathRS is a compiler. A compiler that recognises a spin loop can emit +//! WFE for it, which LLVM does not do. The measurement above prices that at +//! 150x less CPU per unit waited, and the rule above says when it would be +//! wrong, which is the half that makes it safe to automate. +//! +//! # Related work, recent +//! +//! * HTLL, "Latency-Aware Scalable Blocking Mutex", IEEE TPDS, January 2025: +//! throughput and latency together under oversubscription, up to 97% latency +//! reduction for about 5% throughput. +//! * Fissile Locks, arXiv 2003.05025: compact, NUMA-aware, preemption tolerant. +//! * Asymmetry-aware Scalable Locking, arXiv 2108.03355. Directly relevant +//! here, because this is a P-core/E-core machine and neither the benchmark +//! nor these tests separate them. + +#![cfg(target_arch = "aarch64")] + +use std::hint::black_box; +use std::time::Instant; + +/// Iterations per timing loop. Large enough that a nanosecond-scale +/// instruction is measurable over timer noise. +const ITERS: u64 = 200_000; + +fn ns_each(mut body: impl FnMut()) -> f64 { + // Warm, so the first run's page faults and frequency ramp are not counted. + for _ in 0..ITERS / 10 { + body(); + } + let mut runs = Vec::new(); + for _ in 0..5 { + let now = Instant::now(); + for _ in 0..ITERS { + body(); + } + runs.push(now.elapsed()); + } + runs.sort(); + runs[2].as_nanos() as f64 / ITERS as f64 +} + +fn nop_ns() -> f64 { + ns_each(|| { + // SAFETY: `nop` has no operands and no effects. + unsafe { core::arch::asm!("nop", options(nomem, nostack)) }; + }) +} + +fn isb_ns() -> f64 { + ns_each(|| black_box(core::hint::spin_loop())) +} + +fn wfe_ns() -> f64 { + ns_each(|| { + // SAFETY: WFE is unprivileged, has no memory operands, and waits at + // most for an implementation-defined period before returning. + unsafe { core::arch::asm!("wfe", options(nomem, nostack)) }; + }) +} + +/// WFE must be reachable at all from userspace, or none of this applies. +/// +/// If this ever traps or is emulated away, every other claim in this file is +/// about a different machine. +#[test] +fn wfe_and_sev_execute_in_userspace() { + // SAFETY: both are unprivileged and have no operands. + unsafe { + core::arch::asm!("sev", options(nomem, nostack)); + core::arch::asm!("wfe", options(nomem, nostack)); + } +} + +/// WFE is not a no-op here, which is the whole reason it is worth emitting. +/// +/// An implementation is free to make WFE a `nop`, and on such a machine a +/// WFE-based lock is a plain spinlock wearing a costume. This separates the +/// two: `nop` measured 0.3 ns and WFE measured 1336.7 ns, so anything within +/// an order of magnitude of `nop` means WFE is not idling. +#[test] +#[ignore = "timing, run with --ignored"] +fn wfe_actually_idles_rather_than_being_a_nop() { + let nop = nop_ns(); + let wfe = wfe_ns(); + assert!( + wfe > nop * 20.0, + "WFE at {wfe:.1} ns against nop at {nop:.1} ns: WFE is not idling on \ + this machine, so a WFE lock here is a spinlock with extra steps" + ); +} + +/// `core::hint::spin_loop()` is not WFE, and the gap is why the lock has to +/// write the instruction itself. +/// +/// `spin_loop()` is `__isb(SY)`, measured at 8.6 ns against WFE's 1336.7. If +/// these ever converge, either Rust started emitting WFE, in which case a lock +/// should stop hand-writing it, or WFE stopped idling. +#[test] +#[ignore = "timing, run with --ignored"] +fn the_standard_spin_hint_is_not_wfe() { + let isb = isb_ns(); + let wfe = wfe_ns(); + assert!( + wfe > isb * 10.0, + "spin_loop() at {isb:.1} ns and WFE at {wfe:.1} ns are within 10x. \ + core::hint::spin_loop() is __isb(SY) and should be far cheaper; if it \ + is not, check whether Rust now emits WFE and delete the hand-written one" + ); +} + +/// WFE returns on its own, so a waiter cannot deadlock if SEV is missed. +/// +/// This is what makes a WFE lock safe to write: the event register may already +/// be clear, or a SEV may land before the waiter reaches its WFE, and neither +/// wedges. Measured at roughly 1.3 us per WFE, so a hundred of them is +/// bounded well under a second on any machine where WFE idles at all. +#[test] +fn a_waiter_is_never_stuck_when_no_sev_arrives() { + // Drain any pending event so the WFEs below have nothing waiting for them. + // SAFETY: unprivileged, no operands. + unsafe { + core::arch::asm!("sev", options(nomem, nostack)); + core::arch::asm!("wfe", options(nomem, nostack)); + } + let now = Instant::now(); + for _ in 0..100 { + // SAFETY: as above. + unsafe { core::arch::asm!("wfe", options(nomem, nostack)) }; + } + assert!( + now.elapsed() < std::time::Duration::from_secs(5), + "100 WFEs with no SEV took {:?}: WFE is blocking indefinitely here, so \ + a lock must pair every one with a SEV rather than relying on the timeout", + now.elapsed() + ); +} + +/// The rule, as an executable statement: WFE does not yield to the scheduler. +/// +/// Two threads per core, one holding a lock for a long stretch. If WFE reached +/// the scheduler, waiters would stand aside and this would finish in about the +/// time the holders need. It does not, which is why the rule is "threads at +/// most cores". +/// +/// **Expected to be different under KVM**, where WFE traps and the hypervisor +/// yields the vCPU. On Graviton this test is the one to watch: if it starts +/// passing comfortably there, WFE became viable for oversubscribed hosts and +/// the guidance in this file needs revisiting. +#[test] +#[ignore = "heavy and oversubscribed, run with --ignored"] +fn wfe_does_not_yield_to_the_scheduler() { + use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; + use std::sync::Arc; + + let cores = std::thread::available_parallelism().map_or(8, std::num::NonZeroUsize::get); + let threads = cores * 8; + let held = Arc::new(AtomicBool::new(false)); + let done = Arc::new(AtomicU64::new(0)); + + let now = Instant::now(); + std::thread::scope(|scope| { + for _ in 0..threads { + let held = held.clone(); + let done = done.clone(); + scope.spawn(move || { + for _ in 0..8 { + while held + .compare_exchange_weak(false, true, Ordering::Acquire, Ordering::Relaxed) + .is_err() + { + // SAFETY: unprivileged, no operands. + unsafe { core::arch::asm!("wfe", options(nomem, nostack)) }; + } + let mut acc = 0u64; + for n in 0..20_000u64 { + acc = acc.wrapping_mul(0x9e37_79b9).wrapping_add(n); + } + black_box(acc); + held.store(false, Ordering::Release); + // SAFETY: as above. + unsafe { core::arch::asm!("sev", options(nomem, nostack)) }; + done.fetch_add(1, Ordering::Relaxed); + } + }); + } + }); + + assert_eq!(done.load(Ordering::Relaxed), threads as u64 * 8); + // Generous: this exists to prove the run terminates and to print the cost, + // not to pin a number that varies by machine. + assert!( + now.elapsed() < std::time::Duration::from_secs(120), + "{threads} threads on {cores} cores took {:?}", + now.elapsed() + ); +}