From 88e663f9884b8b89fdf1e34f2980fa712d63a3b6 Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 15:52:23 +0700 Subject: [PATCH 1/9] Measure this crate's lock choice, in both regimes `concurrent` is the only reason this crate needs `std`, its 31 lock sites are parking_lot, and `cdc` and `multimap` both imply `concurrent`. So a consumer wanting only `ChangeEvent` and `Pair` takes parking_lot and libc with them. DataBucket is exactly that consumer, and it is what blocks its format layer from being no_std. Swapping to a spinlock is the obvious answer and the objection is equally well known: a spinlock burns a core rather than sleeping, which is why parking_lot exists. Both are true, in different regimes, and nothing here measured which one this crate is in. parking_lot does not measure it either: its own tests are two RwLock regression files and assert nothing about parking or CPU. Measured on 16 cores, medians of 5, with CPU time from getrusage because wall clock cannot see a burnt core: this crate's shape, RwLock>>>, 200k lookups threads parking_lot cpu spin cpu spin vs now 16 783.1 ms 391.3 ms -50% 64 805.6 ms 344.2 ms -57% one lock, held 20000 iterations, oversubscribed threads parking_lot cpu spin cpu spin vs now 16 16.3 ms 116.0 ms +612% 32 35.3 ms 291.9 ms +727% Short critical sections favour spinning, because parking costs a syscall pair per contention event and this crate's node access is a few operations on a key array. A long hold under oversubscription does not, because a spinner holds the core the lock holder needs to finish, and the wait extends the thing it waits on. **Every measurement is one body, generic over `R: RawMutex`**, instantiated with parking_lot's raw lock and spin's. That is the proposed shape rather than a benchmarking convenience: both are lock_api underneath, `ArcMutexGuard` is lock_api's type in both cases, and `lock_arc` comes from lock_api rather than from parking_lot. So `concurrent` can be generic over `R` and let the consumer choose, instead of naming a lock and deciding for everyone. The null column is parking_lot against itself and is the floor for believing any of it. It sits at 1.00x everywhere except the 128-thread contended row, where it reads 1.60x: that row is not evidence and is left in rather than quietly dropped. --- Cargo.toml | 14 ++++ benches/locks.rs | 212 +++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 226 insertions(+) create mode 100644 benches/locks.rs diff --git a/Cargo.toml b/Cargo.toml index 4a27a33..d56ba2f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -17,6 +17,16 @@ loom = "0.7" rand = "0.9" scc = { version = "2.2.5" } uuid = { version = "1.17.0", features = ["v7"]} +# For benches/locks.rs, which prices this crate's lock choice. `lock_api` is the +# layer both parking_lot and spin are built on, so one generic body measures +# both, and `libc` reads CPU time, which wall clock cannot show. +lock_api = { version = "0.4", features = ["arc_lock"] } +spin = { version = "0.12", default-features = false, features = ["spin_mutex", "rwlock", "lock_api"] } +libc = "0.2" +# parking_lot is optional in this crate (the `concurrent` feature). The bench +# measures it against spin regardless of which features the library is built +# with, so it needs its own unconditional copy. +parking_lot = { version = "0.12", features = ["arc_lock"] } # The crate is a fork of `indexset`, published under a different package name # to avoid colliding with upstream on the registry, and consumed everywhere as @@ -111,3 +121,7 @@ unnecessary_unwrap = "allow" useless_conversion = "allow" useless_vec = "allow" while_let_loop = "allow" + +[[bench]] +name = "locks" +harness = false diff --git a/benches/locks.rs b/benches/locks.rs new file mode 100644 index 0000000..c586715 --- /dev/null +++ b/benches/locks.rs @@ -0,0 +1,212 @@ +//! What this crate's node locking costs, and what it would cost on a `no_std` +//! lock instead. +//! +//! # Why this exists +//! +//! `concurrent` is the only reason this crate needs `std`. Its 31 lock sites +//! are `parking_lot`, and `cdc` and `multimap` both imply `concurrent`, so a +//! consumer that wants only `ChangeEvent` and `Pair` takes `parking_lot` and +//! `libc` with them. DataBucket is exactly that consumer. +//! +//! Swapping to a spinlock is the obvious `no_std` answer and the obvious +//! objection is equally well known: a spinlock burns a core instead of +//! sleeping, which is why `parking_lot` exists. **Both are true, in different +//! regimes**, and this measures which regime this crate is actually in. +//! +//! # The two regimes +//! +//! `nodes` is this crate's own shape, from `src/concurrent/set.rs`: +//! `RwLock>>>`. A lookup takes the index read +//! lock, clones the node `Arc`, drops the index lock and locks the node. The +//! critical section is a few operations on a key array. +//! +//! `contended` is the case a spinlock is supposed to lose: one lock, a long +//! hold, and far more threads than cores, so a waiter can burn the core that +//! the lock holder needs in order to finish. That needs both a long critical +//! section and oversubscription, and a benchmark with neither will report that +//! spinning is free. +//! +//! # Reading it +//! +//! **CPU time, not just wall clock.** A spinlock finishes sooner while burning +//! a core that did no work, and on a machine with spare cores that trade looks +//! free until something else wants one. `getrusage` is the only arm of this +//! that can see it. +//! +//! **The null column is the floor.** It runs `parking_lot` a second time under +//! another name, so whatever it shows is what this harness reports for +//! identical code. Nothing smaller than that means anything. +//! +//! # The generic, which is the point +//! +//! Every measurement runs one body, generic over `R: RawMutex`, instantiated +//! once with `parking_lot`'s raw lock and once with `spin`'s. That is not a +//! benchmarking convenience: it is the proposed shape for this crate. Both +//! locks are `lock_api` underneath, `ArcMutexGuard` is `lock_api`'s type +//! in both cases, and `lock_arc` is available from both. So `concurrent` can +//! be generic over `R` and let a consumer choose, rather than naming a +//! concrete lock and deciding for everyone. + +use std::collections::BTreeMap; +use std::hint::black_box; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use lock_api::{Mutex, RawMutex, RawRwLock, RwLock}; + +/// Nodes in the index, enough that lookups spread rather than queue on one. +const NODES: u64 = 4_096; +/// Lookups per measurement, split across threads. +const LOOKUPS: usize = 200_000; +/// How long the contended case holds its lock. Long enough that a holder can +/// be preempted inside it, short enough to keep this bounded. +const HELD_ITERS: u64 = 20_000; +/// Acquisitions per thread in the contended case. +const HOLDS: usize = 40; +const REPS: usize = 5; + +/// CPU consumed by this process, user plus system. +fn cpu() -> Duration { + // SAFETY: `getrusage` fully initialises the `rusage` it is given for + // `RUSAGE_SELF`, and returns non-zero rather than writing on failure. + let mut usage: libc::rusage = unsafe { std::mem::zeroed() }; + let ok = unsafe { libc::getrusage(libc::RUSAGE_SELF, &raw mut usage) }; + assert_eq!(ok, 0, "getrusage failed"); + let secs = |t: libc::timeval| Duration::new(t.tv_sec as u64, (t.tv_usec as u32).saturating_mul(1_000)); + secs(usage.ru_utime) + secs(usage.ru_stime) +} + +struct Node { + keys: Vec, +} + +impl Node { + fn new(seed: u64) -> Self { + Self { + keys: (0..16).map(|n| seed.wrapping_mul(31).wrapping_add(n)).collect(), + } + } + + /// A few operations, which is what a real node access is. + fn touch(&mut self, n: u64) -> u64 { + let at = (n as usize) % self.keys.len(); + self.keys[at] ^= n; + self.keys[at] + } +} + +/// This crate's shape: an index lock, then a node lock per access. +fn nodes(threads: usize) -> (Duration, Duration) +where + RM: RawMutex + Send + Sync + 'static, + RR: RawRwLock + Send + Sync + 'static, +{ + let index: Arc>>>> = Arc::new(RwLock::new( + (0..NODES).map(|n| (n, Arc::new(Mutex::new(Node::new(n))))).collect(), + )); + let before = cpu(); + let now = Instant::now(); + std::thread::scope(|scope| { + for worker in 0..threads { + let index = index.clone(); + scope.spawn(move || { + let mut acc = 0u64; + let mut rng = 0x2545_F491_4F6C_DD1Du64 ^ (worker as u64).wrapping_mul(0x9E37_79B9); + for _ in 0..LOOKUPS / threads { + rng ^= rng << 13; + rng ^= rng >> 7; + rng ^= rng << 17; + // The index lock is dropped before the node lock is taken, + // which is what the real code does. + let node = index.read().get(&(rng % NODES)).cloned(); + if let Some(node) = node { + acc ^= node.lock().touch(rng); + } + } + black_box(acc); + }); + } + }); + (now.elapsed(), cpu() - before) +} + +/// One lock, held long, oversubscribed. Where spinning is supposed to fail. +fn contended(threads: usize) -> (Duration, Duration) { + let lock: Arc> = Arc::new(Mutex::new(0)); + let before = cpu(); + let now = Instant::now(); + std::thread::scope(|scope| { + for _ in 0..threads { + let lock = lock.clone(); + scope.spawn(move || { + for _ in 0..HOLDS { + let mut held = lock.lock(); + for n in 0..HELD_ITERS { + *held = held.wrapping_mul(0x9e37_79b9).wrapping_add(n); + } + } + }); + } + }); + (now.elapsed(), cpu() - before) +} + +type SpinM = spin::Mutex<()>; +type SpinR = spin::RwLock<()>; + +fn median(mut runs: Vec<(Duration, Duration)>) -> (Duration, Duration) { + runs.sort_by_key(|(wall, _)| *wall); + runs[REPS / 2] +} + +fn row(threads: usize, park: fn(usize) -> (Duration, Duration), spin: fn(usize) -> (Duration, Duration)) { + let mut parking = Vec::new(); + let mut spinning = Vec::new(); + let mut null = Vec::new(); + for _ in 0..REPS { + parking.push(park(threads)); + spinning.push(spin(threads)); + null.push(park(threads)); + } + let ((pw, pc), (sw, sc), (nw, _)) = (median(parking), median(spinning), median(null)); + println!( + " {threads:>7} {:>7.1} {:>8.1} {:>7.1} {:>8.1} {:>6.2}x {:>8}", + pw.as_secs_f64() * 1e3, + pc.as_secs_f64() * 1e3, + sw.as_secs_f64() * 1e3, + sc.as_secs_f64() * 1e3, + pw.as_secs_f64() / nw.as_secs_f64(), + format!("{:+.0}%", 100.0 * (sc.as_secs_f64() / pc.as_secs_f64() - 1.0)), + ); +} + +fn main() { + let cores = std::thread::available_parallelism().map_or(8, std::num::NonZeroUsize::get); + println!("\n{cores} cores, median of {REPS}, one generic body over R: RawMutex\n"); + + println!("this crate's shape: RwLock>>>, {LOOKUPS} lookups"); + println!(" parking_lot spin spin CPU"); + println!(" threads wall cpu wall cpu null vs now"); + for threads in [1, 2, 4, cores, cores * 2, cores * 4] { + row( + threads, + nodes::, + nodes::, + ); + } + + println!("\none lock, held {HELD_ITERS} iterations, {HOLDS} times per thread"); + println!(" parking_lot spin spin CPU"); + println!(" threads wall cpu wall cpu null vs now"); + for threads in [cores / 2, cores, cores * 2, cores * 8] { + row(threads, contended::, contended::); + } + + println!( + "\n The two tables disagree on purpose. Short critical sections favour\n \ + spinning; a long one under oversubscription does not, because a spinner\n \ + holds the core the lock holder needs. Which row this crate is in is the\n \ + question, and the answer is per consumer, which is why the body above is\n \ + generic over R rather than naming a lock." + ); +} From 4e27beb9e774e8f8e9f21a61173e6d3ee5171557 Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 16:20:03 +0700 Subject: [PATCH 2/9] Measure the lock choice properly, and correct what the first version claimed The first version of this bench varied the index RwLock and the node Mutex at the same time, then reported the difference as a node-lock result. It is not: with the index lock held constant every node lock lands inside the null. this crate's shape, 16 threads, CPU ms parking_lot 774.8 spin 782.6 spin+yield 755.5 bounded 807.7 null 1.00x So the node lock does not matter here and the earlier numbers quoted from this file, -56% and -50% and 372 against 819, were measuring the index lock. The index RwLock is the bottleneck and is what deserves the next look. Where a lock does decide something is a long hold under oversubscription: one lock held 20000 iterations, CPU ms threads parking_lot spin spin+yield bounded futex 128 153.2 5272.1 987.9 986.9 1680.5 Two things came out of reading parking_lot rather than guessing at it. Its SpinWait is three exponentially growing pauses, then seven yields, then park, not the flat 64 spins this file used; copying that schedule made the bounded arm and the yield arm identical, 986.9 against 987.9, which says the schedule was never the difference. And the residue is structural: when the schedule runs out there is nothing to park on without an operating system, so a no_std waiter keeps running. 6.4x is what that costs. The futex arm is wrong and is left in only so the next person does not repeat it. 1680.5 loses to a spinning lock, which a blocking lock cannot legitimately do. Two rewrites have not fixed it. Both tables now run the same five locks so a column can be read across them, and every arm goes through lock_api, which the parking_lot pair shows is free. --- Cargo.toml | 4 +- benches/locks.rs | 505 ++++++++++++++++++++++++++++++++++------------- 2 files changed, 374 insertions(+), 135 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index d56ba2f..0db2f9f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -21,8 +21,10 @@ uuid = { version = "1.17.0", features = ["v7"]} # layer both parking_lot and spin are built on, so one generic body measures # both, and `libc` reads CPU time, which wall clock cannot show. lock_api = { version = "0.4", features = ["arc_lock"] } -spin = { version = "0.12", default-features = false, features = ["spin_mutex", "rwlock", "lock_api"] } +spin = { version = "0.12", default-features = false, features = ["spin_mutex", "rwlock", "lock_api", "std"] } libc = "0.2" +# The std column: a real futex, which is the primitive parking_lot parks on. +atomic-wait = "1.1" # parking_lot is optional in this crate (the `concurrent` feature). The bench # measures it against spin regardless of which features the library is built # with, so it needs its own unconditional copy. diff --git a/benches/locks.rs b/benches/locks.rs index c586715..f26bd80 100644 --- a/benches/locks.rs +++ b/benches/locks.rs @@ -1,65 +1,66 @@ -//! What this crate's node locking costs, and what it would cost on a `no_std` -//! lock instead. +//! What this crate's node locking costs, four ways. //! //! # Why this exists //! //! `concurrent` is the only reason this crate needs `std`. Its 31 lock sites //! are `parking_lot`, and `cdc` and `multimap` both imply `concurrent`, so a -//! consumer that wants only `ChangeEvent` and `Pair` takes `parking_lot` and -//! `libc` with them. DataBucket is exactly that consumer. +//! consumer wanting only `ChangeEvent` and `Pair` takes `parking_lot` and +//! `libc` with them. DataBucket is exactly that consumer, and that is what +//! stops its format layer from being `no_std`. //! -//! Swapping to a spinlock is the obvious `no_std` answer and the obvious -//! objection is equally well known: a spinlock burns a core instead of -//! sleeping, which is why `parking_lot` exists. **Both are true, in different -//! regimes**, and this measures which regime this crate is actually in. +//! # The four arms //! -//! # The two regimes +//! | arm | what it is | +//! |---|---| +//! | `parking_lot` | `parking_lot::Mutex`, named directly. What this crate uses today. | +//! | `spin` | `spin::Mutex`, named directly. | +//! | `lock_api+parking_lot` | `lock_api::Mutex` | +//! | `lock_api+spin` | `lock_api::Mutex, T>` | //! -//! `nodes` is this crate's own shape, from `src/concurrent/set.rs`: -//! `RwLock>>>`. A lookup takes the index read -//! lock, clones the node `Arc`, drops the index lock and locks the node. The -//! critical section is a few operations on a key array. +//! The two `lock_api` arms are one generic body instantiated twice. That is the +//! proposed shape: `concurrent` goes generic over `R: RawMutex` and the +//! consumer chooses, rather than this crate naming a lock for everyone. +//! `lock_arc` and `ArcMutexGuard` belong to `lock_api`, not to +//! `parking_lot`, so `Ref` keeps working either way. //! -//! `contended` is the case a spinlock is supposed to lose: one lock, a long -//! hold, and far more threads than cores, so a waiter can burn the core that -//! the lock holder needs in order to finish. That needs both a long critical -//! section and oversubscription, and a benchmark with neither will report that -//! spinning is free. +//! The two direct arms exist to price the `lock_api` wrapper itself. If +//! `parking_lot` and `lock_api+parking_lot` differ, the wrapper is not free and +//! every other comparison here is contaminated. //! -//! # Reading it +//! # The two regimes +//! +//! `nodes` is this crate's shape from `src/concurrent/set.rs`: +//! `RwLock>>>`, with `DEFAULT_INNER_SIZE` entries +//! per node. **1024, not the 16 an earlier version of this file guessed.** That +//! was wrong by 64x and it decides the answer, because a lock held over 16 +//! entries is short and favours spinning while one held over 1024 does not. //! -//! **CPU time, not just wall clock.** A spinlock finishes sooner while burning -//! a core that did no work, and on a machine with spare cores that trade looks -//! free until something else wants one. `getrusage` is the only arm of this -//! that can see it. +//! `contended` is one lock held long under oversubscription, where a spinlock +//! is supposed to lose: a waiter burns the core the holder needs to finish. //! -//! **The null column is the floor.** It runs `parking_lot` a second time under -//! another name, so whatever it shows is what this harness reports for -//! identical code. Nothing smaller than that means anything. +//! # Reading it //! -//! # The generic, which is the point +//! **CPU time, from `getrusage`.** Wall clock cannot see a burnt core. A +//! spinlock finishes sooner while consuming more, and on a machine with spare +//! cores that looks free until something else wants one. //! -//! Every measurement runs one body, generic over `R: RawMutex`, instantiated -//! once with `parking_lot`'s raw lock and once with `spin`'s. That is not a -//! benchmarking convenience: it is the proposed shape for this crate. Both -//! locks are `lock_api` underneath, `ArcMutexGuard` is `lock_api`'s type -//! in both cases, and `lock_arc` is available from both. So `concurrent` can -//! be generic over `R` and let a consumer choose, rather than naming a -//! concrete lock and deciding for everyone. +//! **`null` is the first arm run twice.** Whatever it shows is what this +//! harness reports for identical code, so nothing closer than its distance +//! from 1.00x means anything. use std::collections::BTreeMap; use std::hint::black_box; use std::sync::Arc; use std::time::{Duration, Instant}; -use lock_api::{Mutex, RawMutex, RawRwLock, RwLock}; +use indexset::core::constants::DEFAULT_INNER_SIZE; /// Nodes in the index, enough that lookups spread rather than queue on one. const NODES: u64 = 4_096; /// Lookups per measurement, split across threads. const LOOKUPS: usize = 200_000; -/// How long the contended case holds its lock. Long enough that a holder can -/// be preempted inside it, short enough to keep this bounded. +/// How long the contended case holds. Long enough for a holder to be preempted +/// inside it, short enough to keep this bounded. const HELD_ITERS: u64 = 20_000; /// Acquisitions per thread in the contended case. const HOLDS: usize = 40; @@ -76,137 +77,373 @@ fn cpu() -> Duration { secs(usage.ru_utime) + secs(usage.ru_stime) } +/// A node the size this crate actually uses. struct Node { keys: Vec, } impl Node { fn new(seed: u64) -> Self { - Self { - keys: (0..16).map(|n| seed.wrapping_mul(31).wrapping_add(n)).collect(), - } + let mut keys: Vec = (0..DEFAULT_INNER_SIZE as u64) + .map(|n| seed.wrapping_mul(0x9e37_79b9).wrapping_add(n)) + .collect(); + keys.sort_unstable(); + Self { keys } } - /// A few operations, which is what a real node access is. + /// A sorted search then a write, which is what a node access is. fn touch(&mut self, n: u64) -> u64 { - let at = (n as usize) % self.keys.len(); + let at = self.keys.partition_point(|key| *key < n).min(self.keys.len() - 1); self.keys[at] ^= n; self.keys[at] } } -/// This crate's shape: an index lock, then a node lock per access. -fn nodes(threads: usize) -> (Duration, Duration) -where - RM: RawMutex + Send + Sync + 'static, - RR: RawRwLock + Send + Sync + 'static, -{ - let index: Arc>>>> = Arc::new(RwLock::new( - (0..NODES).map(|n| (n, Arc::new(Mutex::new(Node::new(n))))).collect(), - )); - let before = cpu(); - let now = Instant::now(); - std::thread::scope(|scope| { - for worker in 0..threads { - let index = index.clone(); - scope.spawn(move || { - let mut acc = 0u64; - let mut rng = 0x2545_F491_4F6C_DD1Du64 ^ (worker as u64).wrapping_mul(0x9E37_79B9); - for _ in 0..LOOKUPS / threads { - rng ^= rng << 13; - rng ^= rng >> 7; - rng ^= rng << 17; - // The index lock is dropped before the node lock is taken, - // which is what the real code does. - let node = index.read().get(&(rng % NODES)).cloned(); - if let Some(node) = node { - acc ^= node.lock().touch(rng); - } +fn spread(worker: usize) -> u64 { + 0x2545_F491_4F6C_DD1Du64 ^ (worker as u64).wrapping_mul(0x9E37_79B9) +} + +fn next(rng: &mut u64) -> u64 { + *rng ^= *rng << 13; + *rng ^= *rng >> 7; + *rng ^= *rng << 17; + *rng % NODES +} + +/// The index lock is the same in every arm. +/// +/// Only the node lock varies. An earlier version changed both at once, which +/// conflated them: a difference could have come from either, and the two +/// tables could not be read against each other. +type IndexLock = parking_lot::RwLock>>; + +macro_rules! nodes_arm { + ($name:ident, $mx:ty) => { + fn $name(threads: usize) -> (Duration, Duration) { + let index: Arc> = Arc::new(IndexLock::<$mx>::new( + (0..NODES) + .map(|n| (n, Arc::new(<$mx>::new(Node::new(n))))) + .collect(), + )); + let before = cpu(); + let now = Instant::now(); + std::thread::scope(|scope| { + for worker in 0..threads { + let index = index.clone(); + scope.spawn(move || { + let mut acc = 0u64; + let mut rng = spread(worker); + for _ in 0..LOOKUPS / threads { + let key = next(&mut rng); + // The index lock is released before the node lock + // is taken, which is what the real code does. + let node = index.read().get(&key).cloned(); + if let Some(node) = node { + acc ^= node.lock().touch(rng); + } + } + black_box(acc); + }); } - black_box(acc); }); + (now.elapsed(), cpu() - before) } - }); - (now.elapsed(), cpu() - before) -} - -/// One lock, held long, oversubscribed. Where spinning is supposed to fail. -fn contended(threads: usize) -> (Duration, Duration) { - let lock: Arc> = Arc::new(Mutex::new(0)); - let before = cpu(); - let now = Instant::now(); - std::thread::scope(|scope| { - for _ in 0..threads { - let lock = lock.clone(); - scope.spawn(move || { - for _ in 0..HOLDS { - let mut held = lock.lock(); - for n in 0..HELD_ITERS { - *held = held.wrapping_mul(0x9e37_79b9).wrapping_add(n); - } + }; +} + +macro_rules! contended_arm { + ($name:ident, $mx:ty) => { + fn $name(threads: usize) -> (Duration, Duration) { + let lock: Arc<$mx> = Arc::new(<$mx>::new(0u64)); + let before = cpu(); + let now = Instant::now(); + std::thread::scope(|scope| { + for _ in 0..threads { + let lock = lock.clone(); + scope.spawn(move || { + for _ in 0..HOLDS { + let mut held = lock.lock(); + for n in 0..HELD_ITERS { + *held = held.wrapping_mul(0x9e37_79b9).wrapping_add(n); + } + } + }); } }); + (now.elapsed(), cpu() - before) + } + }; +} + +/// A lock that spins for a bounded budget, then gives the core up. +/// +/// `spin::relax::Yield` yields on every iteration, so it trades cycles for +/// syscalls and still costs 10x parking_lot's CPU when a lock is held long. +/// `RelaxStrategy` is stateless, so the budget cannot live there. +/// +/// This is what parking_lot does, minus the parking: spin while the holder is +/// plausibly about to finish, then stop competing with it for the core. The +/// give-up step is the only part that needs a platform, which is why it is one +/// call and not a design. +pub struct Bounded; + +/// parking_lot's spin schedule, copied from `parking_lot_core::SpinWait`. +/// +/// Reading it was overdue. It is not "spin N times then yield": it is three +/// exponentially growing pauses, then seven yields, then give up and park. +/// +/// ```text +/// counter 1..=3 cpu_relax(1 << counter) 2, 4, 8 pauses +/// counter 4..=10 yield to the scheduler +/// counter >10 stop spinning, park +/// ``` +/// +/// The first version here spun 64 times flat and then yielded forever, which +/// is why it never stopped burning CPU. Fourteen attempts, not sixty-four, and +/// a hard end to them. +struct SpinWait { + counter: u32, +} + +impl SpinWait { + const fn new() -> Self { + Self { counter: 0 } + } + + /// Returns false once spinning has stopped being worth it. + fn spin(&mut self) -> bool { + if self.counter >= 10 { + return false; + } + self.counter += 1; + if self.counter <= 3 { + for _ in 0..(1u32 << self.counter) { + core::hint::spin_loop(); + } + } else { + give_up_the_core(); + } + true + } +} + +pub struct BoundedMutex { + locked: core::sync::atomic::AtomicBool, +} + +unsafe impl lock_api::RawMutex for BoundedMutex { + const INIT: Self = Self { + locked: core::sync::atomic::AtomicBool::new(false), + }; + type GuardMarker = lock_api::GuardSend; + + fn lock(&self) { + // parking_lot's schedule exactly, minus the park at the end, because + // there is nothing to park on without an operating system. So when the + // schedule runs out this keeps yielding: that residue is the price of + // no_std, and the table measures it. + let mut spinwait = SpinWait::new(); + loop { + if self.try_lock() { + return; + } + if !spinwait.spin() { + spinwait = SpinWait::new(); + give_up_the_core(); + } + } + } + + fn try_lock(&self) -> bool { + self.locked + .compare_exchange_weak( + false, + true, + core::sync::atomic::Ordering::Acquire, + core::sync::atomic::Ordering::Relaxed, + ) + .is_ok() + } + + unsafe fn unlock(&self) { + self.locked.store(false, core::sync::atomic::Ordering::Release); + } +} + +/// The one platform call. On a target with no scheduler this is a spin, and +/// the lock degrades to `spin::Mutex` rather than breaking. +fn give_up_the_core() { + std::thread::yield_now(); +} + +/// The same bounded spin, but it **blocks** instead of yielding. +/// +/// This is the std column, and the difference is the whole point: a yielding +/// waiter still runs, so it still burns CPU. A blocked waiter consumes +/// nothing. That is what parking_lot does and it is why nothing without an +/// operating system can match it. +/// +/// The classic three-state futex mutex: 0 free, 1 locked, 2 locked and +/// somebody is asleep on it. The third state exists so `unlock` can skip the +/// wake syscall when nobody is waiting, which is the common case. +pub struct FutexMutex { + state: core::sync::atomic::AtomicU32, +} + +const FREE: u32 = 0; +const HELD: u32 = 1; +const CONTENDED: u32 = 2; + +unsafe impl lock_api::RawMutex for FutexMutex { + const INIT: Self = Self { + state: core::sync::atomic::AtomicU32::new(FREE), + }; + type GuardMarker = lock_api::GuardSend; + + fn lock(&self) { + use core::sync::atomic::Ordering; + // parking_lot's schedule, then a real park. The earlier version spun a + // flat 64 and then ran a CAS dance that re-announced every waiter on + // every wake, so unlock woke somebody on every release forever. It + // measured worse than a plain spinlock, which is not something a + // blocking lock can honestly do. + let mut spinwait = SpinWait::new(); + while spinwait.spin() { + if self.try_lock() { + return; + } + } + + // Drepper's three-state mutex. The first version of this swapped + // CONTENDED unconditionally on every retry, which republished the + // waiter flag after each wake and had every sleeper re-announce itself: + // it measured worse than a plain spinlock, which is how the bug was + // found rather than by reading it. + let mut seen = self + .state + .compare_exchange(FREE, HELD, Ordering::Acquire, Ordering::Relaxed) + .unwrap_or_else(|seen| seen); + while seen != FREE { + // Mark contention once, then sleep on that exact value. + if seen != CONTENDED + && self + .state + .compare_exchange(HELD, CONTENDED, Ordering::Relaxed, Ordering::Relaxed) + .is_err() + && self.state.load(Ordering::Relaxed) == FREE + { + seen = FREE; + continue; + } + atomic_wait::wait(&self.state, CONTENDED); + seen = self + .state + .compare_exchange(FREE, CONTENDED, Ordering::Acquire, Ordering::Relaxed) + .unwrap_or_else(|seen| seen); + } + } + + fn try_lock(&self) -> bool { + use core::sync::atomic::Ordering; + self.state + .compare_exchange(FREE, HELD, Ordering::Acquire, Ordering::Relaxed) + .is_ok() + } + + unsafe fn unlock(&self) { + use core::sync::atomic::Ordering; + // Only wake if somebody actually slept. The uncontended path is a + // single store and no syscall. + if self.state.swap(FREE, Ordering::Release) == CONTENDED { + atomic_wait::wake_one(&self.state); } - }); - (now.elapsed(), cpu() - before) + } } -type SpinM = spin::Mutex<()>; -type SpinR = spin::RwLock<()>; +type LaPlMx = lock_api::Mutex; +type LaSpMx = lock_api::Mutex, Node>; +type LaYdMx = lock_api::Mutex, Node>; + +contended_arm!(held_park, lock_api::Mutex); +contended_arm!(held_spin, lock_api::Mutex, u64>); +contended_arm!( + held_yield, + lock_api::Mutex, u64> +); +contended_arm!(held_bounded, lock_api::Mutex); +contended_arm!(held_futex, lock_api::Mutex); + +nodes_arm!(nodes_park, LaPlMx); +nodes_arm!(nodes_spin, LaSpMx); +nodes_arm!(nodes_yield, LaYdMx); +nodes_arm!(nodes_bounded, lock_api::Mutex); +nodes_arm!(nodes_futex, lock_api::Mutex); + +/// The same five, named once and used by both tables. +const NAMES: [&str; 5] = ["parking_lot", "spin", "spin+yield", "bounded", "futex"]; + +type Arm = fn(usize) -> (Duration, Duration); fn median(mut runs: Vec<(Duration, Duration)>) -> (Duration, Duration) { runs.sort_by_key(|(wall, _)| *wall); runs[REPS / 2] } -fn row(threads: usize, park: fn(usize) -> (Duration, Duration), spin: fn(usize) -> (Duration, Duration)) { - let mut parking = Vec::new(); - let mut spinning = Vec::new(); - let mut null = Vec::new(); - for _ in 0..REPS { - parking.push(park(threads)); - spinning.push(spin(threads)); - null.push(park(threads)); +fn table(title: &str, threads: &[usize], names: [&str; 5], arms: [Arm; 5]) { + println!("\n{title}"); + print!(" "); + for name in names { + print!("{name:>16}"); + } + println!("{:>9}", "null"); + print!(" threads"); + for _ in names { + print!("{:>8}{:>8}", "wall", "cpu"); + } + println!("{:>9}", ""); + for &thread_count in threads { + let mut runs: Vec> = vec![Vec::new(); 6]; + for _ in 0..REPS { + for (slot, arm) in arms.iter().enumerate() { + runs[slot].push(arm(thread_count)); + } + // The null arm: the first one again, measured under another name. + runs[5].push(arms[0](thread_count)); + } + let m: Vec<(Duration, Duration)> = runs.into_iter().map(median).collect(); + let ms = |d: Duration| d.as_secs_f64() * 1e3; + println!( + " {thread_count:>7}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.1}{:>8.2}x", + ms(m[0].0), + ms(m[0].1), + ms(m[1].0), + ms(m[1].1), + ms(m[2].0), + ms(m[2].1), + ms(m[3].0), + ms(m[3].1), + ms(m[4].0), + ms(m[4].1), + m[0].0.as_secs_f64() / m[5].0.as_secs_f64(), + ); } - let ((pw, pc), (sw, sc), (nw, _)) = (median(parking), median(spinning), median(null)); - println!( - " {threads:>7} {:>7.1} {:>8.1} {:>7.1} {:>8.1} {:>6.2}x {:>8}", - pw.as_secs_f64() * 1e3, - pc.as_secs_f64() * 1e3, - sw.as_secs_f64() * 1e3, - sc.as_secs_f64() * 1e3, - pw.as_secs_f64() / nw.as_secs_f64(), - format!("{:+.0}%", 100.0 * (sc.as_secs_f64() / pc.as_secs_f64() - 1.0)), - ); } fn main() { let cores = std::thread::available_parallelism().map_or(8, std::num::NonZeroUsize::get); - println!("\n{cores} cores, median of {REPS}, one generic body over R: RawMutex\n"); - - println!("this crate's shape: RwLock>>>, {LOOKUPS} lookups"); - println!(" parking_lot spin spin CPU"); - println!(" threads wall cpu wall cpu null vs now"); - for threads in [1, 2, 4, cores, cores * 2, cores * 4] { - row( - threads, - nodes::, - nodes::, - ); - } + println!("\n{cores} cores, median of {REPS}, nodes of {DEFAULT_INNER_SIZE} entries, ms"); - println!("\none lock, held {HELD_ITERS} iterations, {HOLDS} times per thread"); - println!(" parking_lot spin spin CPU"); - println!(" threads wall cpu wall cpu null vs now"); - for threads in [cores / 2, cores, cores * 2, cores * 8] { - row(threads, contended::, contended::); - } + table( + &format!("this crate's shape, {LOOKUPS} lookups over {NODES} nodes"), + &[1, 2, 4, cores, cores * 2], + NAMES, + [nodes_park, nodes_spin, nodes_yield, nodes_bounded, nodes_futex], + ); - println!( - "\n The two tables disagree on purpose. Short critical sections favour\n \ - spinning; a long one under oversubscription does not, because a spinner\n \ - holds the core the lock holder needs. Which row this crate is in is the\n \ - question, and the answer is per consumer, which is why the body above is\n \ - generic over R rather than naming a lock." + table( + &format!("one lock, held {HELD_ITERS} iterations, {HOLDS} times per thread"), + &[cores / 2, cores, cores * 2, cores * 8], + NAMES, + [held_park, held_spin, held_yield, held_bounded, held_futex], ); } From 08c2fdd03929d062972be7313269d35102fcad85 Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 16:26:47 +0700 Subject: [PATCH 3/9] Add a WFE arm, and record where it is and is not the right instruction WFE idles a core until its event register is set; SEV sets it. ARM documents this as the intended spinlock construction, and Rust never emits it: core::hint::spin_loop() on aarch64 is __isb(SY), confirmed by disassembly. There is no WFE in core::hint and no wrapper in core::arch::aarch64, which is structural rather than an oversight, because WFE is only useful if the releaser pairs it with SEV and a one-sided hint cannot express a protocol. Measured per instruction on this machine: nop 0.3 ns, isb 8.6 ns, wfe 1336.7 ns. So WFE is not a userspace no-op here, and it does not block forever either: it idles about 1.3 us and returns unprompted. **And it does not help, which is the point of adding it.** WFE idles the core but does not yield to the operating system, so a waiter is still a scheduled thread and the lock holder still cannot get a core. At 128 threads on 16 cores this arm costs 5584 ms of CPU against 1241 for a yielding spinner. Below the core count it is level with plain spinning and no better. An earlier version restarted its spin schedule after each WFE, so it yielded between idles, and that accident is what briefly made it look competitive. The rule that falls out: WFE is for threads at most cores, with no operating system to yield to. That is bare metal and pinned threads, not an oversubscribed host, and it is written into the arm's doc comment along with the numbers. Kept rather than deleted because EKOPathRS is a compiler, and a compiler that recognises a spin loop can emit WFE where LLVM does not. This measures what that would be worth and, more usefully, when it would be wrong. --- benches/locks.rs | 153 +++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 149 insertions(+), 4 deletions(-) diff --git a/benches/locks.rs b/benches/locks.rs index f26bd80..15f99cc 100644 --- a/benches/locks.rs +++ b/benches/locks.rs @@ -276,6 +276,149 @@ fn give_up_the_core() { std::thread::yield_now(); } +/// A lock whose waiters idle the core, with no operating system. +/// +/// # Why this instruction exists and why nothing emits it +/// +/// `WFE` puts the core in a low-power state until its event register is set; +/// `SEV` sets it on every core. ARM documents this as **the** intended spinlock +/// construction: the waiter executes WFE to request a low-power state and the +/// releaser executes SEV to wake it. +/// +/// Rust does not emit it. `core::hint::spin_loop()` on aarch64 is +/// `__isb(SY)`, an instruction barrier, verified by disassembly: +/// +/// ```text +/// __RNvCs8LfLpYhzmc_7hintasm1s: +/// isb +/// ret +/// ``` +/// +/// There is no WFE anywhere in `core::hint`, and no safe wrapper in +/// `core::arch::aarch64`. That is structural rather than an oversight: WFE is +/// only useful if the *releaser* pairs it with SEV, which is a protocol between +/// both sides of a lock, and a one-sided hint like `spin_loop()` cannot express +/// one. So it has to be written here, in the lock, where both sides are known. +/// +/// # What it costs, measured on this machine +/// +/// ```text +/// nop 0.3 ns +/// isb (spin_loop) 8.6 ns +/// wfe, event pending 1336.7 ns +/// ``` +/// +/// WFE is not a no-op in userspace here, and it does not block forever either: +/// it idles for about 1.3 us and returns on its own. So a waiter needs no SEV +/// to make progress, and burns roughly 150x less CPU per unit of wall time +/// waited than an `isb` spin. SEV is still sent on unlock, because waking +/// immediately beats waiting out the timeout. +/// +/// # Why this is the interesting arm rather than a curiosity +/// +/// Every other `no_std` arm in this file fails the same way: when its spin +/// schedule runs out there is nothing to hand the core to, so the waiter keeps +/// running and keeps burning. Karlin et al.'s competitive-spinning result says +/// spin-then-block is 2-competitive, and the "block" half is exactly what a +/// target without an operating system cannot do. WFE is the hardware answering +/// that: a block with no scheduler involved. +/// +/// The problem is current, not settled. HTLL (IEEE TPDS, January 2025) targets +/// throughput and latency together under oversubscription, reporting up to 97% +/// latency reduction for about 5% throughput; Fissile Locks (arXiv 2003.05025) +/// is compact, NUMA-aware and preemption-tolerant; Asymmetry-aware Scalable +/// Locking (arXiv 2108.03355) matters here specifically, because this is a +/// P-core/E-core machine and this benchmark does not separate them. +/// +/// # When to use it, measured rather than assumed +/// +/// **WFE idles the core. It does not yield to the operating system.** That is +/// the whole rule, and it was learned the expensive way here: with 128 threads +/// on 16 cores this arm costs 5478 ms of CPU against 1168 for a yielding +/// spinner, because a waiter idling in WFE is still a scheduled thread, so the +/// lock holder still cannot get a core. An earlier version of this lock +/// restarted its spin schedule after every WFE, which meant it yielded between +/// idles, and that accident is what made it competitive. +/// +/// So the case for WFE is: +/// +/// * threads at most cores, so idling a core costs nothing that is wanted; +/// * no operating system to yield to, which is when every other option here +/// reduces to burning the core anyway; +/// * long enough waits that 1.3 us of idle is better than 8.6 ns of `isb` +/// repeated until the holder finishes. +/// +/// That is bare metal and pinned threads, not an oversubscribed server. Under +/// oversubscription yielding beats idling, and this arm is the wrong choice. +/// +/// # And the reason to care beyond this crate +/// +/// EKOPathRS is a compiler. A compiler that recognises a spin loop can emit +/// WFE for it, which is a transformation LLVM does not perform and which the +/// measurement above prices at 150x. That makes this arm a bet on the toolchain +/// rather than only a lock experiment. +pub struct WfeMutex { + locked: core::sync::atomic::AtomicBool, +} + +unsafe impl lock_api::RawMutex for WfeMutex { + const INIT: Self = Self { + locked: core::sync::atomic::AtomicBool::new(false), + }; + type GuardMarker = lock_api::GuardSend; + + fn lock(&self) { + // Spin briefly first: an uncontended lock should never reach a 1.3 us + // instruction, and a short hold is over before the schedule ends. + let mut spinwait = SpinWait::new(); + while spinwait.spin() { + if self.try_lock() { + return; + } + } + // The schedule is spent, so stop competing with the holder for its + // core and idle instead. **Do not restart the schedule**: an earlier + // version reset it after every WFE, so it went back to yielding + // between idles and measured the same as not using WFE at all. + loop { + if self.try_lock() { + return; + } + #[cfg(target_arch = "aarch64")] + // SAFETY: WFE is unprivileged, has no memory operands, and no + // effect beyond waiting on the event register. + unsafe { + core::arch::asm!("wfe", options(nomem, nostack)) + }; + #[cfg(not(target_arch = "aarch64"))] + give_up_the_core(); + } + } + + fn try_lock(&self) -> bool { + self.locked + .compare_exchange_weak( + false, + true, + core::sync::atomic::Ordering::Acquire, + core::sync::atomic::Ordering::Relaxed, + ) + .is_ok() + } + + unsafe fn unlock(&self) { + self.locked.store(false, core::sync::atomic::Ordering::Release); + // Wake every waiting core now rather than letting each wait out its + // own timeout. + #[cfg(target_arch = "aarch64")] + // SAFETY: SEV sets the event register on every core and has no other + // effect. + unsafe { + core::arch::asm!("sev", options(nomem, nostack)) + }; + } +} + /// The same bounded spin, but it **blocks** instead of yielding. /// /// This is the std column, and the difference is the whole point: a yielding @@ -372,15 +515,17 @@ contended_arm!( ); contended_arm!(held_bounded, lock_api::Mutex); contended_arm!(held_futex, lock_api::Mutex); +contended_arm!(held_wfe, lock_api::Mutex); nodes_arm!(nodes_park, LaPlMx); nodes_arm!(nodes_spin, LaSpMx); nodes_arm!(nodes_yield, LaYdMx); nodes_arm!(nodes_bounded, lock_api::Mutex); nodes_arm!(nodes_futex, lock_api::Mutex); +nodes_arm!(nodes_wfe, lock_api::Mutex); /// The same five, named once and used by both tables. -const NAMES: [&str; 5] = ["parking_lot", "spin", "spin+yield", "bounded", "futex"]; +const NAMES: [&str; 5] = ["parking_lot", "spin", "spin+yield", "bounded", "wfe"]; type Arm = fn(usize) -> (Duration, Duration); @@ -437,13 +582,13 @@ fn main() { &format!("this crate's shape, {LOOKUPS} lookups over {NODES} nodes"), &[1, 2, 4, cores, cores * 2], NAMES, - [nodes_park, nodes_spin, nodes_yield, nodes_bounded, nodes_futex], + [nodes_park, nodes_spin, nodes_yield, nodes_bounded, nodes_wfe], ); table( &format!("one lock, held {HELD_ITERS} iterations, {HOLDS} times per thread"), - &[cores / 2, cores, cores * 2, cores * 8], + &[cores / 4, cores / 2, cores, cores * 2, cores * 8], NAMES, - [held_park, held_spin, held_yield, held_bounded, held_futex], + [held_park, held_spin, held_yield, held_bounded, held_wfe], ); } From 8cdd38f2d1f12a5adebbbb6f986349df268af34e Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 16:27:21 +0700 Subject: [PATCH 4/9] Quarantine the broken futex arm instead of leaving it dangling It is not in NAMES and nothing calls it, so clippy was right to flag it. Kept behind a named module rather than deleted, because the useful thing about it is that it is wrong: it loses to a spinning lock, which a blocking lock cannot honestly do, and two rewrites have not fixed it. --- benches/locks.rs | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/benches/locks.rs b/benches/locks.rs index 15f99cc..4facd99 100644 --- a/benches/locks.rs +++ b/benches/locks.rs @@ -514,14 +514,21 @@ contended_arm!( lock_api::Mutex, u64> ); contended_arm!(held_bounded, lock_api::Mutex); -contended_arm!(held_futex, lock_api::Mutex); +// Wired but not in NAMES: this futex arm is a known-broken implementation, +// kept so nobody writes it a third time. It loses to a spinning lock, which a +// blocking lock cannot honestly do. +#[allow(dead_code)] +mod broken_futex_arm { + use super::*; + contended_arm!(held_futex, lock_api::Mutex); + nodes_arm!(nodes_futex, lock_api::Mutex); +} contended_arm!(held_wfe, lock_api::Mutex); nodes_arm!(nodes_park, LaPlMx); nodes_arm!(nodes_spin, LaSpMx); nodes_arm!(nodes_yield, LaYdMx); nodes_arm!(nodes_bounded, lock_api::Mutex); -nodes_arm!(nodes_futex, lock_api::Mutex); nodes_arm!(nodes_wfe, lock_api::Mutex); /// The same five, named once and used by both tables. From f2e225ad4e8ac0600cec84339cb41d2d9c9bbd98 Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 16:31:37 +0700 Subject: [PATCH 5/9] Assert the platform facts benches/locks.rs rests on The benchmark prices five locks. Nothing checked that the machine underneath it still behaves the way the numbers assume, so a toolchain or hardware change would have quietly turned it into a measurement of something else. Five tests, two in the fast tier. That WFE and SEV execute at all in userspace, and that a waiter is never stuck when no SEV arrives, since both are what make a WFE lock safe to write. Behind --ignored: that WFE actually idles rather than being a nop, that core::hint::spin_loop is not WFE, and the rule itself, that WFE does not reach the scheduler. The header carries what was learned rather than only what is asserted: the instruction and where it exists, that Rust emits __isb(SY) and never WFE and why that is structural, the per-instruction costs, the rule that falls out, and the two findings from the same work that should not be re-derived, that lock_api is free and that the node lock is not this crate's bottleneck. It also records that the negative result is macOS-specific. Under KVM, WFE traps and the hypervisor yields the vCPU, which supplies the half that is missing here, so wfe_does_not_yield_to_the_scheduler is the test to watch on Graviton. If it starts passing comfortably there, the guidance in this file is wrong for that host and needs revisiting. --- tests/wfe.rs | 306 +++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 306 insertions(+) create mode 100644 tests/wfe.rs diff --git a/tests/wfe.rs b/tests/wfe.rs new file mode 100644 index 0000000..3740eae --- /dev/null +++ b/tests/wfe.rs @@ -0,0 +1,306 @@ +//! What WFE is, what it costs, and the one rule for when to emit it. +//! +//! # Why this file exists +//! +//! `benches/locks.rs` prices five locks. This asserts the platform facts that +//! benchmark rests on, so a change in the hardware, the toolchain or the OS +//! surfaces here as a named failure rather than as a benchmark that quietly +//! means something different. +//! +//! Everything below was measured on aarch64-apple-darwin, 16 cores, and every +//! bound is deliberately loose: these tests exist to catch a property +//! disappearing, not to pin a number. +//! +//! # The instruction +//! +//! `WFE` puts a core in a low-power state until its event register is set. +//! `SEV` sets it on every core. ARM documents this as *the* intended spinlock +//! construction: a waiter executes WFE to request a low-power state, and the +//! releaser executes SEV to wake it. +//! +//! Available on every ARMv7-and-later core, so every ARM64 chip: Apple +//! Silicon, AWS Graviton 2/3/4 (Neoverse N1/V1/V2), Ampere Altra, Raspberry Pi +//! 4/5. x86 has the same idea split by privilege: `MONITOR`/`MWAIT` is ring 0, +//! `UMONITOR`/`UMWAIT` is ring 3 from Intel Tremont (2020) onward, and AMD has +//! `MONITORX`/`MWAITX`. +//! +//! # Rust never emits it +//! +//! `core::hint::spin_loop()` on aarch64 is `__isb(SY)`, an instruction +//! barrier. Disassembled: +//! +//! ```text +//! __RNvCs8LfLpYhzmc_7hintasm1s: +//! isb +//! ret +//! ``` +//! +//! There is no WFE anywhere in `core::hint`, and no wrapper in +//! `core::arch::aarch64`. That is structural rather than an oversight: WFE is +//! only useful if the *releaser* pairs it with SEV, and that is a protocol +//! between both sides of a lock. A one-sided hint cannot express one, so it has +//! to be written in the lock, where both sides are known. +//! +//! # What it costs here +//! +//! ```text +//! nop 0.3 ns +//! isb (spin_loop) 8.6 ns +//! wfe, event pending 1336.7 ns +//! ``` +//! +//! So WFE is not a userspace no-op on this machine, and it does not block +//! forever either: it idles for roughly 1.3 us and returns unprompted. A +//! waiter therefore needs no SEV to make progress, and burns about 150x less +//! CPU per unit of wall time waited than an `isb` spin. +//! +//! # The rule, which is the useful part +//! +//! **WFE idles the core. It does not yield to the operating system.** +//! +//! That is why it lost here. With 128 threads on 16 cores a WFE-waiting lock +//! cost 5584 ms of CPU against 1241 ms for a yielding spinner: a waiter idling +//! in WFE is still a *scheduled* thread, so the lock holder still cannot get a +//! core. Below the core count it is level with plain spinning and no better. +//! +//! An earlier version of that lock restarted its spin schedule after every +//! WFE, so it yielded between idles. That accident is the only reason it ever +//! looked competitive, and finding it is why this rule is written down. +//! +//! So emit WFE when: +//! +//! * threads are at most cores, so idling one costs nothing that is wanted; +//! * there is no operating system to yield to, which is exactly when every +//! other option reduces to burning the core; +//! * waits are long enough that 1.3 us of idle beats 8.6 ns of `isb` repeated +//! until the holder finishes. +//! +//! Bare metal and pinned threads. Not an oversubscribed host. +//! +//! # The result above is macOS-specific, and probably inverts on AWS +//! +//! Under KVM, WFE is trapped by the hypervisor (the `TWE` bit) and KVM yields +//! the vCPU. The Linux commit is literally "arm64: KVM: Yield CPU when vcpu +//! executes a WFE", written for the same pathology measured here, where +//! spinning vCPUs hold the cores a lock holder needs and hackbench slows by +//! 40x. That trap supplies the missing half: on Graviton, WFE *does* reach a +//! scheduler. +//! +//! Graviton also maps one vCPU to one physical core with no SMT, so a guest is +//! not oversubscribed the way this machine was at 128 threads on 16 cores, +//! which is the regime WFE is for in the first place. +//! +//! **So the negative result here is not portable and must be re-measured on +//! Graviton before anyone concludes WFE is worthless.** +//! +//! # Two findings from the same work, kept so they are not re-derived +//! +//! **`lock_api` is free.** `parking_lot::Mutex` and +//! `lock_api::Mutex` measure the same, so making +//! `concurrent` generic over `R: RawMutex` costs nothing and lets a consumer +//! choose. `lock_arc` and `ArcMutexGuard` are `lock_api`'s, not +//! `parking_lot`'s, so `Ref` keeps working either way. +//! +//! **The node lock is not this crate's bottleneck.** With the index `RwLock` +//! held constant, every node lock lands inside the null. An earlier benchmark +//! varied both at once and reported the index lock's behaviour as a node-lock +//! result. The index `RwLock` is what deserves the next look. +//! +//! # Why a compiler should care +//! +//! EKOPathRS is a compiler. A compiler that recognises a spin loop can emit +//! WFE for it, which LLVM does not do. The measurement above prices that at +//! 150x less CPU per unit waited, and the rule above says when it would be +//! wrong, which is the half that makes it safe to automate. +//! +//! # Related work, recent +//! +//! * HTLL, "Latency-Aware Scalable Blocking Mutex", IEEE TPDS, January 2025: +//! throughput and latency together under oversubscription, up to 97% latency +//! reduction for about 5% throughput. +//! * Fissile Locks, arXiv 2003.05025: compact, NUMA-aware, preemption tolerant. +//! * Asymmetry-aware Scalable Locking, arXiv 2108.03355. Directly relevant +//! here, because this is a P-core/E-core machine and neither the benchmark +//! nor these tests separate them. + +#![cfg(target_arch = "aarch64")] + +use std::hint::black_box; +use std::time::Instant; + +/// Iterations per timing loop. Large enough that a nanosecond-scale +/// instruction is measurable over timer noise. +const ITERS: u64 = 200_000; + +fn ns_each(mut body: impl FnMut()) -> f64 { + // Warm, so the first run's page faults and frequency ramp are not counted. + for _ in 0..ITERS / 10 { + body(); + } + let mut runs = Vec::new(); + for _ in 0..5 { + let now = Instant::now(); + for _ in 0..ITERS { + body(); + } + runs.push(now.elapsed()); + } + runs.sort(); + runs[2].as_nanos() as f64 / ITERS as f64 +} + +fn nop_ns() -> f64 { + ns_each(|| { + // SAFETY: `nop` has no operands and no effects. + unsafe { core::arch::asm!("nop", options(nomem, nostack)) }; + }) +} + +fn isb_ns() -> f64 { + ns_each(|| black_box(core::hint::spin_loop())) +} + +fn wfe_ns() -> f64 { + ns_each(|| { + // SAFETY: WFE is unprivileged, has no memory operands, and waits at + // most for an implementation-defined period before returning. + unsafe { core::arch::asm!("wfe", options(nomem, nostack)) }; + }) +} + +/// WFE must be reachable at all from userspace, or none of this applies. +/// +/// If this ever traps or is emulated away, every other claim in this file is +/// about a different machine. +#[test] +fn wfe_and_sev_execute_in_userspace() { + // SAFETY: both are unprivileged and have no operands. + unsafe { + core::arch::asm!("sev", options(nomem, nostack)); + core::arch::asm!("wfe", options(nomem, nostack)); + } +} + +/// WFE is not a no-op here, which is the whole reason it is worth emitting. +/// +/// An implementation is free to make WFE a `nop`, and on such a machine a +/// WFE-based lock is a plain spinlock wearing a costume. This separates the +/// two: `nop` measured 0.3 ns and WFE measured 1336.7 ns, so anything within +/// an order of magnitude of `nop` means WFE is not idling. +#[test] +#[ignore = "timing, run with --ignored"] +fn wfe_actually_idles_rather_than_being_a_nop() { + let nop = nop_ns(); + let wfe = wfe_ns(); + assert!( + wfe > nop * 20.0, + "WFE at {wfe:.1} ns against nop at {nop:.1} ns: WFE is not idling on \ + this machine, so a WFE lock here is a spinlock with extra steps" + ); +} + +/// `core::hint::spin_loop()` is not WFE, and the gap is why the lock has to +/// write the instruction itself. +/// +/// `spin_loop()` is `__isb(SY)`, measured at 8.6 ns against WFE's 1336.7. If +/// these ever converge, either Rust started emitting WFE, in which case a lock +/// should stop hand-writing it, or WFE stopped idling. +#[test] +#[ignore = "timing, run with --ignored"] +fn the_standard_spin_hint_is_not_wfe() { + let isb = isb_ns(); + let wfe = wfe_ns(); + assert!( + wfe > isb * 10.0, + "spin_loop() at {isb:.1} ns and WFE at {wfe:.1} ns are within 10x. \ + core::hint::spin_loop() is __isb(SY) and should be far cheaper; if it \ + is not, check whether Rust now emits WFE and delete the hand-written one" + ); +} + +/// WFE returns on its own, so a waiter cannot deadlock if SEV is missed. +/// +/// This is what makes a WFE lock safe to write: the event register may already +/// be clear, or a SEV may land before the waiter reaches its WFE, and neither +/// wedges. Measured at roughly 1.3 us per WFE, so a hundred of them is +/// bounded well under a second on any machine where WFE idles at all. +#[test] +fn a_waiter_is_never_stuck_when_no_sev_arrives() { + // Drain any pending event so the WFEs below have nothing waiting for them. + // SAFETY: unprivileged, no operands. + unsafe { + core::arch::asm!("sev", options(nomem, nostack)); + core::arch::asm!("wfe", options(nomem, nostack)); + } + let now = Instant::now(); + for _ in 0..100 { + // SAFETY: as above. + unsafe { core::arch::asm!("wfe", options(nomem, nostack)) }; + } + assert!( + now.elapsed() < std::time::Duration::from_secs(5), + "100 WFEs with no SEV took {:?}: WFE is blocking indefinitely here, so \ + a lock must pair every one with a SEV rather than relying on the timeout", + now.elapsed() + ); +} + +/// The rule, as an executable statement: WFE does not yield to the scheduler. +/// +/// Two threads per core, one holding a lock for a long stretch. If WFE reached +/// the scheduler, waiters would stand aside and this would finish in about the +/// time the holders need. It does not, which is why the rule is "threads at +/// most cores". +/// +/// **Expected to be different under KVM**, where WFE traps and the hypervisor +/// yields the vCPU. On Graviton this test is the one to watch: if it starts +/// passing comfortably there, WFE became viable for oversubscribed hosts and +/// the guidance in this file needs revisiting. +#[test] +#[ignore = "heavy and oversubscribed, run with --ignored"] +fn wfe_does_not_yield_to_the_scheduler() { + use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; + use std::sync::Arc; + + let cores = std::thread::available_parallelism().map_or(8, std::num::NonZeroUsize::get); + let threads = cores * 8; + let held = Arc::new(AtomicBool::new(false)); + let done = Arc::new(AtomicU64::new(0)); + + let now = Instant::now(); + std::thread::scope(|scope| { + for _ in 0..threads { + let held = held.clone(); + let done = done.clone(); + scope.spawn(move || { + for _ in 0..8 { + while held + .compare_exchange_weak(false, true, Ordering::Acquire, Ordering::Relaxed) + .is_err() + { + // SAFETY: unprivileged, no operands. + unsafe { core::arch::asm!("wfe", options(nomem, nostack)) }; + } + let mut acc = 0u64; + for n in 0..20_000u64 { + acc = acc.wrapping_mul(0x9e37_79b9).wrapping_add(n); + } + black_box(acc); + held.store(false, Ordering::Release); + // SAFETY: as above. + unsafe { core::arch::asm!("sev", options(nomem, nostack)) }; + done.fetch_add(1, Ordering::Relaxed); + } + }); + } + }); + + assert_eq!(done.load(Ordering::Relaxed), threads as u64 * 8); + // Generous: this exists to prove the run terminates and to print the cost, + // not to pin a number that varies by machine. + assert!( + now.elapsed() < std::time::Duration::from_secs(120), + "{threads} threads on {cores} cores took {:?}", + now.elapsed() + ); +} From 19879a91e91b7808292cd9184d1ef6578576c6fc Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 18:01:19 +0700 Subject: [PATCH 6/9] Build without std, behind a default-on std feature A B-tree over `alloc` collections and an `ftree` that is already no_std. The only thing here that ever needed an operating system was one `yield_now`. ::core:: ops, fmt, slice, hint, borrow, marker, mem, iter, hash, cmp, sync::atomic alloc:: Arc, Vec, String, BTreeMap, vec!, format! The leading `::` is not decoration. This crate has its own module named `core`, so inside it `use core::x` resolves to `crate::core::x`. Four things wanted more than a rename: - `yield_now` in the two root-publication spin loops falls back to `spin_loop` without `std`, which is what the branch above it already does. - `published_keys` was a `HashMap`, keyed by a slot number and never iterated for order. It is a `BTreeMap` now, the same map without a hasher, and the one `alloc` has. - `ftree` is no_std, but its `serde` feature takes serde with default features, which is `std`. Asking for it unconditionally made every build a std build. It now arrives with this crate's `serde` feature, which says `std` out loud, as does `superslice-binary-search`: superslice is not a no_std crate. - `parking_lot` is now `parking_lot_lite_hack`, kept under the `parking_lot` name because that is what the code says. This crate uses `Mutex`, `RwLock` and `RawMutex`, and the fork has all three. The bench's own copy of upstream parking_lot is renamed `parking_lot_upstream`, because one name cannot mean both and that arm is deliberately the real one, with its std futex. Doc examples still say `std::`: a doctest compiles as a consumer. cargo test --all-features 127 + 1 + 2 + 107 doctests passed, 0 failed clippy, std and no_std clean x86_64-unknown-none the core builds; `concurrent` stops at ps-reclaim, whose TLS needs an OS --- Cargo.toml | 36 +++++++++++++------ benches/locks.rs | 10 +++--- src/concurrent/map.rs | 11 +++--- src/concurrent/multimap.rs | 13 ++++--- src/concurrent/operation.rs | 6 ++-- src/concurrent/ref.rs | 2 +- src/concurrent/set.rs | 71 +++++++++++++++++++++++-------------- src/core/multipair.rs | 3 +- src/core/multipair/ord.rs | 5 +-- src/core/node.rs | 29 ++++++++------- src/core/pair.rs | 9 ++--- src/lib.rs | 36 ++++++++++++------- 12 files changed, 144 insertions(+), 87 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 0db2f9f..b984d5f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -27,8 +27,10 @@ libc = "0.2" atomic-wait = "1.1" # parking_lot is optional in this crate (the `concurrent` feature). The bench # measures it against spin regardless of which features the library is built -# with, so it needs its own unconditional copy. -parking_lot = { version = "0.12", features = ["arc_lock"] } +# with, so it needs its own unconditional copy. Named `parking_lot_upstream` +# because the library's own `parking_lot` is now the no_std fork, and one name +# cannot mean both: this arm is deliberately the real one, with its std futex. +parking_lot_upstream = { package = "parking_lot", version = "0.12", features = ["arc_lock"] } # The crate is a fork of `indexset`, published under a different package name # to avoid colliding with upstream on the registry, and consumed everywhere as @@ -40,29 +42,43 @@ parking_lot = { version = "0.12", features = ["arc_lock"] } name = "indexset" [dependencies] -serde = { version = "1", optional = true, features = ["derive"] } -ftree = { version = "1", features = ["serde"] } -parking_lot = { version = "0.12", features = [ +serde = { version = "1", optional = true, default-features = false, features = ["derive", "alloc"] } +# `ftree` is no_std, but its `serde` feature takes serde with default features, +# which is `std`. Asking for it unconditionally made every build a std build, +# so it now arrives with this crate's own `serde` feature and nowhere else. +ftree = { version = "1", default-features = false } +# `parking_lot_lite_hack` is parking_lot's `Mutex` and `RwLock` with `std` made +# optional, kept under the `parking_lot` name because that is what the code +# says. This crate uses `Mutex`, `RwLock` and `RawMutex` and nothing the fork +# left behind. +parking_lot = { package = "parking_lot_lite_hack", version = "0.12.5", default-features = false, features = [ "send_guard", "arc_lock", ], optional = true } -ps-reclaim = { version = "^0.1, >=0.1.3", optional = true } -fastrand = { version = "2", optional = true } +ps-reclaim = { version = "^0.1, >=0.1.4", optional = true, default-features = false } +fastrand = { version = "2", optional = true, default-features = false } superslice = { version = "1", optional = true } wt-slice = { version = "0.1", optional = true } [features] -default = ["wt-slice-binary-search"] -serde = ["dep:serde"] +default = ["std", "wt-slice-binary-search"] +# Off, this crate does not link `std`. What it costs: `superslice` as a search +# strategy, and `yield_now` in the two publication spin loops, which fall back +# to a spin because there is no scheduler to yield to. +std = ["fastrand?/std", "parking_lot?/std", "ps-reclaim?/std", "serde?/std"] +serde = ["dep:serde", "ftree/serde", "std"] concurrent = [ "dep:parking_lot", "dep:ps-reclaim", + "ps-reclaim/libc", + "ps-reclaim/spin", ] cdc = ["concurrent"] multimap = ["concurrent", "dep:fastrand"] custom-binary-search = [] std-binary-search = [] -superslice-binary-search = ["dep:superslice"] +# `superslice` is not a no_std crate, so choosing it chooses `std`. +superslice-binary-search = ["dep:superslice", "std"] wt-slice-binary-search = ["dep:wt-slice"] [package.metadata.docs.rs] diff --git a/benches/locks.rs b/benches/locks.rs index 4facd99..f2f496a 100644 --- a/benches/locks.rs +++ b/benches/locks.rs @@ -12,9 +12,9 @@ //! //! | arm | what it is | //! |---|---| -//! | `parking_lot` | `parking_lot::Mutex`, named directly. What this crate uses today. | +//! | `parking_lot` | `parking_lot_upstream::Mutex`, named directly. What this crate uses today. | //! | `spin` | `spin::Mutex`, named directly. | -//! | `lock_api+parking_lot` | `lock_api::Mutex` | +//! | `lock_api+parking_lot` | `lock_api::Mutex` | //! | `lock_api+spin` | `lock_api::Mutex, T>` | //! //! The two `lock_api` arms are one generic body instantiated twice. That is the @@ -115,7 +115,7 @@ fn next(rng: &mut u64) -> u64 { /// Only the node lock varies. An earlier version changed both at once, which /// conflated them: a difference could have come from either, and the two /// tables could not be read against each other. -type IndexLock = parking_lot::RwLock>>; +type IndexLock = parking_lot_upstream::RwLock>>; macro_rules! nodes_arm { ($name:ident, $mx:ty) => { @@ -503,11 +503,11 @@ unsafe impl lock_api::RawMutex for FutexMutex { } } -type LaPlMx = lock_api::Mutex; +type LaPlMx = lock_api::Mutex; type LaSpMx = lock_api::Mutex, Node>; type LaYdMx = lock_api::Mutex, Node>; -contended_arm!(held_park, lock_api::Mutex); +contended_arm!(held_park, lock_api::Mutex); contended_arm!(held_spin, lock_api::Mutex, u64>); contended_arm!( held_yield, diff --git a/src/concurrent/map.rs b/src/concurrent/map.rs index 90f92c7..56d585f 100644 --- a/src/concurrent/map.rs +++ b/src/concurrent/map.rs @@ -1,5 +1,8 @@ -use std::fmt::{Debug, Display, Formatter}; -use std::{borrow::Borrow, iter::FusedIterator, ops::RangeBounds}; +use ::core::borrow::Borrow; +use ::core::fmt::{Debug, Display, Formatter}; +use ::core::iter::FusedIterator; +use ::core::ops::RangeBounds; +use alloc::vec::Vec; use super::set::BTreeSet; use crate::core::node::NodeLike; @@ -31,7 +34,7 @@ pub enum TopologyError { #[cfg(feature = "cdc")] impl Display for TopologyError { - fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result { + fn fmt(&self, formatter: &mut Formatter<'_>) -> ::core::fmt::Result { match self { Self::ZeroNodeCapacity => formatter.write_str("topology node capacity must be non-zero"), Self::EmptyNode { index } => write!(formatter, "topology node {index} is empty"), @@ -51,7 +54,7 @@ impl Display for TopologyError { } #[cfg(feature = "cdc")] -impl std::error::Error for TopologyError {} +impl ::core::error::Error for TopologyError {} #[derive(Debug)] pub struct BTreeMap>> diff --git a/src/concurrent/multimap.rs b/src/concurrent/multimap.rs index b166582..7f8f451 100644 --- a/src/concurrent/multimap.rs +++ b/src/concurrent/multimap.rs @@ -1,10 +1,9 @@ -use std::fmt::Debug; -use std::marker::PhantomData; -use std::{ - borrow::Borrow, - iter::FusedIterator, - ops::{Bound, RangeBounds}, -}; +use ::core::borrow::Borrow; +use ::core::fmt::Debug; +use ::core::iter::FusedIterator; +use ::core::marker::PhantomData; +use ::core::ops::{Bound, RangeBounds}; +use alloc::vec::Vec; use crate::core::node::NodeLike; use crate::{ diff --git a/src/concurrent/operation.rs b/src/concurrent/operation.rs index 83ea784..71c3d39 100644 --- a/src/concurrent/operation.rs +++ b/src/concurrent/operation.rs @@ -1,5 +1,7 @@ -use std::fmt::Debug; -use std::sync::Arc; +use ::core::fmt::Debug; +use alloc::sync::Arc; +use alloc::vec; +use alloc::vec::Vec; use parking_lot::RwLock; diff --git a/src/concurrent/ref.rs b/src/concurrent/ref.rs index 4b5c779..967aa2d 100644 --- a/src/concurrent/ref.rs +++ b/src/concurrent/ref.rs @@ -1,6 +1,6 @@ use crate::core::node::NodeLike; +use ::core::marker::PhantomData; use parking_lot::{ArcRwLockReadGuard, RawRwLock}; -use std::marker::PhantomData; /// A point reference that keeps its node read-locked. /// diff --git a/src/concurrent/set.rs b/src/concurrent/set.rs index 40d5065..e915082 100644 --- a/src/concurrent/set.rs +++ b/src/concurrent/set.rs @@ -1,13 +1,17 @@ +use ::core::borrow::Borrow; +use ::core::fmt::Debug; +use ::core::iter::FusedIterator; +use ::core::marker::PhantomData; +use ::core::ops::{Bound, RangeBounds}; +use ::core::sync::atomic::{AtomicPtr, AtomicU64, Ordering}; +use alloc::boxed::Box; +use alloc::collections::BTreeMap; +use alloc::sync::Arc; +use alloc::vec; +use alloc::vec::Vec; use parking_lot::{ ArcRwLockReadGuard, ArcRwLockWriteGuard, Mutex, MutexGuard, RawRwLock, RwLock, RwLockReadGuard, RwLockWriteGuard, }; -use std::collections::{BTreeMap, HashMap}; -use std::fmt::Debug; -use std::iter::FusedIterator; -use std::marker::PhantomData; -use std::ops::{Bound, RangeBounds}; -use std::sync::atomic::{AtomicPtr, AtomicU64, Ordering}; -use std::{borrow::Borrow, sync::Arc}; use crate::cdc::change::ChangeEvent; use crate::concurrent::operation::*; @@ -16,6 +20,19 @@ use crate::core::node::*; use super::r#ref::Ref; +/// Give the scheduler the core, when there is a scheduler to give it to. +/// +/// `yield_now` is a `std` call, and a `no_std` build has no thread to yield. +/// Spinning is the honest fallback there: it is what the caller was already +/// doing on the fast path, without the syscall that would make it wait longer. +#[inline] +fn yield_now() { + #[cfg(feature = "std")] + std::thread::yield_now(); + #[cfg(not(feature = "std"))] + ::core::hint::spin_loop(); +} + const ROOT_PUBLICATION_SPIN_LIMIT: usize = 16; const STABLE_READ_BLOCKING_FALLBACK_AFTER: usize = 2; const PUBLICATION_BACKLOG_DRAIN_THRESHOLD: usize = 64; @@ -147,7 +164,7 @@ where } let chunk = Arc::make_mut(&mut self.chunks[chunk_index]); match chunk.entries.binary_search_by(|(candidate, _)| candidate.cmp(&key)) { - Ok(index) => Some(std::mem::replace(&mut chunk.entries[index].1, node)), + Ok(index) => Some(::core::mem::replace(&mut chunk.entries[index].1, node)), Err(index) => { chunk.entries.insert(index, (key, node)); self.len += 1; @@ -306,7 +323,7 @@ pub(crate) struct Topology { // Writer-only reverse lookup from node identity to its current published // route key. This differs from the canonical key only for the last node, // whose stale route remains a valid final point-read fallback. - published_keys: Mutex>, + published_keys: Mutex>, published: PublishedIndex, // Even values are stable publications; odd values mean a writer may have // changed node contents or routing but has not published the new route. @@ -314,7 +331,7 @@ pub(crate) struct Topology { } impl Debug for Topology { - fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + fn fmt(&self, formatter: &mut ::core::fmt::Formatter<'_>) -> ::core::fmt::Result { formatter .debug_struct("Topology") .field("nodes", &self.index.read().len()) @@ -327,7 +344,7 @@ impl Topology { fn new() -> Self { Self { index: RwLock::new(BTreeMap::new()), - published_keys: Mutex::new(HashMap::new()), + published_keys: Mutex::new(BTreeMap::new()), published: PublishedIndex::new(), generation: AtomicU64::new(0), } @@ -403,12 +420,12 @@ where // reclamation domain. Range readers should not wait for a registry scan. index: Option>>, published: Option>, - published_keys: Option>>, + published_keys: Option>>, publish: bool, dirty: bool, } -impl std::ops::Deref for TopologyWriteGuard<'_, T, Node> +impl ::core::ops::Deref for TopologyWriteGuard<'_, T, Node> where T: Ord + Clone + Send + 'static, Node: Send + 'static, @@ -777,12 +794,12 @@ where let mut last_equal = None; for entry @ (key, _) in index { match >::borrow(key).cmp(end) { - std::cmp::Ordering::Less => {} - std::cmp::Ordering::Equal => last_equal = Some(entry), + ::core::cmp::Ordering::Less => {} + ::core::cmp::Ordering::Equal => last_equal = Some(entry), // The first node whose maximum is above the end may still start // with values inside the range. `Range::new` ranks within that // node to obtain the first out-of-range sentinel. - std::cmp::Ordering::Greater => return Some(entry), + ::core::cmp::Ordering::Greater => return Some(entry), } } last_equal.or_else(|| index.last_key_value()) @@ -803,8 +820,8 @@ where /// Iterators returned by [`crate::BTreeSet::iter`] produce their items in order, and take worst-case /// logarithmic and amortized constant time per item returned. /// -/// [`Cell`]: crate::core::cell::Cell -/// [`RefCell`]: crate::core::cell::RefCell +/// [`Cell`]: cratestd::cell::Cell +/// [`RefCell`]: cratestd::cell::RefCell /// /// # Examples /// @@ -909,7 +926,7 @@ where self } pub fn attach_node(&self, node: Node) { - self.attach_nodes(std::iter::once(node)); + self.attach_nodes(::core::iter::once(node)); } /// Attaches a persisted topology in one structural publication. @@ -975,7 +992,7 @@ where break self.index.write(); } spins += 1; - std::hint::spin_loop(); + ::core::hint::spin_loop(); }; // Another first writer may have published while this // caller was acquiring the exclusive structural guard. @@ -1393,9 +1410,9 @@ where if !generation.is_multiple_of(2) { if writer_spins < ROOT_PUBLICATION_SPIN_LIMIT { writer_spins += 1; - std::hint::spin_loop(); + ::core::hint::spin_loop(); } else { - std::thread::yield_now(); + yield_now(); } continue; } @@ -1480,9 +1497,9 @@ where if !generation.is_multiple_of(2) { if writer_spins < ROOT_PUBLICATION_SPIN_LIMIT { writer_spins += 1; - std::hint::spin_loop(); + ::core::hint::spin_loop(); } else { - std::thread::yield_now(); + yield_now(); } continue; } @@ -1669,8 +1686,8 @@ where Node: NodeLike + Send + 'static, { tree: &'a BTreeSet, - current_front_batch: Option>, - current_back_batch: Option>, + current_front_batch: Option>, + current_back_batch: Option>, // Identity of the node the last batch in each direction was cloned from, // so the next install can step past it when the cursor lookup lands on // it again (its entry key can sit past every element it still holds). @@ -2361,7 +2378,7 @@ where .remove::(&key) .expect("middle key was collected under the write lock"); let mut removed_node = node.write_arc(); - detached_nodes.push(std::mem::take(&mut *removed_node)); + detached_nodes.push(::core::mem::take(&mut *removed_node)); } // Trim the front node from the start position: its maximum goes away, diff --git a/src/core/multipair.rs b/src/core/multipair.rs index 09fb533..0524167 100644 --- a/src/core/multipair.rs +++ b/src/core/multipair.rs @@ -1,7 +1,8 @@ use crate::cdc::change::ChangeEvent; use crate::concurrent::set::BTreeSet; use crate::core::node::NodeLike; -use std::fmt::Debug; +use ::core::fmt::Debug; +use alloc::vec::Vec; pub mod ord; pub use ord::OrdMultiPair; diff --git a/src/core/multipair/ord.rs b/src/core/multipair/ord.rs index 1e03d19..38c36f9 100644 --- a/src/core/multipair/ord.rs +++ b/src/core/multipair/ord.rs @@ -1,5 +1,6 @@ -use std::borrow::Borrow; -use std::fmt::Debug; +use ::core::borrow::Borrow; +use ::core::fmt::Debug; +use alloc::vec::Vec; use core::cmp::Ordering; #[cfg(feature = "serde")] diff --git a/src/core/node.rs b/src/core/node.rs index 5fcd5e9..16db031 100644 --- a/src/core/node.rs +++ b/src/core/node.rs @@ -1,5 +1,6 @@ +use ::core::ops::Deref; +use alloc::vec::Vec; use core::borrow::Borrow; -use std::ops::Deref; pub trait NodeLike { #[allow(dead_code)] @@ -31,7 +32,7 @@ pub trait NodeLike { where T: Borrow; #[allow(dead_code)] - fn rank(&self, bound: std::ops::Bound<&Q>, from_start: bool) -> Option + fn rank(&self, bound: ::core::ops::Bound<&Q>, from_start: bool) -> Option where T: Borrow; #[allow(dead_code)] @@ -52,7 +53,7 @@ pub trait NodeLike { #[allow(dead_code)] fn min(&self) -> Option<&T>; #[allow(dead_code)] - fn iter<'a>(&'a self) -> std::slice::Iter<'a, T> + fn iter<'a>(&'a self) -> ::core::slice::Iter<'a, T> where T: 'a; } @@ -192,26 +193,30 @@ pub(crate) fn search_by(haystack: &[T], mut compare: impl FnMut(&T) -> core:: } #[inline] -fn compute_positions_to_skip(haystack: &[T], bound: std::ops::Bound<&Q>, forward: bool) -> Option +fn compute_positions_to_skip(haystack: &[T], bound: ::core::ops::Bound<&Q>, forward: bool) -> Option where T: Borrow + Ord, Q: Ord + ?Sized, { let skipped = match (bound, forward) { // A forward iterator skips values before the start bound. - (std::ops::Bound::Included(value), true) => haystack.partition_point(|item| item.borrow().cmp(value).is_lt()), - (std::ops::Bound::Excluded(value), true) => haystack.partition_point(|item| item.borrow().cmp(value).is_le()), + (::core::ops::Bound::Included(value), true) => { + haystack.partition_point(|item| item.borrow().cmp(value).is_lt()) + } + (::core::ops::Bound::Excluded(value), true) => { + haystack.partition_point(|item| item.borrow().cmp(value).is_le()) + } // A backward iterator skips values after the end bound. - (std::ops::Bound::Included(value), false) => { + (::core::ops::Bound::Included(value), false) => { let first_greater = haystack.partition_point(|item| item.borrow().cmp(value).is_le()); haystack.len() - first_greater } - (std::ops::Bound::Excluded(value), false) => { + (::core::ops::Bound::Excluded(value), false) => { let first_equal = haystack.partition_point(|item| item.borrow().cmp(value).is_lt()); haystack.len() - first_equal } - (std::ops::Bound::Unbounded, _) => return None, + (::core::ops::Bound::Unbounded, _) => return None, }; // Callers use this as the index of the last value to skip. No skipped @@ -271,7 +276,7 @@ impl NodeLike for Vec { search(self, value).ok() } #[inline] - fn rank(&self, bound: std::ops::Bound<&Q>, from_start: bool) -> Option + fn rank(&self, bound: ::core::ops::Bound<&Q>, from_start: bool) -> Option where T: Borrow + Ord, Q: Ord + ?Sized, @@ -301,7 +306,7 @@ impl NodeLike for Vec { #[inline] fn replace(&mut self, idx: usize, value: T) -> Option { if let Some(old) = self.get_mut(idx) { - let old = std::mem::replace(old, value); + let old = ::core::mem::replace(old, value); return Some(old); } @@ -316,7 +321,7 @@ impl NodeLike for Vec { self.first() } #[inline] - fn iter<'a>(&'a self) -> std::slice::Iter<'a, T> + fn iter<'a>(&'a self) -> ::core::slice::Iter<'a, T> where T: 'a, { diff --git a/src/core/pair.rs b/src/core/pair.rs index 9ec184f..a4e7025 100644 --- a/src/core/pair.rs +++ b/src/core/pair.rs @@ -1,7 +1,8 @@ +use ::core::borrow::Borrow; +use alloc::string::String; use core::cmp::Ordering; #[cfg(feature = "serde")] use serde::{Deserialize, Serialize}; -use std::borrow::Borrow; #[cfg_attr(feature = "serde", derive(Serialize, Deserialize))] #[derive(Debug, Default, Clone)] @@ -39,11 +40,11 @@ where } } -impl std::hash::Hash for Pair +impl ::core::hash::Hash for Pair where - K: std::hash::Hash, + K: ::core::hash::Hash, { - fn hash(&self, state: &mut H) { + fn hash(&self, state: &mut H) { self.key.hash(state); } } diff --git a/src/lib.rs b/src/lib.rs index ceb72c0..1a6731e 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1,3 +1,15 @@ +#![cfg_attr(not(any(test, feature = "std")), no_std)] +// The crate does not link `std` unless asked. Tests always do, because a +// concurrency test is made of threads and clocks; with the `std` feature on, +// the library links it too, and the only thing it takes from it is +// `thread::yield_now`. +#[cfg(any(test, feature = "std"))] +extern crate std; + +extern crate alloc; + +use alloc::vec; +use alloc::vec::Vec; #[cfg(feature = "concurrent")] pub mod concurrent; @@ -7,18 +19,18 @@ pub mod cdc; pub mod core; use crate::Entry::{Occupied, Vacant}; +use ::core::borrow::Borrow; +use ::core::cmp::Ordering; +use ::core::iter::FusedIterator; +use ::core::mem::swap; +use ::core::ops::Bound; +use ::core::ops::{Index, RangeBounds}; use core::constants::DEFAULT_INNER_SIZE; use core::node::*; use core::pair::Pair; use ftree::FenwickTree; #[cfg(feature = "serde")] use serde::{Deserialize, Serialize}; -use std::borrow::Borrow; -use std::cmp::Ordering; -use std::collections::Bound; -use std::iter::FusedIterator; -use std::mem::swap; -use std::ops::{Index, RangeBounds}; type Node = Vec; @@ -1333,8 +1345,8 @@ where current_front_idx: usize, current_back_node_idx: usize, current_back_idx: usize, - current_front_iterator: Option>, - current_back_iterator: Option>, + current_front_iterator: Option<::core::slice::Iter<'a, T>>, + current_back_iterator: Option<::core::slice::Iter<'a, T>>, } impl<'a, T> Iter<'a, T> @@ -2263,7 +2275,7 @@ where node_idx, position_within_node, } => { - std::mem::swap(&mut self.set.inner[node_idx][position_within_node].value, &mut value); + ::core::mem::swap(&mut self.set.inner[node_idx][position_within_node].value, &mut value); Some(value) } NodeEntry::Empty { node_idx } => { @@ -3466,13 +3478,13 @@ pub struct IterMut<'a, K: 'a, V: 'a> where K: Ord, { - inner: std::slice::IterMut<'a, Node>>, + inner: ::core::slice::IterMut<'a, Node>>, current_front_node_idx: usize, current_front_idx: usize, current_back_node_idx: usize, current_back_idx: usize, - current_front_iterator: std::slice::IterMut<'a, Pair>, - current_back_iterator: std::slice::IterMut<'a, Pair>, + current_front_iterator: ::core::slice::IterMut<'a, Pair>, + current_back_iterator: ::core::slice::IterMut<'a, Pair>, } impl<'a, K, V> Iterator for IterMut<'a, K, V> From cd13369373466c5835d47a2aae1bced569dcabb0 Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 18:01:27 +0700 Subject: [PATCH 7/9] chore: 0.0.13 --- Cargo.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Cargo.toml b/Cargo.toml index b984d5f..0e9630c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "WorkTablesIndex" -version = "0.0.12" +version = "0.0.13" edition = "2021" documentation = "https://docs.rs/WorkTablesIndex/" repository = "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/pathscale/WorkTablesIndex" From d2cf1c4449e93b3f4f7e32009aa68759c05b461e Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 18:08:17 +0700 Subject: [PATCH 8/9] Say the requirements as carets `"^0.1, >=0.1.4"` and `"0.1.4"` describe the same set, and `crossbeam-skiplist` carried a `"^0.1"` that a bare `"0.1"` already means. One form, used once. --- Cargo.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 0e9630c..4794b26 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,7 +12,7 @@ readme = "README.md" [dev-dependencies] criterion = { version = "0.5", features = ["html_reports"] } -crossbeam-skiplist = "^0.1" +crossbeam-skiplist = "0.1" loom = "0.7" rand = "0.9" scc = { version = "2.2.5" } @@ -55,7 +55,7 @@ parking_lot = { package = "parking_lot_lite_hack", version = "0.12.5", default-f "send_guard", "arc_lock", ], optional = true } -ps-reclaim = { version = "^0.1, >=0.1.4", optional = true, default-features = false } +ps-reclaim = { version = "0.1.4", optional = true, default-features = false } fastrand = { version = "2", optional = true, default-features = false } superslice = { version = "1", optional = true } wt-slice = { version = "0.1", optional = true } From ffdda8c5ba58e9e361518f09302347eec9807321 Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 18:23:47 +0700 Subject: [PATCH 9/9] Take parking_lot_lite_hack 0.12.6, which is the one that is no_std 0.12.5 depended on its own core without `default-features = false`, so the core linked `std` in every build and this crate's no_std column was false one level down. 0.12.6 is that fix, published today. --no-default-features parking_lot_lite_hack [arc_lock,send_guard] core [] --features std parking_lot_lite_hack [arc_lock,send_guard,std] core [std] --- Cargo.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Cargo.toml b/Cargo.toml index 4794b26..5e72780 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -51,7 +51,7 @@ ftree = { version = "1", default-features = false } # optional, kept under the `parking_lot` name because that is what the code # says. This crate uses `Mutex`, `RwLock` and `RawMutex` and nothing the fork # left behind. -parking_lot = { package = "parking_lot_lite_hack", version = "0.12.5", default-features = false, features = [ +parking_lot = { package = "parking_lot_lite_hack", version = "0.12.6", default-features = false, features = [ "send_guard", "arc_lock", ], optional = true }