Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 17 additions & 2 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -13,16 +13,31 @@ jobs:
- uses: dtolnay/rust-toolchain@stable
with:
components: rustfmt, clippy
targets: thumbv7em-none-eabihf
targets: x86_64-unknown-none
- uses: Swatinem/rust-cache@v2
- run: cargo fmt --all -- --check
- run: cargo clippy --all-targets --no-default-features -- -D warnings
- run: cargo test --all-targets --no-default-features
- run: cargo check --lib --no-default-features --target thumbv7em-none-eabihf
- run: cargo clippy --all-targets --features arctic -- -D warnings
- run: cargo test --all-targets --features arctic
- run: cargo clippy --all-targets --all-features -- -D warnings
- run: cargo test --all-targets --all-features
# A crate that says `#![no_std]` and is only ever built on a host with one
# proves nothing. This target has no `std` to find.
- run: cargo build --target x86_64-unknown-none --no-default-features
- run: cargo build --target x86_64-unknown-none --no-default-features --features hydrate
# The backends need an OS for their thread-local storage, so they cannot
# be built for a bare target - but none of them may pull `std` in.
- name: No backend enables std
run: |
for features in arctic congee wti arctic,congee,wti; do
graph=$(cargo tree -e normal --no-default-features --features "$features" -f '{p} [{f}]')
echo "--- $features"
echo "$graph" | grep -E 'arctic-wt|congee-wt|WorkTablesIndex|ps-reclaim|parking_lot'
if echo "$graph" | grep -E '(arctic-wt|congee-wt|WorkTablesIndex|ps-reclaim|parking_lot[a-z_]*) v[^[]*\[[^]]*std'; then
echo "std leaked into --features $features"; exit 1
fi
done

package:
name: Release dry run
Expand Down
38 changes: 34 additions & 4 deletions Cargo.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[package]
name = "worktable-vec"
version = "0.1.3"
version = "0.1.5"
edition = "2024"
rust-version = "1.85"
license = "MIT OR Apache-2.0"
Expand All @@ -12,11 +12,41 @@ categories = ["data-structures"]

[features]
default = []
# On by default for nobody: this crate is `#![no_std]` and the core never needed
# `std`. The feature exists so a backend that does need one can say so, and so a
# caller without a standard library can prove it is not getting one.
std = ["arctic?/std", "congee?/std", "wti?/std"]
# Load and unload rows as pages. Optional because it is the only thing here
# that needs a serializer.
hydrate = ["dep:rkyv", "dep:embedded-io"]
arctic = ["dep:arctic"]
congee = ["dep:congee"]
wti = ["dep:wti"]

[dependencies]
arctic = { package = "arctic-wt", version = "^0.1, >=0.1.9", optional = true }
congee = { package = "congee-wt", version = "^0.4, >=0.4.4", optional = true }
wti = { package = "WorkTablesIndex", version = "^0.0, >=0.0.11", optional = true, default-features = false, features = ["concurrent"] }
rkyv = { version = "0.8.17", default-features = false, features = ["alloc", "bytecheck"], optional = true }
# The I/O is a trait, not a filesystem. `embedded-io` already implements it for
# `&[u8]` and `Vec<u8>`, and `embedded-io-adapters` wraps a `std::fs::File`, so
# a caller with an OS gets real files and a caller without one plugs in whatever
# it has. Nothing here needs `std` either way.
embedded-io = { version = "0.7", default-features = false, features = ["alloc"], optional = true }
# `default-features = false` and an explicit SMR backend, so enabling `arctic`
# does not silently enable `std`. 0.1.11 is the first release where that is
# possible: before it, `smr-ps-reclaim` forced `std` on.
arctic = { package = "arctic-wt", version = "0.1.11", optional = true, default-features = false, features = ["smr-ps-reclaim"] }
congee = { package = "congee-wt", version = "0.4.5", optional = true, default-features = false }
wti = { package = "WorkTablesIndex", version = "0.0.13", optional = true, default-features = false, features = ["concurrent"] }

[dev-dependencies]
# Only the tests need a real file, and this is what turns one into the trait.
embedded-io-adapters = { version = "0.7", features = ["std"] }
rkyv = { version = "0.8.17", features = ["alloc", "bytecheck"] }

[[example]]
name = "disk_cost"
# It measures the page codec, so it needs it.
required-features = ["hydrate"]

[[example]]
name = "where_time_goes"
required-features = ["hydrate"]
82 changes: 82 additions & 0 deletions examples/disk_cost.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,82 @@
//! What a flush and an open cost against a real file.
//!
//! Bounded on purpose: fixed row counts, fixed reps, and the fixture is built
//! once. A benchmark without a ceiling is a hang.

use std::time::Instant;

use embedded_io_adapters::std::FromStd;
use worktable_vec::LinearTable;

const ROWS: usize = 200_000;
const REPS: usize = 5;

fn main() -> Result<(), Box<dyn std::error::Error>> {
let mut table = LinearTable::new();
for n in 0..ROWS as u64 {
table.push(
n,
format!("row {n} with enough text to be worth serializing"),
);
}
let path = std::env::temp_dir().join("worktable-vec-disk-cost.wtv");

let mut wrote = Vec::new();
let mut read = Vec::new();
let mut bytes = 0u64;

for _ in 0..REPS {
let now = Instant::now();
{
let file = std::fs::File::create(&path)?;
table.unload_to(&mut FromStd::new(file))?;
}
wrote.push(now.elapsed());
bytes = std::fs::metadata(&path)?.len();

let now = Instant::now();
let back =
LinearTable::<u64, String>::load_from(&mut FromStd::new(std::fs::File::open(&path)?))
.map_err(|error| format!("{error}"))?;
read.push(now.elapsed());
assert_eq!(back.len(), ROWS);
}

wrote.sort();
read.sort();
let flush = wrote[REPS / 2];
let open = read[REPS / 2];
let mb = bytes as f64 / (1024.0 * 1024.0);

println!(
"\n{ROWS} rows, {mb:.1} MiB on disk, {} pages, median of {REPS}\n",
bytes as usize / (4096 * 4)
);
println!(
" flush {:>7.1} ms {:>7.1} MiB/s {:>6.0} ns/row",
flush.as_secs_f64() * 1e3,
mb / flush.as_secs_f64(),
flush.as_secs_f64() * 1e9 / ROWS as f64
);
println!(
" open {:>7.1} ms {:>7.1} MiB/s {:>6.0} ns/row",
open.as_secs_f64() * 1e3,
mb / open.as_secs_f64(),
open.as_secs_f64() * 1e9 / ROWS as f64
);

// The floor: what the same rows cost with no pages, no checksum and no
// fingerprint, just one archive straight to the file. Anything this codec
// adds shows up as the gap.
let now = Instant::now();
let raw = rkyv::to_bytes::<rkyv::rancor::Error>(&table.rows().to_vec())?;
let encode = now.elapsed();
println!(
"\n rkyv alone, no pages {:>7.1} ms encode, {} MiB",
encode.as_secs_f64() * 1e3,
raw.len() / (1024 * 1024)
);

let _ = std::fs::remove_file(&path);
Ok(())
}
61 changes: 61 additions & 0 deletions examples/where_time_goes.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
//! What an open actually spends its time on, before anyone parallelises it.
//!
//! A speedup on a synthetic loop says nothing about this path. Page decode is
//! the only embarrassingly parallel part, and it is only worth threads if it is
//! most of the time.

use std::hint::black_box;
use std::time::Instant;

use worktable_vec::{LinearTable, PAGE_SIZE};

const ROWS: usize = 200_000;
const REPS: usize = 5;

fn main() -> Result<(), Box<dyn std::error::Error>> {
let mut table = LinearTable::new();
for n in 0..ROWS as u64 {
table.push(
n,
format!("row {n} with enough text to be worth serializing"),
);
}
let bytes = table.unload()?;
let pages = bytes.len() / PAGE_SIZE;

let mut whole = Vec::new();
let mut per_page = Vec::new();
for _ in 0..REPS {
let now = Instant::now();
let back = LinearTable::<u64, String>::load(black_box(&bytes))?;
whole.push(now.elapsed());
black_box(back.len());

// Every page decoded on its own, which is what a worker would do, with
// no vector to append into and no ordering to preserve.
let now = Instant::now();
let mut n = 0usize;
for page in bytes.chunks_exact(PAGE_SIZE) {
n += LinearTable::<u64, String>::load(black_box(page))?.len();
}
per_page.push(now.elapsed());
black_box(n);
}
whole.sort();
per_page.sort();
let (w, p) = (whole[REPS / 2], per_page[REPS / 2]);
println!("\n{ROWS} rows, {pages} pages, median of {REPS}\n");
println!(
" load, whole file {:>7.1} ms",
w.as_secs_f64() * 1e3
);
println!(
" page decode alone {:>7.1} ms {:.0}% of it",
p.as_secs_f64() * 1e3,
100.0 * p.as_secs_f64() / w.as_secs_f64()
);
println!("\n Amdahl ceiling at 16 cores, if only decode parallelises:");
let s = 1.0 - p.as_secs_f64() / w.as_secs_f64();
println!(" {:.2}x", 1.0 / (s + (1.0 - s) / 16.0));
Ok(())
}
Loading
Loading