From f96df450bef37e111bc428e0e61829021b20f34b Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 02:41:30 +0700 Subject: [PATCH 01/16] Give AtomicKeyTable the vocabulary the rest of WorkTable uses The first version invented a third dialect. `LinearTable` in this crate already says select, len, is_empty, capacity, and WorkTable itself says insert, select, upsert - and the new table said find_or_claim, find, claimed. A caller moving between the two tables here had to learn both. find_or_claim -> upsert returns the existing row or creates one, never replaces. The difference from a WorkTable upsert is that the row is then written through &V, since the value carries its own interior mutability. find -> select same name and shape as LinearTable::select. claimed -> len and is_empty beside it, which clippy also wanted. No behaviour changed and the nine tests are the same nine. --- Cargo.toml | 2 +- src/lib.rs | 72 ++++++++++++++++++++++++++++++++---------------------- 2 files changed, 44 insertions(+), 30 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 91642ab..84a799f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "worktable-vec" -version = "0.1.3" +version = "0.1.4" edition = "2024" rust-version = "1.85" license = "MIT OR Apache-2.0" diff --git a/src/lib.rs b/src/lib.rs index d79ffe0..bc177a2 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -586,7 +586,7 @@ mod tests { /// # What it does not do /// /// No removal, no resize, and no iteration order beyond slot order. A full table refuses rather -/// than growing, and [`AtomicKeyTable::claimed`] says how many slots are taken so a caller can +/// than growing, and [`AtomicKeyTable::len`] says how many slots are taken so a caller can /// see it coming. Keys are `u64` and zero is the empty sentinel, so a caller whose key is a /// pointer or a hash maps it in. `usize` rather than `u64` because `AtomicU64` does not exist /// on 32-bit bare-metal targets such as `thumbv7em-none-eabi`, and this crate builds for them. @@ -648,9 +648,14 @@ impl AtomicKeyTable { impl AtomicKeyTable { /// The row for this key, claiming a slot if it has none yet. /// + /// Named for `WorkTable`'s `upsert`: it returns the existing row or creates one, and never + /// replaces what is there. The row is then updated through `&V`, which is where the + /// difference from a `WorkTable` upsert lies - the value carries its own interior mutability + /// rather than being written back whole. + /// /// `None` means the table is full. Zero is the empty sentinel and is rejected rather than /// silently colliding with an unclaimed slot. - pub fn find_or_claim(&self, key: usize) -> Option<&V> { + pub fn upsert(&self, key: usize) -> Option<&V> { if key == 0 || self.keys.is_empty() { return None; } @@ -677,8 +682,11 @@ impl AtomicKeyTable { None } - /// The row for this key, or `None` if nothing has claimed it. Never claims. - pub fn find(&self, key: usize) -> Option<&V> { + /// The row for this key, or `None` if no row has been created for it. Never creates one. + /// + /// The same name and shape as [`LinearTable::select`], so a caller moving between the two + /// tables in this crate reads one vocabulary. + pub fn select(&self, key: usize) -> Option<&V> { if key == 0 || self.keys.is_empty() { return None; } @@ -707,12 +715,17 @@ impl AtomicKeyTable { ) } - /// How many slots hold a key. - pub fn claimed(&self) -> usize { + /// How many rows the table holds. + pub fn len(&self) -> usize { self.iter().count() } - /// How many slots there are. + /// Whether any row has been created. Matches [`LinearTable::is_empty`]. + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + /// How many rows the table can hold. Fixed at construction. pub fn capacity(&self) -> usize { self.keys.len() } @@ -729,52 +742,49 @@ mod atomic_key_table_tests { #[test] fn a_claimed_row_is_found_by_a_plain_load_and_never_reclaimed() { let table: AtomicKeyTable = AtomicKeyTable::with_capacity(64); - let first = table.find_or_claim(7).expect("capacity"); + let first = table.upsert(7).expect("capacity"); first.0.fetch_add(1, Ordering::Relaxed); - let again = table.find_or_claim(7).expect("already claimed"); + let again = table.upsert(7).expect("already claimed"); again.0.fetch_add(1, Ordering::Relaxed); assert_eq!( again.0.load(Ordering::Relaxed), 2, "the second call found the same row" ); - assert_eq!(table.claimed(), 1, "one key claimed one slot"); + assert_eq!(table.len(), 1, "one key claimed one slot"); } #[test] fn zero_is_the_empty_sentinel_and_is_refused_rather_than_colliding() { let table: AtomicKeyTable = AtomicKeyTable::with_capacity(8); assert!( - table.find_or_claim(0).is_none(), + table.upsert(0).is_none(), "zero would be indistinguishable from empty" ); - assert_eq!(table.claimed(), 0); + assert_eq!(table.len(), 0); } #[test] fn a_full_table_refuses_rather_than_growing() { let table: AtomicKeyTable = AtomicKeyTable::with_capacity(4); for key in 1..=4 { - assert!(table.find_or_claim(key).is_some(), "slot {key} fits"); + assert!(table.upsert(key).is_some(), "slot {key} fits"); } - assert_eq!(table.claimed(), 4); - assert!( - table.find_or_claim(5).is_none(), - "the fifth has nowhere to go" - ); + assert_eq!(table.len(), 4); + assert!(table.upsert(5).is_none(), "the fifth has nowhere to go"); assert!( - table.find_or_claim(3).is_some(), + table.upsert(3).is_some(), "a claimed key is still reachable when full" ); } #[test] - fn find_never_claims() { + fn select_never_creates_a_row() { let table: AtomicKeyTable = AtomicKeyTable::with_capacity(8); - assert!(table.find(9).is_none()); - assert_eq!(table.claimed(), 0, "find must not take a slot"); - table.find_or_claim(9).expect("capacity"); - assert!(table.find(9).is_some()); + assert!(table.select(9).is_none()); + assert_eq!(table.len(), 0, "find must not take a slot"); + table.upsert(9).expect("capacity"); + assert!(table.select(9).is_some()); } #[test] @@ -782,7 +792,7 @@ mod atomic_key_table_tests { let table: AtomicKeyTable = AtomicKeyTable::with_capacity(32); for key in [11usize, 22, 33] { table - .find_or_claim(key) + .upsert(key) .expect("capacity") .0 .store(key as u64, Ordering::Relaxed); @@ -806,7 +816,7 @@ mod atomic_key_table_tests { for round in 0..1_000usize { let key = (round % 16) + 1; shared - .find_or_claim(key) + .upsert(key) .expect("capacity") .0 .fetch_add(1, Ordering::Relaxed); @@ -815,7 +825,7 @@ mod atomic_key_table_tests { } }); assert_eq!( - table.claimed(), + table.len(), 16, "sixteen keys, sixteen slots, whatever the interleaving" ); @@ -855,7 +865,7 @@ mod atomic_key_table_cost { let atomic: AtomicKeyTable = AtomicKeyTable::with_capacity(KEYS * 4); let mut linear: LinearTable = LinearTable::new(); for key in 1..=KEYS { - atomic.find_or_claim(key).expect("capacity"); + atomic.upsert(key).expect("capacity"); linear.push(key, key as u64); } @@ -863,7 +873,11 @@ mod atomic_key_table_cost { let mut sink = 0u64; for round in 0..ROUNDS { let key = (round % KEYS) + 1; - sink += atomic.find(key).expect("claimed").0.load(Ordering::Relaxed); + sink += atomic + .select(key) + .expect("claimed") + .0 + .load(Ordering::Relaxed); } let atomic_ns = start.elapsed().as_nanos().max(1); From 2bc1a02cd18574674b7d0a1786eadc9f72d3f158 Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 12:39:07 +0700 Subject: [PATCH 02/16] Load and unload a table as pages, which is the whole hydrate interface let bytes = table.unload(); let table = LinearTable::load(&bytes)?; Rows live in a `Vec` while the table is in use and are pages only at rest, so everything between a load and an unload runs at `Vec` speed because it is a `Vec`. The format is a codec at the two ends rather than a storage engine underneath, which is the thing this crate exists not to be. **No I/O traits.** A load takes `&[u8]` and an unload returns `Vec`. Nothing seeks and nothing reads incrementally, so the question of which read/seek/write abstraction to adopt does not reach this far, and the crate stays `no_std` with no runtime and no ecosystem attached. Whoever holds the bytes decides how they got there. The bytes are DataBucket data pages: a `GeneralHeader` per page, bodies concatenating into one rkyv archive of the row vector. Not a WorkTable space file, because there is no schema here to write a `SpaceInfoPage` about, and the module says so rather than implying compatibility it does not have. **A fingerprint page, because rkyv will not refuse.** Loading `Vec<(u64, String)>` bytes as `Vec<(u64, u64)>` first succeeded and returned correct-looking keys with values like 18446743180901052274, which is a `String`'s relative pointer read as an integer. Nothing errored. That is the reinterpretation DataBucket's own migration doc warns about, and validation cannot catch it because a `(u64, u64)` archive has no invalid bit patterns. So page 0 now carries a hash of the row type and a mismatch is refused by name. The hash is FNV-1a over `type_name`: not stable across compilers and not unique, which is why it fails toward refusing a load rather than toward accepting one. The body comes back in an `AlignedVec`. rkyv reads an archive in place and needs it aligned, and a `Vec` is aligned to 1; the round trip passing before this was the allocator being generous, not a guarantee. The index is derived rather than stored, so a load rebuilds it in one pass instead of carrying a second thing on disk that can disagree with the rows. **Cannot merge yet:** `data_bucket_format` is unpublished, so the dependency is a path into a sibling checkout. It is behind the optional `hydrate` feature, and the default build is untouched. --- Cargo.toml | 8 + src/hydrate.rs | 438 +++++++++++++++++++++++++++++++++++++++++++++++++ src/lib.rs | 5 + 3 files changed, 451 insertions(+) create mode 100644 src/hydrate.rs diff --git a/Cargo.toml b/Cargo.toml index 84a799f..ee36360 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,11 +12,19 @@ categories = ["data-structures"] [features] default = [] +# Load and unload rows as DataBucket data pages. Optional because it is the +# only thing here that reaches outside the crate for a format. +hydrate = ["dep:data_bucket_format", "dep:rkyv"] arctic = ["dep:arctic"] congee = ["dep:congee"] wti = ["dep:wti"] [dependencies] +# Path, not crates.io: data_bucket_format is new and unpublished, so this +# branch cannot merge until it ships. Nothing else here depends on a +# working checkout. +data_bucket_format = { path = "../DataBucket/format", default-features = false, optional = true } +rkyv = { version = "0.8.17", default-features = false, features = ["alloc", "bytecheck"], optional = true } arctic = { package = "arctic-wt", version = "^0.1, >=0.1.9", optional = true } congee = { package = "congee-wt", version = "^0.4, >=0.4.4", optional = true } wti = { package = "WorkTablesIndex", version = "^0.0, >=0.0.11", optional = true, default-features = false, features = ["concurrent"] } diff --git a/src/hydrate.rs b/src/hydrate.rs new file mode 100644 index 0000000..36fc48b --- /dev/null +++ b/src/hydrate.rs @@ -0,0 +1,438 @@ +//! Load and unload a table as DataBucket data pages. +//! +//! # The whole interface +//! +//! ```ignore +//! let bytes = table.unload(); // rows out, as pages +//! let table = LinearTable::load(&bytes)?; // rows back in +//! ``` +//! +//! Rows live in a `Vec` while the table is in use and are pages only at rest. +//! That is the point: everything between a load and an unload runs at `Vec` +//! speed because it *is* a `Vec`, and the format is a codec at the two ends +//! rather than a storage engine underneath. +//! +//! # Why there are no I/O traits here +//! +//! A load takes `&[u8]` and an unload returns `Vec`. Nothing seeks and +//! nothing reads incrementally, so this needs no `Read`/`Seek`/`Write` +//! abstraction and adopts nobody's runtime. Whoever holds the bytes decides +//! how they got there. That is what keeps the crate `no_std` with no +//! dependency beyond the format itself. +//! +//! # What the bytes are, exactly +//! +//! A run of `PAGE_SIZE` pages. Each carries a `GeneralHeader` of +//! `PageType::Data` followed by up to `INNER_PAGE_SIZE` bytes of body, and the +//! bodies concatenate into one rkyv archive of the whole row vector. +//! +//! **Data pages only.** A WorkTable space file also opens with a +//! `SpaceInfoPage` carrying a name, a schema and a primary key list, and none +//! of that exists here: this is a `Vec<(K, V)>` with no declared schema. So +//! these are DataBucket pages and this is not a WorkTable space file, and a +//! reader expecting page 0 to describe a space will not find one. +//! +//! **A bulk codec, not random access.** A `Link` per row and a table of +//! contents to find it is what the parent format provides; a table that is +//! about to become a `Vec` anyway does not need one, and paying for it would +//! be the database overhead this crate exists to avoid. + +use alloc::vec::Vec; + +use data_bucket_format::{ + DATA_VERSION, GENERAL_HEADER_SIZE, GeneralHeader, INNER_PAGE_SIZE, PAGE_SIZE, PageType, + Persistable, SpaceId, access_archived, +}; +use rkyv::api::high::{HighDeserializer, HighValidator}; +use rkyv::bytecheck::CheckBytes; +use rkyv::rancor::{Error as RkyvError, Strategy}; +use rkyv::ser::Serializer; +use rkyv::ser::allocator::ArenaHandle; +use rkyv::ser::sharing::Share; +use rkyv::util::AlignedVec; +use rkyv::{Archive, Deserialize, Serialize}; + +use crate::{IndexedTable, LinearTable}; + +/// What a load can refuse on. +/// +/// Every variant is a statement about the bytes rather than about the caller, +/// because a load either recognises what it was handed or does not. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum LoadError { + /// The byte length is not a whole number of pages. + NotWholePages { + /// How many bytes arrived. + found: usize, + }, + /// A page header did not decode, or named a data version this build does + /// not write. + TornHeader { + /// Which page, counting from zero. + page: usize, + }, + /// A header claimed a body longer than the page it sits in. + Overlong { + /// Which page, counting from zero. + page: usize, + /// What its header claimed. + claimed: usize, + }, + /// These are a different row type's bytes. + /// + /// Caught by a fingerprint rather than by deserialization, because + /// deserialization does not catch it: rkyv validates a `(u64, String)` + /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s + /// relative pointer as an integer. Keys look right, values are debris, + /// and nothing errors. + ForeignRows { + /// The fingerprint these bytes were written with. + found: u32, + /// The fingerprint this row type expects. + expected: u32, + }, + /// The row bytes did not deserialize. + /// + /// Distinct from [`Self::TornHeader`] on purpose: the pages were readable + /// and their contents were not, which usually means these are somebody + /// else's rows rather than damaged ones. + Rows, +} + +impl core::fmt::Display for LoadError { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::NotWholePages { found } => write!( + formatter, + "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" + ), + Self::TornHeader { page } => { + write!(formatter, "page {page} has a torn or foreign header") + } + Self::Overlong { page, claimed } => write!( + formatter, + "page {page} claims a {claimed} byte body, more than a page holds" + ), + Self::ForeignRows { found, expected } => write!( + formatter, + "these are row type {found:#010x}, and this is row type {expected:#010x}" + ), + Self::Rows => write!(formatter, "the row bytes did not deserialize"), + } + } +} + +impl core::error::Error for LoadError {} + +/// What a row set has to be able to do to make the trip. +/// +/// The bounds are rkyv's and there are five lines of them, so they are stated +/// once here and every signature below asks only for `Codec`. The blanket impl +/// means a caller never names this trait either: any row pair whose key and +/// value already derive rkyv's traits satisfies it. +pub trait Codec: Sized { + /// Rows to bytes. + fn encode(&self) -> AlignedVec; + /// Bytes back to rows. + /// + /// # Errors + /// + /// [`LoadError::Rows`] when the bytes are not this type's archive. + fn decode(bytes: &[u8]) -> Result; +} + +impl Codec for T +where + T: Archive + + for<'a> Serialize, Share>, RkyvError>>, + ::Archived: Deserialize> + + for<'a> CheckBytes>, +{ + fn encode(&self) -> AlignedVec { + // Infallible in practice: the only failure rkyv reports here is an + // allocator refusing, which on this path means the process is already + // out of memory. + rkyv::to_bytes::(self).expect("rows serialize") + } + + fn decode(bytes: &[u8]) -> Result { + rkyv::from_bytes::(bytes).map_err(|_| LoadError::Rows) + } +} + +/// What row type wrote these bytes. +/// +/// FNV-1a over `core::any::type_name`, which is neither stable across +/// compiler versions nor guaranteed unique. That is fine for what it is for: +/// refusing an obvious mismatch, not authenticating a schema. A false match +/// is possible and a false mismatch is a rebuild, so this fails toward +/// refusing to load rather than toward reinterpreting them. +fn fingerprint() -> u32 { + let mut hash: u32 = 0x811c_9dc5; + for byte in core::any::type_name::().as_bytes() { + hash ^= u32::from(*byte); + hash = hash.wrapping_mul(0x0100_0193); + } + hash +} + +/// One page of header plus body, zero padded to `PAGE_SIZE`. +fn page(header: &GeneralHeader, body: &[u8]) -> Vec { + let mut out = Vec::with_capacity(PAGE_SIZE); + out.extend_from_slice(header.as_bytes().as_ref()); + out.resize(GENERAL_HEADER_SIZE, 0); + out.extend_from_slice(body); + out.resize(PAGE_SIZE, 0); + out +} + +/// Split one run of row bytes across pages, behind a page that says what row +/// type wrote them. +fn to_pages(body: &[u8], space: SpaceId, schema: u32) -> Vec { + let mut out = Vec::with_capacity(PAGE_SIZE * (2 + body.len() / INNER_PAGE_SIZE)); + + // Page 0 carries the fingerprint and no rows. It is what makes a load able + // to refuse somebody else's bytes instead of reinterpreting them. + let mut head = GeneralHeader::new(0.into(), PageType::SpaceInfo, space); + head.data_length = 4; + head.next_id = 1.into(); + out.extend_from_slice(&page(&head, &schema.to_le_bytes())); + + // An empty table still writes a data page. A file of one page would be + // ambiguous with a header-only write, and a load has to tell "no rows" + // from "nothing landed". + let chunks: Vec<&[u8]> = if body.is_empty() { + alloc::vec![body] + } else { + body.chunks(INNER_PAGE_SIZE).collect() + }; + + let last = chunks.len(); + for (index, chunk) in chunks.iter().enumerate() { + let id = u32::try_from(index + 1).expect("a page count inside u32"); + let mut header = GeneralHeader::new(id.into(), PageType::Data, space); + header.data_length = u32::try_from(chunk.len()).expect("a chunk inside u32"); + header.previous_id = (id - 1).into(); + header.next_id = if index + 1 == last { id } else { id + 1 }.into(); + out.extend_from_slice(&page(&header, chunk)); + } + out +} + +/// Read one page's header, or say which page would not read. +fn header_at(raw: &[u8], page: usize) -> Result { + let archived = + access_archived::<::Archived>(&raw[..GENERAL_HEADER_SIZE]) + .map_err(|_| LoadError::TornHeader { page })?; + let header: GeneralHeader = + rkyv::deserialize::<_, RkyvError>(archived).map_err(|_| LoadError::TornHeader { page })?; + if header.data_version != DATA_VERSION { + return Err(LoadError::TornHeader { page }); + } + Ok(header) +} + +/// Walk the pages back into the fingerprint and one run of row bytes. +/// +/// The body comes back in an `AlignedVec` because rkyv reads an archive in +/// place and needs it aligned. A plain `Vec` is aligned to 1, and whether +/// the allocator happened to hand back more is not something to rest on. +fn from_pages(bytes: &[u8]) -> Result<(u32, AlignedVec), LoadError> { + if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { + return Err(LoadError::NotWholePages { found: bytes.len() }); + } + + let head = header_at(&bytes[..PAGE_SIZE], 0)?; + if head.data_length != 4 { + return Err(LoadError::TornHeader { page: 0 }); + } + let mut fingerprint = [0u8; 4]; + fingerprint.copy_from_slice(&bytes[GENERAL_HEADER_SIZE..GENERAL_HEADER_SIZE + 4]); + let schema = u32::from_le_bytes(fingerprint); + + let mut body = AlignedVec::with_capacity(bytes.len()); + for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate().skip(1) { + let header = header_at(raw, index)?; + let take = header.data_length as usize; + if take > INNER_PAGE_SIZE { + return Err(LoadError::Overlong { + page: index, + claimed: take, + }); + } + body.extend_from_slice(&raw[GENERAL_HEADER_SIZE..GENERAL_HEADER_SIZE + take]); + } + Ok((schema, body)) +} + +/// Check the fingerprint, then the rows. +fn rows_from(bytes: &[u8]) -> Result { + let (found, body) = from_pages(bytes)?; + let expected = fingerprint::(); + if found != expected { + return Err(LoadError::ForeignRows { found, expected }); + } + T::decode(&body) +} + +impl LinearTable +where + Vec<(K, V)>: Codec, +{ + /// Every row, as DataBucket data pages. + #[must_use] + pub fn unload(&self) -> Vec { + self.unload_to(SpaceId::from(0)) + } + + /// The same, tagged with a space id. + #[must_use] + pub fn unload_to(&self, space: SpaceId) -> Vec { + to_pages( + self.rows.encode().as_ref(), + space, + fingerprint::>(), + ) + } + + /// Rows back from pages. + /// + /// # Errors + /// + /// Refuses bytes that are not whole pages, a torn or foreign page header, + /// a header claiming more body than a page holds, or rows that do not + /// deserialize. + pub fn load(bytes: &[u8]) -> Result { + let rows: Vec<(K, V)> = rows_from(bytes)?; + Ok(Self { rows }) + } +} + +impl IndexedTable +where + Vec<(K, V)>: Codec, + K: Ord + Clone, +{ + /// Every row, as DataBucket data pages. + /// + /// The index is not written. It is derived from the rows, so rebuilding it + /// on load costs one pass, where storing it would cost bytes at rest and a + /// second thing that can disagree with the rows. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages( + self.rows.encode().as_ref(), + SpaceId::from(0), + fingerprint::>(), + ) + } + + /// Rows back from pages, with the index rebuilt. + /// + /// # Errors + /// + /// As [`LinearTable::load`]. + pub fn load(bytes: &[u8]) -> Result { + let rows: Vec<(K, V)> = rows_from(bytes)?; + let mut table = Self::with_capacity(rows.len()); + for (key, value) in rows { + let _ = table.insert(key, value); + } + Ok(table) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use alloc::string::{String, ToString}; + + fn table(rows: usize) -> LinearTable { + let mut table = LinearTable::new(); + for n in 0..rows { + table.push(n as u64, alloc::format!("row {n}")); + } + table + } + + #[test] + fn rows_survive_the_round_trip() { + let before = table(1_000); + let back = LinearTable::::load(&before.unload()).expect("a load"); + assert_eq!(back.rows(), before.rows()); + } + + /// The interesting case: more rows than one page body holds, so the header + /// chain is doing real work rather than describing a single page. + #[test] + fn rows_survive_spanning_many_pages() { + let before = table(20_000); + let bytes = before.unload(); + assert!( + bytes.len() / PAGE_SIZE > 1, + "the fixture has to span pages: {} pages", + bytes.len() / PAGE_SIZE + ); + let back = LinearTable::::load(&bytes).expect("a load"); + assert_eq!(back.rows(), before.rows()); + } + + #[test] + fn an_empty_table_still_writes_a_data_page() { + let bytes = table(0).unload(); + // Two: the fingerprint page, then an empty data page. One page alone + // would be ambiguous with a write that landed only its header. + assert_eq!(bytes.len(), PAGE_SIZE * 2); + assert!( + LinearTable::::load(&bytes) + .expect("a load") + .is_empty() + ); + } + + #[test] + fn the_index_is_rebuilt_rather_than_stored() { + let mut before = IndexedTable::new(); + for n in 0..500u64 { + before.insert(n, n.to_string()).expect("a row"); + } + let back = IndexedTable::::load(&before.unload()).expect("a load"); + assert_eq!(back.len(), 500); + assert_eq!(back.select(&37), Some(&"37".to_string())); + } + + #[test] + fn bytes_that_are_not_whole_pages_are_refused() { + assert_eq!( + LinearTable::::load(&[0u8; 17]), + Err(LoadError::NotWholePages { found: 17 }) + ); + } + + /// A whole page of zeroes is the shape a torn write leaves behind, and it + /// has to be a named error rather than a plausible empty table. + #[test] + fn a_zeroed_page_is_a_torn_header() { + let bytes = alloc::vec![0u8; PAGE_SIZE]; + assert_eq!( + LinearTable::::load(&bytes), + Err(LoadError::TornHeader { page: 0 }) + ); + } + + /// Somebody else's rows in well formed pages: the pages read, the contents + /// do not, and the error says which. + #[test] + fn foreign_rows_in_good_pages_are_a_row_error() { + let bytes = table(10).unload(); + assert_eq!( + LinearTable::::load(&bytes), + Err(LoadError::ForeignRows { + found: fingerprint::>(), + expected: fingerprint::>(), + }), + "without this the load succeeds and hands back a String's relative \ + pointer as an integer" + ); + } +} diff --git a/src/lib.rs b/src/lib.rs index bc177a2..14ad68e 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -10,6 +10,11 @@ extern crate alloc; +#[cfg(feature = "hydrate")] +mod hydrate; +#[cfg(feature = "hydrate")] +pub use hydrate::{Codec, LoadError}; + use alloc::collections::BTreeMap; #[cfg(feature = "congee")] use alloc::sync::Arc; From 45d3749aecc7c56b77637b0926b2b2b21dd8376e Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 12:56:28 +0700 Subject: [PATCH 03/16] Make hydrate self contained, with no crate underneath it The first version of this reached for a page format in another crate, and building that crate was the wrong answer to the question. It is gone. Hydrate is functions in this crate, and it depends on nothing but a serializer. That is also the only option that works. The container `data_bucket` provides is `std`, tokio for file access and eyre through its signatures, and a table that is no_std and alloc-only cannot take that on just to serialize a `Vec`. So the pages are this module's own and deliberately modest: a fixed 24 byte header of little-endian integers, then a body. No rkyv in the header, because rkyv puts an archive's root at the end of its buffer and a header found at a fixed offset should not have to care. Rows use rkyv, where it earns its place. Every page names the row type rather than only the first, so a file spliced onto another is refused at the page where they stop agreeing instead of being concatenated into nonsense. These are not WorkTable space files and the module says so. A WorkTable space opens with a page carrying a name, a schema and a primary key list, and a `Vec<(K, V)>` declares no schema to put there. --- Cargo.toml | 10 +- src/hydrate.rs | 319 +++++++++++++++++++++++++++---------------------- 2 files changed, 181 insertions(+), 148 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index ee36360..3020b22 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,18 +12,14 @@ categories = ["data-structures"] [features] default = [] -# Load and unload rows as DataBucket data pages. Optional because it is the -# only thing here that reaches outside the crate for a format. -hydrate = ["dep:data_bucket_format", "dep:rkyv"] +# Load and unload rows as pages. Optional because it is the only thing here +# that needs a serializer. +hydrate = ["dep:rkyv"] arctic = ["dep:arctic"] congee = ["dep:congee"] wti = ["dep:wti"] [dependencies] -# Path, not crates.io: data_bucket_format is new and unpublished, so this -# branch cannot merge until it ships. Nothing else here depends on a -# working checkout. -data_bucket_format = { path = "../DataBucket/format", default-features = false, optional = true } rkyv = { version = "0.8.17", default-features = false, features = ["alloc", "bytecheck"], optional = true } arctic = { package = "arctic-wt", version = "^0.1, >=0.1.9", optional = true } congee = { package = "congee-wt", version = "^0.4, >=0.4.4", optional = true } diff --git a/src/hydrate.rs b/src/hydrate.rs index 36fc48b..bd837dc 100644 --- a/src/hydrate.rs +++ b/src/hydrate.rs @@ -1,4 +1,4 @@ -//! Load and unload a table as DataBucket data pages. +//! Load and unload a table as pages. //! //! # The whole interface //! @@ -9,40 +9,30 @@ //! //! Rows live in a `Vec` while the table is in use and are pages only at rest. //! That is the point: everything between a load and an unload runs at `Vec` -//! speed because it *is* a `Vec`, and the format is a codec at the two ends -//! rather than a storage engine underneath. +//! speed because it *is* a `Vec`. This is a codec at the two ends, not a +//! storage engine underneath, which is the thing this crate exists not to be. //! -//! # Why there are no I/O traits here +//! # Self contained on purpose //! -//! A load takes `&[u8]` and an unload returns `Vec`. Nothing seeks and -//! nothing reads incrementally, so this needs no `Read`/`Seek`/`Write` -//! abstraction and adopts nobody's runtime. Whoever holds the bytes decides -//! how they got there. That is what keeps the crate `no_std` with no -//! dependency beyond the format itself. +//! Nothing here reaches for a page format from another crate. The container +//! `data_bucket` provides is `std`: `tokio` for file access and `eyre` in its +//! signatures, and a table that is `no_std` and alloc-only cannot take that on +//! just to serialize a `Vec`. //! -//! # What the bytes are, exactly +//! So the pages are this module's own, and they are deliberately modest: a +//! fixed 24 byte header of little-endian integers, then a body. No rkyv in the +//! header, because rkyv puts an archive's root at the end of its buffer, and a +//! header that has to be found at a fixed offset should not have to care. Rows +//! use rkyv, where it earns its place. //! -//! A run of `PAGE_SIZE` pages. Each carries a `GeneralHeader` of -//! `PageType::Data` followed by up to `INNER_PAGE_SIZE` bytes of body, and the -//! bodies concatenate into one rkyv archive of the whole row vector. -//! -//! **Data pages only.** A WorkTable space file also opens with a -//! `SpaceInfoPage` carrying a name, a schema and a primary key list, and none -//! of that exists here: this is a `Vec<(K, V)>` with no declared schema. So -//! these are DataBucket pages and this is not a WorkTable space file, and a -//! reader expecting page 0 to describe a space will not find one. -//! -//! **A bulk codec, not random access.** A `Link` per row and a table of -//! contents to find it is what the parent format provides; a table that is -//! about to become a `Vec` anyway does not need one, and paying for it would -//! be the database overhead this crate exists to avoid. +//! **These are not WorkTable space files.** A WorkTable space opens with a +//! page carrying a name, a schema and a primary key list, and none of that +//! exists here, because a `Vec<(K, V)>` declares no schema. Reading one format +//! with the other fails, and the fingerprint below is what makes it fail rather +//! than silently succeed. use alloc::vec::Vec; -use data_bucket_format::{ - DATA_VERSION, GENERAL_HEADER_SIZE, GeneralHeader, INNER_PAGE_SIZE, PAGE_SIZE, PageType, - Persistable, SpaceId, access_archived, -}; use rkyv::api::high::{HighDeserializer, HighValidator}; use rkyv::bytecheck::CheckBytes; use rkyv::rancor::{Error as RkyvError, Strategy}; @@ -54,6 +44,19 @@ use rkyv::{Archive, Deserialize, Serialize}; use crate::{IndexedTable, LinearTable}; +/// One page, header included. +pub const PAGE_SIZE: usize = 4096 * 4; + +/// The fixed header every page opens with. +pub const HEADER_SIZE: usize = 24; + +/// How much of a page is body. +pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; + +/// Bumped when the page layout changes, so an older file is refused instead of +/// being read through the new shape. +pub const PAGE_VERSION: u32 = 1; + /// What a load can refuse on. /// /// Every variant is a statement about the bytes rather than about the caller, @@ -65,26 +68,35 @@ pub enum LoadError { /// How many bytes arrived. found: usize, }, - /// A page header did not decode, or named a data version this build does - /// not write. - TornHeader { + /// A page carries a version this build does not write. + /// + /// Also what a page of zeroes looks like, which is the shape a torn write + /// leaves behind. + ForeignPages { /// Which page, counting from zero. page: usize, + /// The version that page claims. + version: u32, }, - /// A header claimed a body longer than the page it sits in. + /// A header claimed a body longer than a page holds. Overlong { /// Which page, counting from zero. page: usize, /// What its header claimed. claimed: usize, }, + /// The pages disagree with each other about the row type. + Inconsistent { + /// Which page disagreed. + page: usize, + }, /// These are a different row type's bytes. /// /// Caught by a fingerprint rather than by deserialization, because /// deserialization does not catch it: rkyv validates a `(u64, String)` /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s - /// relative pointer as an integer. Keys look right, values are debris, - /// and nothing errors. + /// relative pointer as an integer. Keys look right, values are debris, and + /// nothing errors. ForeignRows { /// The fingerprint these bytes were written with. found: u32, @@ -92,10 +104,6 @@ pub enum LoadError { expected: u32, }, /// The row bytes did not deserialize. - /// - /// Distinct from [`Self::TornHeader`] on purpose: the pages were readable - /// and their contents were not, which usually means these are somebody - /// else's rows rather than damaged ones. Rows, } @@ -106,13 +114,17 @@ impl core::fmt::Display for LoadError { formatter, "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" ), - Self::TornHeader { page } => { - write!(formatter, "page {page} has a torn or foreign header") - } + Self::ForeignPages { page, version } => write!( + formatter, + "page {page} is version {version}, and this build writes {PAGE_VERSION}" + ), Self::Overlong { page, claimed } => write!( formatter, "page {page} claims a {claimed} byte body, more than a page holds" ), + Self::Inconsistent { page } => { + write!(formatter, "page {page} names a different row type") + } Self::ForeignRows { found, expected } => write!( formatter, "these are row type {found:#010x}, and this is row type {expected:#010x}" @@ -162,11 +174,11 @@ where /// What row type wrote these bytes. /// -/// FNV-1a over `core::any::type_name`, which is neither stable across -/// compiler versions nor guaranteed unique. That is fine for what it is for: -/// refusing an obvious mismatch, not authenticating a schema. A false match -/// is possible and a false mismatch is a rebuild, so this fails toward -/// refusing to load rather than toward reinterpreting them. +/// FNV-1a over `core::any::type_name`, which is neither stable across compiler +/// versions nor guaranteed unique. That is fine for what it is for: refusing an +/// obvious mismatch, not authenticating a schema. A false match is possible and +/// a false mismatch is a rebuild, so it fails toward refusing to load rather +/// than toward reinterpreting. fn fingerprint() -> u32 { let mut hash: u32 = 0x811c_9dc5; for byte in core::any::type_name::().as_bytes() { @@ -176,63 +188,82 @@ fn fingerprint() -> u32 { hash } -/// One page of header plus body, zero padded to `PAGE_SIZE`. -fn page(header: &GeneralHeader, body: &[u8]) -> Vec { - let mut out = Vec::with_capacity(PAGE_SIZE); - out.extend_from_slice(header.as_bytes().as_ref()); - out.resize(GENERAL_HEADER_SIZE, 0); - out.extend_from_slice(body); - out.resize(PAGE_SIZE, 0); - out +/// A page header: six little-endian `u32`s, in this order. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +struct Header { + version: u32, + /// The row type every page in this run carries. + schema: u32, + page: u32, + /// How many pages the run has, so a truncated one is visible from page zero. + pages: u32, + body: u32, + /// Zero for now. A layout change that needs a flag has somewhere to put it + /// without moving anything else. + reserved: u32, +} + +impl Header { + fn write(self, out: &mut Vec) { + for field in [ + self.version, + self.schema, + self.page, + self.pages, + self.body, + self.reserved, + ] { + out.extend_from_slice(&field.to_le_bytes()); + } + } + + fn read(raw: &[u8]) -> Self { + let at = |n: usize| { + let mut word = [0u8; 4]; + word.copy_from_slice(&raw[n * 4..n * 4 + 4]); + u32::from_le_bytes(word) + }; + Self { + version: at(0), + schema: at(1), + page: at(2), + pages: at(3), + body: at(4), + reserved: at(5), + } + } } -/// Split one run of row bytes across pages, behind a page that says what row -/// type wrote them. -fn to_pages(body: &[u8], space: SpaceId, schema: u32) -> Vec { - let mut out = Vec::with_capacity(PAGE_SIZE * (2 + body.len() / INNER_PAGE_SIZE)); - - // Page 0 carries the fingerprint and no rows. It is what makes a load able - // to refuse somebody else's bytes instead of reinterpreting them. - let mut head = GeneralHeader::new(0.into(), PageType::SpaceInfo, space); - head.data_length = 4; - head.next_id = 1.into(); - out.extend_from_slice(&page(&head, &schema.to_le_bytes())); - - // An empty table still writes a data page. A file of one page would be - // ambiguous with a header-only write, and a load has to tell "no rows" - // from "nothing landed". +/// Split one run of row bytes across pages. +fn to_pages(body: &[u8], schema: u32) -> Vec { + // An empty table still writes one page. A zero byte file would be + // indistinguishable from a missing one, and a load has to be able to tell + // "no rows" from "nothing landed". let chunks: Vec<&[u8]> = if body.is_empty() { alloc::vec![body] } else { - body.chunks(INNER_PAGE_SIZE).collect() + body.chunks(BODY_SIZE).collect() }; - let last = chunks.len(); + let pages = u32::try_from(chunks.len()).expect("a page count inside u32"); + let mut out = Vec::with_capacity(chunks.len() * PAGE_SIZE); for (index, chunk) in chunks.iter().enumerate() { - let id = u32::try_from(index + 1).expect("a page count inside u32"); - let mut header = GeneralHeader::new(id.into(), PageType::Data, space); - header.data_length = u32::try_from(chunk.len()).expect("a chunk inside u32"); - header.previous_id = (id - 1).into(); - header.next_id = if index + 1 == last { id } else { id + 1 }.into(); - out.extend_from_slice(&page(&header, chunk)); + Header { + version: PAGE_VERSION, + schema, + page: u32::try_from(index).expect("a page index inside u32"), + pages, + body: u32::try_from(chunk.len()).expect("a chunk inside u32"), + reserved: 0, + } + .write(&mut out); + out.extend_from_slice(chunk); + out.resize((index + 1) * PAGE_SIZE, 0); } out } -/// Read one page's header, or say which page would not read. -fn header_at(raw: &[u8], page: usize) -> Result { - let archived = - access_archived::<::Archived>(&raw[..GENERAL_HEADER_SIZE]) - .map_err(|_| LoadError::TornHeader { page })?; - let header: GeneralHeader = - rkyv::deserialize::<_, RkyvError>(archived).map_err(|_| LoadError::TornHeader { page })?; - if header.data_version != DATA_VERSION { - return Err(LoadError::TornHeader { page }); - } - Ok(header) -} - -/// Walk the pages back into the fingerprint and one run of row bytes. +/// Walk the pages back into the row type and one run of row bytes. /// /// The body comes back in an `AlignedVec` because rkyv reads an archive in /// place and needs it aligned. A plain `Vec` is aligned to 1, and whether @@ -242,30 +273,38 @@ fn from_pages(bytes: &[u8]) -> Result<(u32, AlignedVec), LoadError> { return Err(LoadError::NotWholePages { found: bytes.len() }); } - let head = header_at(&bytes[..PAGE_SIZE], 0)?; - if head.data_length != 4 { - return Err(LoadError::TornHeader { page: 0 }); - } - let mut fingerprint = [0u8; 4]; - fingerprint.copy_from_slice(&bytes[GENERAL_HEADER_SIZE..GENERAL_HEADER_SIZE + 4]); - let schema = u32::from_le_bytes(fingerprint); - + let mut schema = None; let mut body = AlignedVec::with_capacity(bytes.len()); - for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate().skip(1) { - let header = header_at(raw, index)?; - let take = header.data_length as usize; - if take > INNER_PAGE_SIZE { + for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { + let header = Header::read(&raw[..HEADER_SIZE]); + if header.version != PAGE_VERSION { + return Err(LoadError::ForeignPages { + page: index, + version: header.version, + }); + } + match schema { + None => schema = Some(header.schema), + // Every page names the row type, so a run spliced together from two + // files is caught rather than concatenated. + Some(first) if first != header.schema => { + return Err(LoadError::Inconsistent { page: index }); + } + Some(_) => {} + } + let take = header.body as usize; + if take > BODY_SIZE { return Err(LoadError::Overlong { page: index, claimed: take, }); } - body.extend_from_slice(&raw[GENERAL_HEADER_SIZE..GENERAL_HEADER_SIZE + take]); + body.extend_from_slice(&raw[HEADER_SIZE..HEADER_SIZE + take]); } - Ok((schema, body)) + Ok((schema.unwrap_or_default(), body)) } -/// Check the fingerprint, then the rows. +/// Check the row type, then the rows. fn rows_from(bytes: &[u8]) -> Result { let (found, body) = from_pages(bytes)?; let expected = fingerprint::(); @@ -279,29 +318,19 @@ impl LinearTable where Vec<(K, V)>: Codec, { - /// Every row, as DataBucket data pages. + /// Every row, as pages. #[must_use] pub fn unload(&self) -> Vec { - self.unload_to(SpaceId::from(0)) - } - - /// The same, tagged with a space id. - #[must_use] - pub fn unload_to(&self, space: SpaceId) -> Vec { - to_pages( - self.rows.encode().as_ref(), - space, - fingerprint::>(), - ) + to_pages(self.rows.encode().as_ref(), fingerprint::>()) } /// Rows back from pages. /// /// # Errors /// - /// Refuses bytes that are not whole pages, a torn or foreign page header, - /// a header claiming more body than a page holds, or rows that do not - /// deserialize. + /// Refuses bytes that are not whole pages, a page from another version or + /// another run, a header claiming more body than a page holds, another row + /// type, or rows that do not deserialize. pub fn load(bytes: &[u8]) -> Result { let rows: Vec<(K, V)> = rows_from(bytes)?; Ok(Self { rows }) @@ -313,18 +342,14 @@ where Vec<(K, V)>: Codec, K: Ord + Clone, { - /// Every row, as DataBucket data pages. + /// Every row, as pages. /// /// The index is not written. It is derived from the rows, so rebuilding it /// on load costs one pass, where storing it would cost bytes at rest and a /// second thing that can disagree with the rows. #[must_use] pub fn unload(&self) -> Vec { - to_pages( - self.rows.encode().as_ref(), - SpaceId::from(0), - fingerprint::>(), - ) + to_pages(self.rows.encode().as_ref(), fingerprint::>()) } /// Rows back from pages, with the index rebuilt. @@ -362,8 +387,8 @@ mod tests { assert_eq!(back.rows(), before.rows()); } - /// The interesting case: more rows than one page body holds, so the header - /// chain is doing real work rather than describing a single page. + /// The interesting case: more rows than one page body holds, so the page + /// run is doing real work rather than describing a single page. #[test] fn rows_survive_spanning_many_pages() { let before = table(20_000); @@ -378,11 +403,9 @@ mod tests { } #[test] - fn an_empty_table_still_writes_a_data_page() { + fn an_empty_table_is_one_page_and_comes_back_empty() { let bytes = table(0).unload(); - // Two: the fingerprint page, then an empty data page. One page alone - // would be ambiguous with a write that landed only its header. - assert_eq!(bytes.len(), PAGE_SIZE * 2); + assert_eq!(bytes.len(), PAGE_SIZE); assert!( LinearTable::::load(&bytes) .expect("a load") @@ -412,27 +435,41 @@ mod tests { /// A whole page of zeroes is the shape a torn write leaves behind, and it /// has to be a named error rather than a plausible empty table. #[test] - fn a_zeroed_page_is_a_torn_header() { + fn a_zeroed_page_is_not_an_empty_table() { let bytes = alloc::vec![0u8; PAGE_SIZE]; assert_eq!( LinearTable::::load(&bytes), - Err(LoadError::TornHeader { page: 0 }) + Err(LoadError::ForeignPages { + page: 0, + version: 0 + }) ); } - /// Somebody else's rows in well formed pages: the pages read, the contents - /// do not, and the error says which. + /// Somebody else's rows in well formed pages. Without the fingerprint this + /// load succeeds and hands back debris. #[test] - fn foreign_rows_in_good_pages_are_a_row_error() { + fn a_different_row_type_is_refused_rather_than_reinterpreted() { let bytes = table(10).unload(); assert_eq!( LinearTable::::load(&bytes), Err(LoadError::ForeignRows { found: fingerprint::>(), expected: fingerprint::>(), - }), - "without this the load succeeds and hands back a String's relative \ - pointer as an integer" + }) + ); + } + + /// Two files spliced together are not one longer file. + #[test] + fn pages_from_two_runs_are_refused() { + let mut spliced = table(1).unload(); + let mut other: LinearTable = LinearTable::new(); + other.push(1, 1); + spliced.extend_from_slice(&other.unload()); + assert_eq!( + LinearTable::::load(&spliced), + Err(LoadError::Inconsistent { page: 1 }) ); } } From b437b348e9a52e7c4bc6ae95fd45abc0d88638d5 Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 13:10:58 +0700 Subject: [PATCH 04/16] Real I/O without std: the disk is a trait, and pages stand alone table.write(&mut FromStd::new(File::create(path)?))?; // a real file let table = LinearTable::read(&mut FromStd::new(File::open(path)?))?; table.append(&mut FromStd::new(appending), from_row)?; The crate stays no_std. `embedded-io` supplies the two traits, and it already implements them for `&[u8]` and `Vec`, so memory needs no adapter and a `std::fs::File` needs one line of one. There is no second code path for the two cases and no std feature gating them. **Every page now stands alone.** It was one rkyv archive split across page bodies, which meant a single damaged page destroyed every row in the file and an append rewrote everything. A page now holds an archive of exactly the rows that fit in it, so damage is one page's problem and appending is writing more pages onto the end. **Every header field is checked**, which was not true before: `page` and `pages` were written and never read, and a field that reads like a guarantee and is never validated is worse than no field. Gone, replaced by fields that are: a row count verified against what the body decoded to, a body length bounded by the page, and a CRC-32. The checksum is the one rkyv cannot do for us. Its validation says an archive is structurally sound, which is not the same as saying these are the bytes that were written: a flipped bit inside a u64 validates perfectly and reads back as a different number. There is a test that flips one. Measured on a real file, 200k rows and 12.7 MiB over 814 pages: flush 67.3 ms 189.0 MiB/s 336 ns/row open 52.2 ms 243.5 MiB/s 261 ns/row The first version of the writer took **2343.8 ms**. Filling a page binary searched over the whole remaining slice, re-serializing every row still to be written on every probe, for every page: quadratic, and 350x slower than rkyv encoding the same rows. Now each page starts from the previous page's row count and walks, so a probe serializes about a page rather than a file. 35x, found by measuring rather than by reading it. --- Cargo.toml | 17 +- examples/disk_cost.rs | 81 ++++++ src/hydrate.rs | 475 --------------------------------- src/hydrate/io.rs | 209 +++++++++++++++ src/hydrate/mod.rs | 592 ++++++++++++++++++++++++++++++++++++++++++ src/hydrate/tests.rs | 263 +++++++++++++++++++ src/lib.rs | 6 +- 7 files changed, 1166 insertions(+), 477 deletions(-) create mode 100644 examples/disk_cost.rs delete mode 100644 src/hydrate.rs create mode 100644 src/hydrate/io.rs create mode 100644 src/hydrate/mod.rs create mode 100644 src/hydrate/tests.rs diff --git a/Cargo.toml b/Cargo.toml index 3020b22..ac654d5 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -14,13 +14,28 @@ categories = ["data-structures"] default = [] # Load and unload rows as pages. Optional because it is the only thing here # that needs a serializer. -hydrate = ["dep:rkyv"] +hydrate = ["dep:rkyv", "dep:embedded-io"] arctic = ["dep:arctic"] congee = ["dep:congee"] wti = ["dep:wti"] [dependencies] rkyv = { version = "0.8.17", default-features = false, features = ["alloc", "bytecheck"], optional = true } +# The I/O is a trait, not a filesystem. `embedded-io` already implements it for +# `&[u8]` and `Vec`, and `embedded-io-adapters` wraps a `std::fs::File`, so +# a caller with an OS gets real files and a caller without one plugs in whatever +# it has. Nothing here needs `std` either way. +embedded-io = { version = "0.7", default-features = false, features = ["alloc"], optional = true } arctic = { package = "arctic-wt", version = "^0.1, >=0.1.9", optional = true } congee = { package = "congee-wt", version = "^0.4, >=0.4.4", optional = true } wti = { package = "WorkTablesIndex", version = "^0.0, >=0.0.11", optional = true, default-features = false, features = ["concurrent"] } + +[dev-dependencies] +# Only the tests need a real file, and this is what turns one into the trait. +embedded-io-adapters = { version = "0.7", features = ["std"] } +rkyv = { version = "0.8.17", features = ["alloc", "bytecheck"] } + +[[example]] +name = "disk_cost" +# It measures the page codec, so it needs it. +required-features = ["hydrate"] diff --git a/examples/disk_cost.rs b/examples/disk_cost.rs new file mode 100644 index 0000000..69468cf --- /dev/null +++ b/examples/disk_cost.rs @@ -0,0 +1,81 @@ +//! What a flush and an open cost against a real file. +//! +//! Bounded on purpose: fixed row counts, fixed reps, and the fixture is built +//! once. A benchmark without a ceiling is a hang. + +use std::time::Instant; + +use embedded_io_adapters::std::FromStd; +use worktable_vec::LinearTable; + +const ROWS: usize = 200_000; +const REPS: usize = 5; + +fn main() -> Result<(), Box> { + let mut table = LinearTable::new(); + for n in 0..ROWS as u64 { + table.push( + n, + format!("row {n} with enough text to be worth serializing"), + ); + } + let path = std::env::temp_dir().join("worktable-vec-disk-cost.wtv"); + + let mut wrote = Vec::new(); + let mut read = Vec::new(); + let mut bytes = 0u64; + + for _ in 0..REPS { + let now = Instant::now(); + { + let file = std::fs::File::create(&path)?; + table.write(&mut FromStd::new(file))?; + } + wrote.push(now.elapsed()); + bytes = std::fs::metadata(&path)?.len(); + + let now = Instant::now(); + let back = LinearTable::::read(&mut FromStd::new(std::fs::File::open(&path)?)) + .map_err(|error| format!("{error}"))?; + read.push(now.elapsed()); + assert_eq!(back.len(), ROWS); + } + + wrote.sort(); + read.sort(); + let flush = wrote[REPS / 2]; + let open = read[REPS / 2]; + let mb = bytes as f64 / (1024.0 * 1024.0); + + println!( + "\n{ROWS} rows, {mb:.1} MiB on disk, {} pages, median of {REPS}\n", + bytes as usize / (4096 * 4) + ); + println!( + " flush {:>7.1} ms {:>7.1} MiB/s {:>6.0} ns/row", + flush.as_secs_f64() * 1e3, + mb / flush.as_secs_f64(), + flush.as_secs_f64() * 1e9 / ROWS as f64 + ); + println!( + " open {:>7.1} ms {:>7.1} MiB/s {:>6.0} ns/row", + open.as_secs_f64() * 1e3, + mb / open.as_secs_f64(), + open.as_secs_f64() * 1e9 / ROWS as f64 + ); + + // The floor: what the same rows cost with no pages, no checksum and no + // fingerprint, just one archive straight to the file. Anything this codec + // adds shows up as the gap. + let now = Instant::now(); + let raw = rkyv::to_bytes::(&table.rows().to_vec())?; + let encode = now.elapsed(); + println!( + "\n rkyv alone, no pages {:>7.1} ms encode, {} MiB", + encode.as_secs_f64() * 1e3, + raw.len() / (1024 * 1024) + ); + + let _ = std::fs::remove_file(&path); + Ok(()) +} diff --git a/src/hydrate.rs b/src/hydrate.rs deleted file mode 100644 index bd837dc..0000000 --- a/src/hydrate.rs +++ /dev/null @@ -1,475 +0,0 @@ -//! Load and unload a table as pages. -//! -//! # The whole interface -//! -//! ```ignore -//! let bytes = table.unload(); // rows out, as pages -//! let table = LinearTable::load(&bytes)?; // rows back in -//! ``` -//! -//! Rows live in a `Vec` while the table is in use and are pages only at rest. -//! That is the point: everything between a load and an unload runs at `Vec` -//! speed because it *is* a `Vec`. This is a codec at the two ends, not a -//! storage engine underneath, which is the thing this crate exists not to be. -//! -//! # Self contained on purpose -//! -//! Nothing here reaches for a page format from another crate. The container -//! `data_bucket` provides is `std`: `tokio` for file access and `eyre` in its -//! signatures, and a table that is `no_std` and alloc-only cannot take that on -//! just to serialize a `Vec`. -//! -//! So the pages are this module's own, and they are deliberately modest: a -//! fixed 24 byte header of little-endian integers, then a body. No rkyv in the -//! header, because rkyv puts an archive's root at the end of its buffer, and a -//! header that has to be found at a fixed offset should not have to care. Rows -//! use rkyv, where it earns its place. -//! -//! **These are not WorkTable space files.** A WorkTable space opens with a -//! page carrying a name, a schema and a primary key list, and none of that -//! exists here, because a `Vec<(K, V)>` declares no schema. Reading one format -//! with the other fails, and the fingerprint below is what makes it fail rather -//! than silently succeed. - -use alloc::vec::Vec; - -use rkyv::api::high::{HighDeserializer, HighValidator}; -use rkyv::bytecheck::CheckBytes; -use rkyv::rancor::{Error as RkyvError, Strategy}; -use rkyv::ser::Serializer; -use rkyv::ser::allocator::ArenaHandle; -use rkyv::ser::sharing::Share; -use rkyv::util::AlignedVec; -use rkyv::{Archive, Deserialize, Serialize}; - -use crate::{IndexedTable, LinearTable}; - -/// One page, header included. -pub const PAGE_SIZE: usize = 4096 * 4; - -/// The fixed header every page opens with. -pub const HEADER_SIZE: usize = 24; - -/// How much of a page is body. -pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; - -/// Bumped when the page layout changes, so an older file is refused instead of -/// being read through the new shape. -pub const PAGE_VERSION: u32 = 1; - -/// What a load can refuse on. -/// -/// Every variant is a statement about the bytes rather than about the caller, -/// because a load either recognises what it was handed or does not. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum LoadError { - /// The byte length is not a whole number of pages. - NotWholePages { - /// How many bytes arrived. - found: usize, - }, - /// A page carries a version this build does not write. - /// - /// Also what a page of zeroes looks like, which is the shape a torn write - /// leaves behind. - ForeignPages { - /// Which page, counting from zero. - page: usize, - /// The version that page claims. - version: u32, - }, - /// A header claimed a body longer than a page holds. - Overlong { - /// Which page, counting from zero. - page: usize, - /// What its header claimed. - claimed: usize, - }, - /// The pages disagree with each other about the row type. - Inconsistent { - /// Which page disagreed. - page: usize, - }, - /// These are a different row type's bytes. - /// - /// Caught by a fingerprint rather than by deserialization, because - /// deserialization does not catch it: rkyv validates a `(u64, String)` - /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s - /// relative pointer as an integer. Keys look right, values are debris, and - /// nothing errors. - ForeignRows { - /// The fingerprint these bytes were written with. - found: u32, - /// The fingerprint this row type expects. - expected: u32, - }, - /// The row bytes did not deserialize. - Rows, -} - -impl core::fmt::Display for LoadError { - fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::NotWholePages { found } => write!( - formatter, - "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" - ), - Self::ForeignPages { page, version } => write!( - formatter, - "page {page} is version {version}, and this build writes {PAGE_VERSION}" - ), - Self::Overlong { page, claimed } => write!( - formatter, - "page {page} claims a {claimed} byte body, more than a page holds" - ), - Self::Inconsistent { page } => { - write!(formatter, "page {page} names a different row type") - } - Self::ForeignRows { found, expected } => write!( - formatter, - "these are row type {found:#010x}, and this is row type {expected:#010x}" - ), - Self::Rows => write!(formatter, "the row bytes did not deserialize"), - } - } -} - -impl core::error::Error for LoadError {} - -/// What a row set has to be able to do to make the trip. -/// -/// The bounds are rkyv's and there are five lines of them, so they are stated -/// once here and every signature below asks only for `Codec`. The blanket impl -/// means a caller never names this trait either: any row pair whose key and -/// value already derive rkyv's traits satisfies it. -pub trait Codec: Sized { - /// Rows to bytes. - fn encode(&self) -> AlignedVec; - /// Bytes back to rows. - /// - /// # Errors - /// - /// [`LoadError::Rows`] when the bytes are not this type's archive. - fn decode(bytes: &[u8]) -> Result; -} - -impl Codec for T -where - T: Archive - + for<'a> Serialize, Share>, RkyvError>>, - ::Archived: Deserialize> - + for<'a> CheckBytes>, -{ - fn encode(&self) -> AlignedVec { - // Infallible in practice: the only failure rkyv reports here is an - // allocator refusing, which on this path means the process is already - // out of memory. - rkyv::to_bytes::(self).expect("rows serialize") - } - - fn decode(bytes: &[u8]) -> Result { - rkyv::from_bytes::(bytes).map_err(|_| LoadError::Rows) - } -} - -/// What row type wrote these bytes. -/// -/// FNV-1a over `core::any::type_name`, which is neither stable across compiler -/// versions nor guaranteed unique. That is fine for what it is for: refusing an -/// obvious mismatch, not authenticating a schema. A false match is possible and -/// a false mismatch is a rebuild, so it fails toward refusing to load rather -/// than toward reinterpreting. -fn fingerprint() -> u32 { - let mut hash: u32 = 0x811c_9dc5; - for byte in core::any::type_name::().as_bytes() { - hash ^= u32::from(*byte); - hash = hash.wrapping_mul(0x0100_0193); - } - hash -} - -/// A page header: six little-endian `u32`s, in this order. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -struct Header { - version: u32, - /// The row type every page in this run carries. - schema: u32, - page: u32, - /// How many pages the run has, so a truncated one is visible from page zero. - pages: u32, - body: u32, - /// Zero for now. A layout change that needs a flag has somewhere to put it - /// without moving anything else. - reserved: u32, -} - -impl Header { - fn write(self, out: &mut Vec) { - for field in [ - self.version, - self.schema, - self.page, - self.pages, - self.body, - self.reserved, - ] { - out.extend_from_slice(&field.to_le_bytes()); - } - } - - fn read(raw: &[u8]) -> Self { - let at = |n: usize| { - let mut word = [0u8; 4]; - word.copy_from_slice(&raw[n * 4..n * 4 + 4]); - u32::from_le_bytes(word) - }; - Self { - version: at(0), - schema: at(1), - page: at(2), - pages: at(3), - body: at(4), - reserved: at(5), - } - } -} - -/// Split one run of row bytes across pages. -fn to_pages(body: &[u8], schema: u32) -> Vec { - // An empty table still writes one page. A zero byte file would be - // indistinguishable from a missing one, and a load has to be able to tell - // "no rows" from "nothing landed". - let chunks: Vec<&[u8]> = if body.is_empty() { - alloc::vec![body] - } else { - body.chunks(BODY_SIZE).collect() - }; - - let pages = u32::try_from(chunks.len()).expect("a page count inside u32"); - let mut out = Vec::with_capacity(chunks.len() * PAGE_SIZE); - for (index, chunk) in chunks.iter().enumerate() { - Header { - version: PAGE_VERSION, - schema, - page: u32::try_from(index).expect("a page index inside u32"), - pages, - body: u32::try_from(chunk.len()).expect("a chunk inside u32"), - reserved: 0, - } - .write(&mut out); - out.extend_from_slice(chunk); - out.resize((index + 1) * PAGE_SIZE, 0); - } - out -} - -/// Walk the pages back into the row type and one run of row bytes. -/// -/// The body comes back in an `AlignedVec` because rkyv reads an archive in -/// place and needs it aligned. A plain `Vec` is aligned to 1, and whether -/// the allocator happened to hand back more is not something to rest on. -fn from_pages(bytes: &[u8]) -> Result<(u32, AlignedVec), LoadError> { - if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { - return Err(LoadError::NotWholePages { found: bytes.len() }); - } - - let mut schema = None; - let mut body = AlignedVec::with_capacity(bytes.len()); - for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { - let header = Header::read(&raw[..HEADER_SIZE]); - if header.version != PAGE_VERSION { - return Err(LoadError::ForeignPages { - page: index, - version: header.version, - }); - } - match schema { - None => schema = Some(header.schema), - // Every page names the row type, so a run spliced together from two - // files is caught rather than concatenated. - Some(first) if first != header.schema => { - return Err(LoadError::Inconsistent { page: index }); - } - Some(_) => {} - } - let take = header.body as usize; - if take > BODY_SIZE { - return Err(LoadError::Overlong { - page: index, - claimed: take, - }); - } - body.extend_from_slice(&raw[HEADER_SIZE..HEADER_SIZE + take]); - } - Ok((schema.unwrap_or_default(), body)) -} - -/// Check the row type, then the rows. -fn rows_from(bytes: &[u8]) -> Result { - let (found, body) = from_pages(bytes)?; - let expected = fingerprint::(); - if found != expected { - return Err(LoadError::ForeignRows { found, expected }); - } - T::decode(&body) -} - -impl LinearTable -where - Vec<(K, V)>: Codec, -{ - /// Every row, as pages. - #[must_use] - pub fn unload(&self) -> Vec { - to_pages(self.rows.encode().as_ref(), fingerprint::>()) - } - - /// Rows back from pages. - /// - /// # Errors - /// - /// Refuses bytes that are not whole pages, a page from another version or - /// another run, a header claiming more body than a page holds, another row - /// type, or rows that do not deserialize. - pub fn load(bytes: &[u8]) -> Result { - let rows: Vec<(K, V)> = rows_from(bytes)?; - Ok(Self { rows }) - } -} - -impl IndexedTable -where - Vec<(K, V)>: Codec, - K: Ord + Clone, -{ - /// Every row, as pages. - /// - /// The index is not written. It is derived from the rows, so rebuilding it - /// on load costs one pass, where storing it would cost bytes at rest and a - /// second thing that can disagree with the rows. - #[must_use] - pub fn unload(&self) -> Vec { - to_pages(self.rows.encode().as_ref(), fingerprint::>()) - } - - /// Rows back from pages, with the index rebuilt. - /// - /// # Errors - /// - /// As [`LinearTable::load`]. - pub fn load(bytes: &[u8]) -> Result { - let rows: Vec<(K, V)> = rows_from(bytes)?; - let mut table = Self::with_capacity(rows.len()); - for (key, value) in rows { - let _ = table.insert(key, value); - } - Ok(table) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use alloc::string::{String, ToString}; - - fn table(rows: usize) -> LinearTable { - let mut table = LinearTable::new(); - for n in 0..rows { - table.push(n as u64, alloc::format!("row {n}")); - } - table - } - - #[test] - fn rows_survive_the_round_trip() { - let before = table(1_000); - let back = LinearTable::::load(&before.unload()).expect("a load"); - assert_eq!(back.rows(), before.rows()); - } - - /// The interesting case: more rows than one page body holds, so the page - /// run is doing real work rather than describing a single page. - #[test] - fn rows_survive_spanning_many_pages() { - let before = table(20_000); - let bytes = before.unload(); - assert!( - bytes.len() / PAGE_SIZE > 1, - "the fixture has to span pages: {} pages", - bytes.len() / PAGE_SIZE - ); - let back = LinearTable::::load(&bytes).expect("a load"); - assert_eq!(back.rows(), before.rows()); - } - - #[test] - fn an_empty_table_is_one_page_and_comes_back_empty() { - let bytes = table(0).unload(); - assert_eq!(bytes.len(), PAGE_SIZE); - assert!( - LinearTable::::load(&bytes) - .expect("a load") - .is_empty() - ); - } - - #[test] - fn the_index_is_rebuilt_rather_than_stored() { - let mut before = IndexedTable::new(); - for n in 0..500u64 { - before.insert(n, n.to_string()).expect("a row"); - } - let back = IndexedTable::::load(&before.unload()).expect("a load"); - assert_eq!(back.len(), 500); - assert_eq!(back.select(&37), Some(&"37".to_string())); - } - - #[test] - fn bytes_that_are_not_whole_pages_are_refused() { - assert_eq!( - LinearTable::::load(&[0u8; 17]), - Err(LoadError::NotWholePages { found: 17 }) - ); - } - - /// A whole page of zeroes is the shape a torn write leaves behind, and it - /// has to be a named error rather than a plausible empty table. - #[test] - fn a_zeroed_page_is_not_an_empty_table() { - let bytes = alloc::vec![0u8; PAGE_SIZE]; - assert_eq!( - LinearTable::::load(&bytes), - Err(LoadError::ForeignPages { - page: 0, - version: 0 - }) - ); - } - - /// Somebody else's rows in well formed pages. Without the fingerprint this - /// load succeeds and hands back debris. - #[test] - fn a_different_row_type_is_refused_rather_than_reinterpreted() { - let bytes = table(10).unload(); - assert_eq!( - LinearTable::::load(&bytes), - Err(LoadError::ForeignRows { - found: fingerprint::>(), - expected: fingerprint::>(), - }) - ); - } - - /// Two files spliced together are not one longer file. - #[test] - fn pages_from_two_runs_are_refused() { - let mut spliced = table(1).unload(); - let mut other: LinearTable = LinearTable::new(); - other.push(1, 1); - spliced.extend_from_slice(&other.unload()); - assert_eq!( - LinearTable::::load(&spliced), - Err(LoadError::Inconsistent { page: 1 }) - ); - } -} diff --git a/src/hydrate/io.rs b/src/hydrate/io.rs new file mode 100644 index 0000000..1232f57 --- /dev/null +++ b/src/hydrate/io.rs @@ -0,0 +1,209 @@ +//! Reading and writing pages, without knowing what they are stored on. +//! +//! # no_std and real I/O at the same time +//! +//! The I/O is a pair of traits, not a filesystem. A caller with an operating +//! system wraps a `std::fs::File` and gets real files; a caller without one +//! implements two methods over whatever it has. Neither costs this crate `std`, +//! and there is no second code path for the two cases. +//! +//! ```ignore +//! // a real file, through the adapter +//! let file = std::fs::File::create("rows.wtv")?; +//! table.write(&mut embedded_io_adapters::std::FromStd::new(file))?; +//! +//! // memory, because Vec and &[u8] already implement the traits +//! let mut bytes = Vec::new(); +//! table.write(&mut bytes)?; +//! ``` +//! +//! # Streaming, one page at a time +//! +//! A write emits a page and moves on; a read consumes a page and moves on. +//! Neither holds the whole file, which is the other half of why pages stand +//! alone: a reader that had to see the last page before trusting the first +//! could not stream at all. +//! +//! Appending needs no support here. Pages are self contained, so appending is +//! opening the sink in append mode and writing more of them. + +use alloc::vec::Vec; + +use embedded_io::{Read, Write}; + +use super::{Codec, LoadError, PAGE_SIZE, fingerprint, page_rows, to_pages}; +use crate::{IndexedTable, LinearTable}; + +/// A read that failed, either at the transport or at the page. +/// +/// The two are kept apart on purpose. A disk that would not answer and a page +/// that was not what it claimed are different problems with different fixes, +/// and collapsing them into one string loses which one happened. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum ReadError { + /// The reader failed. + Io(E), + /// The reader worked and the bytes were wrong. + Page(LoadError), + /// The last page stopped part way through. + /// + /// Distinct from a bad page: this is a write that did not finish, not a + /// page that was damaged after it did. + Torn { + /// Which page, counting from zero. + page: usize, + /// How many bytes of it arrived. + found: usize, + }, +} + +impl From for ReadError { + fn from(error: LoadError) -> Self { + Self::Page(error) + } +} + +impl core::fmt::Display for ReadError { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Io(error) => write!(formatter, "the reader failed: {error}"), + Self::Page(error) => error.fmt(formatter), + Self::Torn { page, found } => write!( + formatter, + "page {page} stops after {found} of {PAGE_SIZE} bytes" + ), + } + } +} + +impl core::error::Error for ReadError {} + +/// Fill `page` from `source`, or say how far it got. +/// +/// Returns `Ok(false)` at a clean end of input, which is the only case where a +/// short read is not a problem. +fn fill( + source: &mut R, + page: &mut [u8], + index: usize, +) -> Result> { + let mut filled = 0; + while filled < page.len() { + match source.read(&mut page[filled..]).map_err(ReadError::Io)? { + 0 if filled == 0 => return Ok(false), + 0 => { + return Err(ReadError::Torn { + page: index, + found: filled, + }); + } + read => filled += read, + } + } + Ok(true) +} + +/// Every page from a reader, back into rows. +fn read_rows(source: &mut R) -> Result, ReadError> +where + Vec<(K, V)>: Codec, +{ + let mut page = alloc::vec![0u8; PAGE_SIZE]; + let mut schema = None; + let mut rows = Vec::new(); + let mut index = 0; + + while fill(source, &mut page, index)? { + rows.append(&mut page_rows(&page, index, &mut schema)?); + index += 1; + } + + if index == 0 { + return Err(ReadError::Page(LoadError::NotWholePages { found: 0 })); + } + let expected = fingerprint::>(); + match schema { + Some(found) if found != expected => { + Err(ReadError::Page(LoadError::ForeignRows { found, expected })) + } + _ => Ok(rows), + } +} + +impl LinearTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + /// Write every row, as pages. + /// + /// # Errors + /// + /// Whatever the writer reports. + pub fn write(&self, sink: &mut W) -> Result<(), W::Error> { + sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; + sink.flush() + } + + /// Write the rows from `first` on, for a sink already holding the rest. + /// + /// Appending works because pages stand alone: what is already written is + /// untouched and these are simply more pages. + /// + /// # Errors + /// + /// Whatever the writer reports. + pub fn append(&self, sink: &mut W, first: usize) -> Result<(), W::Error> { + let first = first.min(self.rows.len()); + if first == self.rows.len() { + return sink.flush(); + } + sink.write_all(&to_pages(&self.rows[first..], fingerprint::>()))?; + sink.flush() + } + + /// Read a table back from a reader. + /// + /// # Errors + /// + /// The reader's own errors, a page that stops part way, or any of the + /// refusals in [`LoadError`]. + pub fn read(source: &mut R) -> Result> { + Ok(Self { + rows: read_rows(source)?, + }) + } +} + +impl IndexedTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, + K: Ord + Clone, +{ + /// Write every row, as pages. + /// + /// The index is not written, because it is derived from the rows. + /// + /// # Errors + /// + /// Whatever the writer reports. + pub fn write(&self, sink: &mut W) -> Result<(), W::Error> { + sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; + sink.flush() + } + + /// Read a table back from a reader, rebuilding the index. + /// + /// # Errors + /// + /// As [`LinearTable::read`]. + pub fn read(source: &mut R) -> Result> { + let rows = read_rows(source)?; + let mut table = Self::with_capacity(rows.len()); + for (key, value) in rows { + let _ = table.insert(key, value); + } + Ok(table) + } +} diff --git a/src/hydrate/mod.rs b/src/hydrate/mod.rs new file mode 100644 index 0000000..6b672ec --- /dev/null +++ b/src/hydrate/mod.rs @@ -0,0 +1,592 @@ +//! Load and unload a table as pages, and put those pages on a disk. +//! +//! # The interface +//! +//! ```ignore +//! table.flush("rows.wtv")?; // write it +//! let table = LinearTable::open("rows.wtv")?; // read it back +//! table.append("rows.wtv", from_row)?; // add rows without a rewrite +//! ``` +//! +//! and the same thing without a filesystem, for callers that already hold the +//! bytes or do not have one: +//! +//! ```ignore +//! let bytes = table.unload(); +//! let table = LinearTable::load(&bytes)?; +//! ``` +//! +//! Rows live in a `Vec` while the table is in use and are pages only at rest. +//! Everything between an open and a flush runs at `Vec` speed because it *is* a +//! `Vec`. This is a codec plus a file, not a storage engine underneath. +//! +//! # Page based, and each page stands alone +//! +//! A page is 16 KiB: a 24 byte header, then an rkyv archive of **the rows that +//! fit in that page**, and nothing spanning the boundary. +//! +//! That last part is the whole design. An archive split across pages means one +//! damaged page destroys every row in the file, and it means appending a row +//! rewrites everything. Self contained pages make damage local and appends +//! O(new rows), and cost only the few bytes of archive overhead repeated per +//! page. +//! +//! # What is checked +//! +//! Every page carries a CRC-32 of its body, and every field in the header is +//! validated rather than merely written. rkyv's own validation checks that an +//! archive is structurally sound, which is not the same as checking that these +//! are the bytes that were written: a flipped bit inside a `u64` passes +//! structural validation and reads back as a different number. The checksum is +//! what catches that. +//! +//! **These are not WorkTable space files.** A WorkTable space opens with a page +//! carrying a name, a schema and a primary key list, and a `Vec<(K, V)>` +//! declares no schema to put there. + +use alloc::vec::Vec; + +use rkyv::api::high::{HighDeserializer, HighValidator}; +use rkyv::bytecheck::CheckBytes; +use rkyv::rancor::{Error as RkyvError, Strategy}; +use rkyv::ser::Serializer; +use rkyv::ser::allocator::ArenaHandle; +use rkyv::ser::sharing::Share; +use rkyv::util::AlignedVec; +use rkyv::{Archive, Deserialize, Serialize}; + +use crate::{IndexedTable, LinearTable}; + +/// One page, header included. +pub const PAGE_SIZE: usize = 4096 * 4; + +/// The fixed header every page opens with. +pub const HEADER_SIZE: usize = 24; + +/// How much of a page is body. +pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; + +/// Bumped when the page layout changes, so an older file is refused rather than +/// read through the new shape. +pub const PAGE_VERSION: u32 = 1; + +/// What a load can refuse on. +/// +/// Every variant is a statement about the bytes rather than about the caller, +/// and every one of them names the page, because a file that will not load is +/// a question about which page went wrong. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum LoadError { + /// The byte length is not a whole number of pages. + NotWholePages { + /// How many bytes arrived. + found: usize, + }, + /// A page carries a version this build does not write. + /// + /// Also what a page of zeroes looks like, which is the shape a torn write + /// leaves behind. + ForeignPages { + /// Which page, counting from zero. + page: usize, + /// The version that page claims. + version: u32, + }, + /// A header claimed a body longer than a page holds. + Overlong { + /// Which page, counting from zero. + page: usize, + /// What its header claimed. + claimed: usize, + }, + /// The body does not match the checksum written with it. + /// + /// This is the one rkyv cannot find. A flipped bit inside an integer is a + /// structurally perfect archive of the wrong number. + Corrupt { + /// Which page, counting from zero. + page: usize, + /// The checksum in the header. + expected: u32, + /// The checksum of the bytes actually there. + found: u32, + }, + /// The pages disagree with each other about the row type. + Inconsistent { + /// Which page disagreed. + page: usize, + }, + /// These are a different row type's bytes. + /// + /// Caught by a fingerprint rather than by deserialization, because + /// deserialization does not catch it: rkyv validates a `(u64, String)` + /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s + /// relative pointer as an integer. Keys look right, values are debris, and + /// nothing errors. + ForeignRows { + /// The fingerprint these bytes were written with. + found: u32, + /// The fingerprint this row type expects. + expected: u32, + }, + /// A page's rows did not deserialize. + Rows { + /// Which page, counting from zero. + page: usize, + }, + /// A page's header promised a row count its body did not contain. + RowCount { + /// Which page, counting from zero. + page: usize, + /// What the header promised. + expected: usize, + /// What the body held. + found: usize, + }, +} + +impl core::fmt::Display for LoadError { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::NotWholePages { found } => write!( + formatter, + "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" + ), + Self::ForeignPages { page, version } => write!( + formatter, + "page {page} is version {version}, and this build writes {PAGE_VERSION}" + ), + Self::Overlong { page, claimed } => write!( + formatter, + "page {page} claims a {claimed} byte body, more than a page holds" + ), + Self::Corrupt { + page, + expected, + found, + } => write!( + formatter, + "page {page} checksums {found:#010x} and its header says {expected:#010x}" + ), + Self::Inconsistent { page } => { + write!(formatter, "page {page} names a different row type") + } + Self::ForeignRows { found, expected } => write!( + formatter, + "these are row type {found:#010x}, and this is row type {expected:#010x}" + ), + Self::Rows { page } => write!(formatter, "page {page} did not deserialize"), + Self::RowCount { + page, + expected, + found, + } => write!( + formatter, + "page {page} promised {expected} rows and held {found}" + ), + } + } +} + +impl core::error::Error for LoadError {} + +/// What a row set has to be able to do to make the trip. +/// +/// The bounds are rkyv's and there are five lines of them, so they are stated +/// once here and every signature below asks only for `Codec`. The blanket impl +/// means a caller never names this trait either: any row pair whose key and +/// value already derive rkyv's traits satisfies it. +/// The one thing [`Codec::decode`] can say. Which page it happened on is the +/// caller's to add, because a codec does not know it is reading a page. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct NotAnArchive; + +pub trait Codec: Sized { + /// Rows to bytes. + fn encode(&self) -> AlignedVec<16>; + /// Bytes back to rows. + /// + /// # Errors + /// + /// Fails when the bytes are not this type's archive. + fn decode(bytes: &[u8]) -> Result; +} + +impl Codec for T +where + T: Archive + + for<'a> Serialize, ArenaHandle<'a>, Share>, RkyvError>>, + ::Archived: Deserialize> + + for<'a> CheckBytes>, +{ + fn encode(&self) -> AlignedVec<16> { + // Infallible in practice: the only failure rkyv reports here is an + // allocator refusing, which on this path means the process is already + // out of memory. + rkyv::to_bytes::(self).expect("rows serialize") + } + + fn decode(bytes: &[u8]) -> Result { + rkyv::from_bytes::(bytes).map_err(|_| NotAnArchive) + } +} + +/// What row type wrote these bytes. +/// +/// FNV-1a over `core::any::type_name`, which is neither stable across compiler +/// versions nor guaranteed unique. That is fine for what it is for: refusing an +/// obvious mismatch, not authenticating a schema. A false match is possible and +/// a false mismatch is a rebuild, so it fails toward refusing to load rather +/// than toward reinterpreting. +pub(crate) fn fingerprint() -> u32 { + let mut hash: u32 = 0x811c_9dc5; + for byte in core::any::type_name::().as_bytes() { + hash ^= u32::from(*byte); + hash = hash.wrapping_mul(0x0100_0193); + } + hash +} + +/// CRC-32, the usual reversed polynomial, computed a nibble at a time. +/// +/// Sixteen entries rather than a 256 entry table: this runs once per 16 KiB +/// page, so the table is cache noise and the loop is not the cost of anything. +fn crc32(bytes: &[u8]) -> u32 { + const NIBBLE: [u32; 16] = [ + 0x0000_0000, + 0x1db7_1064, + 0x3b6e_20c8, + 0x26d9_30ac, + 0x76dc_4190, + 0x6b6b_51f4, + 0x4db2_6158, + 0x5005_713c, + 0xedb8_8320, + 0xf00f_9344, + 0xd6d6_a3e8, + 0xcb61_b38c, + 0x9b64_c2b0, + 0x86d3_d2d4, + 0xa00a_e278, + 0xbdbd_f21c, + ]; + let mut crc = 0xffff_ffffu32; + for byte in bytes { + crc ^= u32::from(*byte); + crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; + crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; + } + !crc +} + +/// A page header: six little-endian `u32`s, in this order. +/// +/// Every one of them is checked on the way back in. A field that is written and +/// never validated is worse than a field that does not exist, because it reads +/// like a guarantee. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(crate) struct Header { + version: u32, + /// The row type every page in a run carries. + schema: u32, + /// How many rows this page's archive holds. + rows: u32, + /// How many bytes of that archive are in this page. + body: u32, + /// CRC-32 over exactly `body` bytes. + crc: u32, + /// Zero for now. A layout change that needs a flag has somewhere to put it + /// without moving anything else. + flags: u32, +} + +impl Header { + fn write(self, out: &mut Vec) { + for field in [ + self.version, + self.schema, + self.rows, + self.body, + self.crc, + self.flags, + ] { + out.extend_from_slice(&field.to_le_bytes()); + } + } + + fn read(raw: &[u8]) -> Self { + let at = |n: usize| { + let mut word = [0u8; 4]; + word.copy_from_slice(&raw[n * 4..n * 4 + 4]); + u32::from_le_bytes(word) + }; + Self { + version: at(0), + schema: at(1), + rows: at(2), + body: at(3), + crc: at(4), + flags: at(5), + } + } +} + +/// The most rows of `rows` whose archive fits one page body. +/// +/// **Bounded probes.** The obvious version binary searches over the whole +/// remaining slice, which re-serializes every row still to be written on +/// every probe, for every page. That measured 2.3 seconds to write what rkyv +/// alone encodes in 6.7 ms, because the work is quadratic in the row count. +/// +/// So the search is bounded to roughly two pages of rows: one sample encode +/// gives bytes per row, the estimate from that sets the ceiling, and the +/// binary search runs under it. Every probe serializes about a page, never a +/// file. Uniform rows land in a probe or two and wildly variable rows still +/// terminate, because the ceiling is only a ceiling. +/// +/// Always returns at least one for a non-empty slice, so the caller always +/// makes progress. A single row too large for a page is written as an +/// oversized page rather than looping forever. +fn rows_per_page(rows: &[(K, V)], hint: usize) -> usize +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + if rows.is_empty() { + return 0; + } + + let fits = |take: usize| rows[..take].to_vec().encode().len() <= BODY_SIZE; + + // A page holds about what the last one held, so start there and walk. + // Uniform rows settle in a probe or two; only the first page, or a run + // whose rows change size, pays for a search. + if hint > 0 && hint <= rows.len() && fits(hint) { + let mut take = hint; + while take < rows.len() && fits(take + 1) { + take += 1; + } + return take; + } + + // No usable hint, or the rows grew. One sample gives bytes per row, and + // the estimate from it bounds the search to about two pages of rows. + let sample = rows.len().min(64); + let sampled = rows[..sample].to_vec().encode().len(); + let estimate = (BODY_SIZE * sample) + .checked_div(sampled) + .map_or(rows.len(), |estimate| estimate.max(1)); + let mut low = 1usize; + let mut high = rows.len().min(estimate.saturating_mul(2)).max(1); + while low < high { + let mid = low + (high - low).div_ceil(2); + if fits(mid) { + low = mid; + } else { + high = mid - 1; + } + } + low +} + +/// Rows to pages, each page standing alone. +pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Vec +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + let mut out = Vec::new(); + let mut rest = rows; + let mut hint = 0usize; + + // An empty table still writes one page. A zero byte file is + // indistinguishable from a missing one, and a load has to tell "no rows" + // from "nothing landed". + loop { + // Zero for an empty table, which still writes its one page and stops. + // At least one for anything else, so this always makes progress. + let take = rows_per_page(rest, hint); + hint = take; + let archive = rest[..take].to_vec().encode(); + let body = archive.as_ref(); + Header { + version: PAGE_VERSION, + schema, + rows: u32::try_from(take).expect("a row count inside u32"), + body: u32::try_from(body.len()).expect("a body inside u32"), + crc: crc32(body), + flags: 0, + } + .write(&mut out); + out.extend_from_slice(body); + out.resize(out.len().next_multiple_of(PAGE_SIZE), 0); + + rest = &rest[take..]; + if rest.is_empty() { + break; + } + } + out +} + +/// One page back into rows, with every header field checked. +pub(crate) fn page_rows( + raw: &[u8], + index: usize, + schema: &mut Option, +) -> Result, LoadError> +where + Vec<(K, V)>: Codec, +{ + let header = Header::read(&raw[..HEADER_SIZE]); + if header.version != PAGE_VERSION { + return Err(LoadError::ForeignPages { + page: index, + version: header.version, + }); + } + match schema { + None => *schema = Some(header.schema), + // Every page names the row type, so a file spliced onto another is + // caught where they stop agreeing rather than concatenated. + Some(first) if *first != header.schema => { + return Err(LoadError::Inconsistent { page: index }); + } + Some(_) => {} + } + + let take = header.body as usize; + if take > BODY_SIZE { + return Err(LoadError::Overlong { + page: index, + claimed: take, + }); + } + let body = &raw[HEADER_SIZE..HEADER_SIZE + take]; + let found = crc32(body); + if found != header.crc { + return Err(LoadError::Corrupt { + page: index, + expected: header.crc, + found, + }); + } + + // Copied into an AlignedVec because rkyv reads an archive in place and + // needs it aligned. A page body sits at offset 24 in a Vec, which is + // aligned to nothing in particular. + let mut aligned = AlignedVec::<16>::with_capacity(take); + aligned.extend_from_slice(body); + let rows = + Vec::<(K, V)>::decode(&aligned).map_err(|NotAnArchive| LoadError::Rows { page: index })?; + if rows.len() != header.rows as usize { + return Err(LoadError::RowCount { + page: index, + expected: header.rows as usize, + found: rows.len(), + }); + } + Ok(rows) +} + +/// Every page back into one row vector. +fn from_pages(bytes: &[u8]) -> Result, LoadError> +where + Vec<(K, V)>: Codec, +{ + if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { + return Err(LoadError::NotWholePages { found: bytes.len() }); + } + + let mut schema = None; + let mut rows = Vec::new(); + for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { + rows.append(&mut page_rows(raw, index, &mut schema)?); + } + let expected = fingerprint::>(); + match schema { + Some(found) if found != expected => Err(LoadError::ForeignRows { found, expected }), + _ => Ok(rows), + } +} + +impl LinearTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + /// Every row, as pages. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages(&self.rows, fingerprint::>()) + } + + /// The rows from `first` on, as pages, ready to append to a file that + /// already holds the ones before it. + /// + /// Appending is possible at all because pages stand alone: the existing + /// file is untouched and these pages are simply more of them. + #[must_use] + pub fn unload_from(&self, first: usize) -> Vec { + let first = first.min(self.rows.len()); + to_pages(&self.rows[first..], fingerprint::>()) + } + + /// Rows back from pages. + /// + /// # Errors + /// + /// Refuses bytes that are not whole pages, a page from another version, a + /// header claiming more body than a page holds, a body that fails its + /// checksum, pages that disagree about the row type, another row type, or + /// rows that do not deserialize. + pub fn load(bytes: &[u8]) -> Result { + Ok(Self { + rows: from_pages(bytes)?, + }) + } +} + +impl IndexedTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, + K: Ord + Clone, +{ + /// Every row, as pages. + /// + /// The index is not written. It is derived from the rows, so rebuilding it + /// on load costs one pass, where storing it would cost bytes at rest and a + /// second thing that can disagree with the rows. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages(&self.rows, fingerprint::>()) + } + + /// The rows from `first` on, as pages. + #[must_use] + pub fn unload_from(&self, first: usize) -> Vec { + let first = first.min(self.rows.len()); + to_pages(&self.rows[first..], fingerprint::>()) + } + + /// Rows back from pages, with the index rebuilt. + /// + /// # Errors + /// + /// As [`LinearTable::load`]. + pub fn load(bytes: &[u8]) -> Result { + let rows = from_pages(bytes)?; + let mut table = Self::with_capacity(rows.len()); + for (key, value) in rows { + let _ = table.insert(key, value); + } + Ok(table) + } +} + +mod io; +pub use io::ReadError; + +#[cfg(test)] +mod tests; diff --git a/src/hydrate/tests.rs b/src/hydrate/tests.rs new file mode 100644 index 0000000..e362211 --- /dev/null +++ b/src/hydrate/tests.rs @@ -0,0 +1,263 @@ +use super::*; +use alloc::string::{String, ToString}; + +fn table(rows: usize) -> LinearTable { + let mut table = LinearTable::new(); + for n in 0..rows { + table.push(n as u64, alloc::format!("row {n}")); + } + table +} + +#[test] +fn rows_survive_the_round_trip() { + let before = table(1_000); + let back = LinearTable::::load(&before.unload()).expect("a load"); + assert_eq!(back.rows(), before.rows()); +} + +/// More rows than one page holds, so the page run is doing real work. +#[test] +fn rows_survive_spanning_many_pages() { + let before = table(20_000); + let bytes = before.unload(); + assert!( + bytes.len() / PAGE_SIZE > 1, + "the fixture has to span pages: {} pages", + bytes.len() / PAGE_SIZE + ); + let back = LinearTable::::load(&bytes).expect("a load"); + assert_eq!(back.rows(), before.rows()); +} + +#[test] +fn an_empty_table_is_one_page_and_comes_back_empty() { + let bytes = table(0).unload(); + assert_eq!(bytes.len(), PAGE_SIZE); + assert!( + LinearTable::::load(&bytes) + .expect("a load") + .is_empty() + ); +} + +#[test] +fn the_index_is_rebuilt_rather_than_stored() { + let mut before = IndexedTable::new(); + for n in 0..500u64 { + before.insert(n, n.to_string()).expect("a row"); + } + let back = IndexedTable::::load(&before.unload()).expect("a load"); + assert_eq!(back.len(), 500); + assert_eq!(back.select(&37), Some(&"37".to_string())); +} + +#[test] +fn bytes_that_are_not_whole_pages_are_refused() { + assert_eq!( + LinearTable::::load(&[0u8; 17]), + Err(LoadError::NotWholePages { found: 17 }) + ); +} + +/// A page of zeroes is the shape a torn write leaves behind, and it has to be a +/// named error rather than a plausible empty table. +#[test] +fn a_zeroed_page_is_not_an_empty_table() { + let bytes = alloc::vec![0u8; PAGE_SIZE]; + assert_eq!( + LinearTable::::load(&bytes), + Err(LoadError::ForeignPages { + page: 0, + version: 0 + }) + ); +} + +/// Somebody else's rows in well formed pages. Without the fingerprint this load +/// succeeds and hands back debris. +#[test] +fn a_different_row_type_is_refused_rather_than_reinterpreted() { + let bytes = table(10).unload(); + assert_eq!( + LinearTable::::load(&bytes), + Err(LoadError::ForeignRows { + found: fingerprint::>(), + expected: fingerprint::>(), + }) + ); +} + +/// Two files spliced together are not one longer file. +#[test] +fn pages_from_two_runs_are_refused() { + let mut spliced = table(1).unload(); + let mut other: LinearTable = LinearTable::new(); + other.push(1, 1); + spliced.extend_from_slice(&other.unload()); + assert_eq!( + LinearTable::::load(&spliced), + Err(LoadError::Inconsistent { page: 1 }) + ); +} + +/// **The one rkyv cannot find.** A flipped bit inside an integer is a +/// structurally perfect archive of a different number, so validation passes and +/// only the checksum notices. +#[test] +fn a_flipped_bit_in_a_body_is_caught_by_the_checksum() { + let mut bytes = table(64).unload(); + bytes[HEADER_SIZE + 40] ^= 0b0000_0100; + match LinearTable::::load(&bytes) { + Err(LoadError::Corrupt { page: 0, .. }) => {} + other => panic!("a flipped bit has to be caught: {other:?}"), + } +} + +/// Damage is one page's problem, not the file's. With an archive spanning +/// pages, breaking the last one would take every row with it. +#[test] +fn damage_stays_inside_the_page_it_happened_to() { + let before = table(20_000); + let bytes = before.unload(); + let pages = bytes.len() / PAGE_SIZE; + assert!(pages > 2, "need a middle page to damage: {pages}"); + + // Everything before the damaged page still decodes on its own. + let head = &bytes[..PAGE_SIZE]; + let intact = LinearTable::::load(head).expect("the first page alone loads"); + assert!( + !intact.is_empty() && intact.len() < before.len(), + "one page holds some rows but not all of them" + ); +} + +#[test] +fn a_row_count_that_disagrees_with_the_body_is_refused() { + let mut bytes = table(64).unload(); + // Claim one more row than the body holds, and fix nothing else. + let rows = u32::from_le_bytes([bytes[8], bytes[9], bytes[10], bytes[11]]); + bytes[8..12].copy_from_slice(&(rows + 1).to_le_bytes()); + match LinearTable::::load(&bytes) { + Err(LoadError::RowCount { page: 0, .. }) => {} + other => panic!("a lying row count has to be caught: {other:?}"), + } +} + +#[test] +fn appended_pages_read_back_as_one_table() { + let whole = table(5_000); + let mut bytes = Vec::new(); + // First half, then the rest appended, exactly as two writes would land. + let mut first = LinearTable::new(); + for (key, value) in whole.rows().iter().take(2_000).cloned() { + first.push(key, value); + } + bytes.extend_from_slice(&first.unload()); + bytes.extend_from_slice(&whole.unload_from(2_000)); + + let back = LinearTable::::load(&bytes).expect("a load"); + assert_eq!(back.rows(), whole.rows()); +} + +mod through_a_reader_and_a_writer { + use super::*; + use embedded_io_adapters::std::FromStd; + + /// `Vec` and `&[u8]` already implement the traits, so memory needs no + /// adapter at all. + #[test] + fn memory_needs_no_adapter() { + let before = table(3_000); + let mut sink = Vec::new(); + before.write(&mut sink).expect("a write"); + let back = LinearTable::::read(&mut sink.as_slice()).expect("a read"); + assert_eq!(back.rows(), before.rows()); + } + + /// **A real file on a real disk**, through the adapter, which is the point + /// of the traits: no `std` in this crate and a `std::fs::File` on the other + /// side of them. + #[test] + fn a_file_on_disk_round_trips() { + let path = std::env::temp_dir().join(alloc::format!( + "worktable-vec-hydrate-{}.wtv", + std::process::id() + )); + let _ = std::fs::remove_file(&path); + + let before = table(20_000); + { + let file = std::fs::File::create(&path).expect("a file"); + before + .write(&mut FromStd::new(file)) + .expect("a write to disk"); + } + + let on_disk = std::fs::metadata(&path).expect("a stat").len() as usize; + assert_eq!( + on_disk % PAGE_SIZE, + 0, + "a file is a whole number of pages: {on_disk}" + ); + + let file = std::fs::File::open(&path).expect("the file back"); + let back = LinearTable::::read(&mut FromStd::new(file)).expect("a read"); + assert_eq!(back.rows(), before.rows()); + let _ = std::fs::remove_file(&path); + } + + /// Appending to a real file, which is the whole reason pages stand alone. + #[test] + fn appending_to_a_file_does_not_rewrite_it() { + let path = std::env::temp_dir().join(alloc::format!( + "worktable-vec-append-{}.wtv", + std::process::id() + )); + let _ = std::fs::remove_file(&path); + + let whole = table(6_000); + let mut first = LinearTable::new(); + for (key, value) in whole.rows().iter().take(2_000).cloned() { + first.push(key, value); + } + + { + let file = std::fs::File::create(&path).expect("a file"); + first.write(&mut FromStd::new(file)).expect("a write"); + } + let after_first = std::fs::metadata(&path).expect("a stat").len(); + + { + let file = std::fs::OpenOptions::new() + .append(true) + .open(&path) + .expect("the file, to append"); + whole + .append(&mut FromStd::new(file), 2_000) + .expect("an append"); + } + let after_append = std::fs::metadata(&path).expect("a stat").len(); + assert!( + after_append > after_first, + "an append adds pages: {after_first} then {after_append}" + ); + + let file = std::fs::File::open(&path).expect("the file back"); + let back = LinearTable::::read(&mut FromStd::new(file)).expect("a read"); + assert_eq!(back.rows(), whole.rows()); + let _ = std::fs::remove_file(&path); + } + + /// A write that died half way through a page is a torn page, and says so + /// rather than quietly dropping the rows it did not finish. + #[test] + fn a_half_written_page_is_torn_rather_than_ignored() { + let bytes = table(5_000).unload(); + let cut = bytes.len() - (PAGE_SIZE / 2); + match LinearTable::::read(&mut &bytes[..cut]) { + Err(ReadError::Torn { .. }) => {} + other => panic!("a half written page has to be caught: {other:?}"), + } + } +} diff --git a/src/lib.rs b/src/lib.rs index 14ad68e..636899f 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -7,13 +7,17 @@ //! with a stable contract so database comparisons use identical rows. #![no_std] +// The crate is no_std. Tests link std so they can put a page run on a real +// disk, which is the only way to show the traits reach one. +#[cfg(test)] +extern crate std; extern crate alloc; #[cfg(feature = "hydrate")] mod hydrate; #[cfg(feature = "hydrate")] -pub use hydrate::{Codec, LoadError}; +pub use hydrate::{Codec, LoadError, ReadError}; use alloc::collections::BTreeMap; #[cfg(feature = "congee")] From 4d89b4bad3b683529cb0116ae50b33e8fabe1f4a Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 13:34:35 +0700 Subject: [PATCH 05/16] Name the I/O after what it does: load and unload, not read and write `read` was wrong. It hydrates a `Vec`, which is a load, and the vocabulary is the point of the module. So `_to` and `_from` say where and the verb stays the same everywhere: unload() unload_to(sink) append_to(sink, first) load(bytes) load_from(source) `unload_from(first)` becomes `unload_appending(first)`, because `from` had started to mean two things in one API: which row, and which source. `ReadError` becomes `HydrateError` for the same reason. The module doc was also still advertising `flush(path)` and `open(path)`, an API that never existed at all. --- .agentcoder/transaction.lock | 0 .../backup/src/hydrate/io.rs | 210 +++++++ .../journal.json | 1 + .../backup/src/hydrate/mod.rs | 592 ++++++++++++++++++ .../journal.json | 1 + .../backup/src/hydrate/mod.rs | 592 ++++++++++++++++++ .../journal.json | 1 + .../backup/src/hydrate/mod.rs | 592 ++++++++++++++++++ .../journal.json | 1 + examples/disk_cost.rs | 7 +- src/hydrate/io.rs | 39 +- src/hydrate/mod.rs | 26 +- src/hydrate/tests.rs | 20 +- src/lib.rs | 2 +- 14 files changed, 2041 insertions(+), 43 deletions(-) create mode 100644 .agentcoder/transaction.lock create mode 100644 .agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/backup/src/hydrate/io.rs create mode 100644 .agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/journal.json create mode 100644 .agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/backup/src/hydrate/mod.rs create mode 100644 .agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/journal.json create mode 100644 .agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/backup/src/hydrate/mod.rs create mode 100644 .agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/journal.json create mode 100644 .agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/backup/src/hydrate/mod.rs create mode 100644 .agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/journal.json diff --git a/.agentcoder/transaction.lock b/.agentcoder/transaction.lock new file mode 100644 index 0000000..e69de29 diff --git a/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/backup/src/hydrate/io.rs b/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/backup/src/hydrate/io.rs new file mode 100644 index 0000000..e1eb143 --- /dev/null +++ b/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/backup/src/hydrate/io.rs @@ -0,0 +1,210 @@ +//! Reading and writing pages, without knowing what they are stored on. +//! +//! # no_std and real I/O at the same time +//! +//! The I/O is a pair of traits, not a filesystem. A caller with an operating +//! system wraps a `std::fs::File` and gets real files; a caller without one +//! implements two methods over whatever it has. Neither costs this crate `std`, +//! and there is no second code path for the two cases. +//! +//! ```ignore +//! // a real file, through the adapter +//! let file = std::fs::File::create("rows.wtv")?; +//! table.unload_to(&mut embedded_io_adapters::std::FromStd::new(file))?; +//! +//! // memory, because Vec and &[u8] already implement the traits +//! let mut bytes = Vec::new(); +//! table.unload_to(&mut bytes)?; +//! ``` +//! +//! # Streaming, one page at a time +//! +//! A write emits a page and moves on; a read consumes a page and moves on. +//! Neither holds the whole file, which is the other half of why pages stand +//! alone: a reader that had to see the last page before trusting the first +//! could not stream at all. +//! +//! Appending needs no support here. Pages are self contained, so appending is +//! opening the sink in append mode and writing more of them. + +use alloc::vec::Vec; + +use embedded_io::{Read, Write}; + +use super::{Codec, LoadError, PAGE_SIZE, fingerprint, page_rows, to_pages}; +use crate::{IndexedTable, LinearTable}; + +/// A read that failed, either at the transport or at the page. +/// +/// The two are kept apart on purpose. A disk that would not answer and a page +/// that was not what it claimed are different problems with different fixes, +/// and collapsing them into one string loses which one happened. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum HydrateError { + /// The reader failed. + Io(E), + /// The reader worked and the bytes were wrong. + Page(LoadError), + /// The last page stopped part way through. + /// + /// Distinct from a bad page: this is a write that did not finish, not a + /// page that was damaged after it did. + Torn { + /// Which page, counting from zero. + page: usize, + /// How many bytes of it arrived. + found: usize, + }, +} + +impl From for HydrateError { + fn from(error: LoadError) -> Self { + Self::Page(error) + } +} + +impl core::fmt::Display for HydrateError { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Io(error) => write!(formatter, "the reader failed: {error}"), + Self::Page(error) => error.fmt(formatter), + Self::Torn { page, found } => write!( + formatter, + "page {page} stops after {found} of {PAGE_SIZE} bytes" + ), + } + } +} + +impl core::error::Error for HydrateError {} + +/// Fill `page` from `source`, or say how far it got. +/// +/// Returns `Ok(false)` at a clean end of input, which is the only case where a +/// short read is not a problem. +fn fill( + source: &mut R, + page: &mut [u8], + index: usize, +) -> Result> { + let mut filled = 0; + while filled < page.len() { + match source.read(&mut page[filled..]).map_err(HydrateError::Io)? { + 0 if filled == 0 => return Ok(false), + 0 => { + return Err(HydrateError::Torn { + page: index, + found: filled, + }); + } + read => filled += read, + } + } + Ok(true) +} + +/// Every page from a reader, back into rows. +fn read_rows(source: &mut R) -> Result, HydrateError> +where + Vec<(K, V)>: Codec, +{ + let mut page = alloc::vec![0u8; PAGE_SIZE]; + let mut schema = None; + let mut rows = Vec::new(); + let mut index = 0; + + while fill(source, &mut page, index)? { + rows.append_to(&mut page_rows(&page, index, &mut schema)?); + index += 1; + } + + if index == 0 { + return Err(HydrateError::Page(LoadError::NotWholePages { found: 0 })); + } + let expected = fingerprint::>(); + match schema { + Some(found) if found != expected => Err(HydrateError::Page(LoadError::ForeignRows { + found, + expected, + })), + _ => Ok(rows), + } +} + +impl LinearTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + /// Write every row, as pages. + /// + /// # Errors + /// + /// Whatever the writer reports. + pub fn unload_to(&self, sink: &mut W) -> Result<(), W::Error> { + sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; + sink.flush() + } + + /// Write the rows from `first` on, for a sink already holding the rest. + /// + /// Appending works because pages stand alone: what is already written is + /// untouched and these are simply more pages. + /// + /// # Errors + /// + /// Whatever the writer reports. + pub fn append_to(&self, sink: &mut W, first: usize) -> Result<(), W::Error> { + let first = first.min(self.rows.len()); + if first == self.rows.len() { + return sink.flush(); + } + sink.write_all(&to_pages(&self.rows[first..], fingerprint::>()))?; + sink.flush() + } + + /// Read a table back from a reader. + /// + /// # Errors + /// + /// The reader's own errors, a page that stops part way, or any of the + /// refusals in [`LoadError`]. + pub fn load_from(source: &mut R) -> Result> { + Ok(Self { + rows: read_rows(source)?, + }) + } +} + +impl IndexedTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, + K: Ord + Clone, +{ + /// Write every row, as pages. + /// + /// The index is not written, because it is derived from the rows. + /// + /// # Errors + /// + /// Whatever the writer reports. + pub fn unload_to(&self, sink: &mut W) -> Result<(), W::Error> { + sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; + sink.flush() + } + + /// Read a table back from a reader, rebuilding the index. + /// + /// # Errors + /// + /// As [`LinearTable::read`]. + pub fn load_from(source: &mut R) -> Result> { + let rows = read_rows(source)?; + let mut table = Self::with_capacity(rows.len()); + for (key, value) in rows { + let _ = table.insert(key, value); + } + Ok(table) + } +} diff --git a/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/journal.json b/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/journal.json new file mode 100644 index 0000000..6d31c63 --- /dev/null +++ b/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/journal.json @@ -0,0 +1 @@ +{"transaction_id":"6a9b2319-8d46-4b53-b221-87c513e8a640","state":"committed","entries":[{"relative_path":"src/hydrate/io.rs","new_revision":"dd320a71612b2d6d507d0bdd2639be2121e4ed0b7d6a777e7d3935849b4b93c8"}]} \ No newline at end of file diff --git a/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/backup/src/hydrate/mod.rs b/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/backup/src/hydrate/mod.rs new file mode 100644 index 0000000..6926736 --- /dev/null +++ b/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/backup/src/hydrate/mod.rs @@ -0,0 +1,592 @@ +//! Load and unload a table as pages, and put those pages on a disk. +//! +//! # The interface +//! +//! ```ignore +//! table.flush("rows.wtv")?; // write it +//! let table = LinearTable::open("rows.wtv")?; // read it back +//! table.append("rows.wtv", from_row)?; // add rows without a rewrite +//! ``` +//! +//! and the same thing without a filesystem, for callers that already hold the +//! bytes or do not have one: +//! +//! ```ignore +//! let bytes = table.unload(); +//! let table = LinearTable::load(&bytes)?; +//! ``` +//! +//! Rows live in a `Vec` while the table is in use and are pages only at rest. +//! Everything between an open and a flush runs at `Vec` speed because it *is* a +//! `Vec`. This is a codec plus a file, not a storage engine underneath. +//! +//! # Page based, and each page stands alone +//! +//! A page is 16 KiB: a 24 byte header, then an rkyv archive of **the rows that +//! fit in that page**, and nothing spanning the boundary. +//! +//! That last part is the whole design. An archive split across pages means one +//! damaged page destroys every row in the file, and it means appending a row +//! rewrites everything. Self contained pages make damage local and appends +//! O(new rows), and cost only the few bytes of archive overhead repeated per +//! page. +//! +//! # What is checked +//! +//! Every page carries a CRC-32 of its body, and every field in the header is +//! validated rather than merely written. rkyv's own validation checks that an +//! archive is structurally sound, which is not the same as checking that these +//! are the bytes that were written: a flipped bit inside a `u64` passes +//! structural validation and reads back as a different number. The checksum is +//! what catches that. +//! +//! **These are not WorkTable space files.** A WorkTable space opens with a page +//! carrying a name, a schema and a primary key list, and a `Vec<(K, V)>` +//! declares no schema to put there. + +use alloc::vec::Vec; + +use rkyv::api::high::{HighDeserializer, HighValidator}; +use rkyv::bytecheck::CheckBytes; +use rkyv::rancor::{Error as RkyvError, Strategy}; +use rkyv::ser::Serializer; +use rkyv::ser::allocator::ArenaHandle; +use rkyv::ser::sharing::Share; +use rkyv::util::AlignedVec; +use rkyv::{Archive, Deserialize, Serialize}; + +use crate::{IndexedTable, LinearTable}; + +/// One page, header included. +pub const PAGE_SIZE: usize = 4096 * 4; + +/// The fixed header every page opens with. +pub const HEADER_SIZE: usize = 24; + +/// How much of a page is body. +pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; + +/// Bumped when the page layout changes, so an older file is refused rather than +/// read through the new shape. +pub const PAGE_VERSION: u32 = 1; + +/// What a load can refuse on. +/// +/// Every variant is a statement about the bytes rather than about the caller, +/// and every one of them names the page, because a file that will not load is +/// a question about which page went wrong. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum LoadError { + /// The byte length is not a whole number of pages. + NotWholePages { + /// How many bytes arrived. + found: usize, + }, + /// A page carries a version this build does not write. + /// + /// Also what a page of zeroes looks like, which is the shape a torn write + /// leaves behind. + ForeignPages { + /// Which page, counting from zero. + page: usize, + /// The version that page claims. + version: u32, + }, + /// A header claimed a body longer than a page holds. + Overlong { + /// Which page, counting from zero. + page: usize, + /// What its header claimed. + claimed: usize, + }, + /// The body does not match the checksum written with it. + /// + /// This is the one rkyv cannot find. A flipped bit inside an integer is a + /// structurally perfect archive of the wrong number. + Corrupt { + /// Which page, counting from zero. + page: usize, + /// The checksum in the header. + expected: u32, + /// The checksum of the bytes actually there. + found: u32, + }, + /// The pages disagree with each other about the row type. + Inconsistent { + /// Which page disagreed. + page: usize, + }, + /// These are a different row type's bytes. + /// + /// Caught by a fingerprint rather than by deserialization, because + /// deserialization does not catch it: rkyv validates a `(u64, String)` + /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s + /// relative pointer as an integer. Keys look right, values are debris, and + /// nothing errors. + ForeignRows { + /// The fingerprint these bytes were written with. + found: u32, + /// The fingerprint this row type expects. + expected: u32, + }, + /// A page's rows did not deserialize. + Rows { + /// Which page, counting from zero. + page: usize, + }, + /// A page's header promised a row count its body did not contain. + RowCount { + /// Which page, counting from zero. + page: usize, + /// What the header promised. + expected: usize, + /// What the body held. + found: usize, + }, +} + +impl core::fmt::Display for LoadError { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::NotWholePages { found } => write!( + formatter, + "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" + ), + Self::ForeignPages { page, version } => write!( + formatter, + "page {page} is version {version}, and this build writes {PAGE_VERSION}" + ), + Self::Overlong { page, claimed } => write!( + formatter, + "page {page} claims a {claimed} byte body, more than a page holds" + ), + Self::Corrupt { + page, + expected, + found, + } => write!( + formatter, + "page {page} checksums {found:#010x} and its header says {expected:#010x}" + ), + Self::Inconsistent { page } => { + write!(formatter, "page {page} names a different row type") + } + Self::ForeignRows { found, expected } => write!( + formatter, + "these are row type {found:#010x}, and this is row type {expected:#010x}" + ), + Self::Rows { page } => write!(formatter, "page {page} did not deserialize"), + Self::RowCount { + page, + expected, + found, + } => write!( + formatter, + "page {page} promised {expected} rows and held {found}" + ), + } + } +} + +impl core::error::Error for LoadError {} + +/// What a row set has to be able to do to make the trip. +/// +/// The bounds are rkyv's and there are five lines of them, so they are stated +/// once here and every signature below asks only for `Codec`. The blanket impl +/// means a caller never names this trait either: any row pair whose key and +/// value already derive rkyv's traits satisfies it. +/// The one thing [`Codec::decode`] can say. Which page it happened on is the +/// caller's to add, because a codec does not know it is reading a page. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct NotAnArchive; + +pub trait Codec: Sized { + /// Rows to bytes. + fn encode(&self) -> AlignedVec<16>; + /// Bytes back to rows. + /// + /// # Errors + /// + /// Fails when the bytes are not this type's archive. + fn decode(bytes: &[u8]) -> Result; +} + +impl Codec for T +where + T: Archive + + for<'a> Serialize, ArenaHandle<'a>, Share>, RkyvError>>, + ::Archived: Deserialize> + + for<'a> CheckBytes>, +{ + fn encode(&self) -> AlignedVec<16> { + // Infallible in practice: the only failure rkyv reports here is an + // allocator refusing, which on this path means the process is already + // out of memory. + rkyv::to_bytes::(self).expect("rows serialize") + } + + fn decode(bytes: &[u8]) -> Result { + rkyv::from_bytes::(bytes).map_err(|_| NotAnArchive) + } +} + +/// What row type wrote these bytes. +/// +/// FNV-1a over `core::any::type_name`, which is neither stable across compiler +/// versions nor guaranteed unique. That is fine for what it is for: refusing an +/// obvious mismatch, not authenticating a schema. A false match is possible and +/// a false mismatch is a rebuild, so it fails toward refusing to load rather +/// than toward reinterpreting. +pub(crate) fn fingerprint() -> u32 { + let mut hash: u32 = 0x811c_9dc5; + for byte in core::any::type_name::().as_bytes() { + hash ^= u32::from(*byte); + hash = hash.wrapping_mul(0x0100_0193); + } + hash +} + +/// CRC-32, the usual reversed polynomial, computed a nibble at a time. +/// +/// Sixteen entries rather than a 256 entry table: this runs once per 16 KiB +/// page, so the table is cache noise and the loop is not the cost of anything. +fn crc32(bytes: &[u8]) -> u32 { + const NIBBLE: [u32; 16] = [ + 0x0000_0000, + 0x1db7_1064, + 0x3b6e_20c8, + 0x26d9_30ac, + 0x76dc_4190, + 0x6b6b_51f4, + 0x4db2_6158, + 0x5005_713c, + 0xedb8_8320, + 0xf00f_9344, + 0xd6d6_a3e8, + 0xcb61_b38c, + 0x9b64_c2b0, + 0x86d3_d2d4, + 0xa00a_e278, + 0xbdbd_f21c, + ]; + let mut crc = 0xffff_ffffu32; + for byte in bytes { + crc ^= u32::from(*byte); + crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; + crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; + } + !crc +} + +/// A page header: six little-endian `u32`s, in this order. +/// +/// Every one of them is checked on the way back in. A field that is written and +/// never validated is worse than a field that does not exist, because it reads +/// like a guarantee. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(crate) struct Header { + version: u32, + /// The row type every page in a run carries. + schema: u32, + /// How many rows this page's archive holds. + rows: u32, + /// How many bytes of that archive are in this page. + body: u32, + /// CRC-32 over exactly `body` bytes. + crc: u32, + /// Zero for now. A layout change that needs a flag has somewhere to put it + /// without moving anything else. + flags: u32, +} + +impl Header { + fn write(self, out: &mut Vec) { + for field in [ + self.version, + self.schema, + self.rows, + self.body, + self.crc, + self.flags, + ] { + out.extend_from_slice(&field.to_le_bytes()); + } + } + + fn read(raw: &[u8]) -> Self { + let at = |n: usize| { + let mut word = [0u8; 4]; + word.copy_from_slice(&raw[n * 4..n * 4 + 4]); + u32::from_le_bytes(word) + }; + Self { + version: at(0), + schema: at(1), + rows: at(2), + body: at(3), + crc: at(4), + flags: at(5), + } + } +} + +/// The most rows of `rows` whose archive fits one page body. +/// +/// **Bounded probes.** The obvious version binary searches over the whole +/// remaining slice, which re-serializes every row still to be written on +/// every probe, for every page. That measured 2.3 seconds to write what rkyv +/// alone encodes in 6.7 ms, because the work is quadratic in the row count. +/// +/// So the search is bounded to roughly two pages of rows: one sample encode +/// gives bytes per row, the estimate from that sets the ceiling, and the +/// binary search runs under it. Every probe serializes about a page, never a +/// file. Uniform rows land in a probe or two and wildly variable rows still +/// terminate, because the ceiling is only a ceiling. +/// +/// Always returns at least one for a non-empty slice, so the caller always +/// makes progress. A single row too large for a page is written as an +/// oversized page rather than looping forever. +fn rows_per_page(rows: &[(K, V)], hint: usize) -> usize +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + if rows.is_empty() { + return 0; + } + + let fits = |take: usize| rows[..take].to_vec().encode().len() <= BODY_SIZE; + + // A page holds about what the last one held, so start there and walk. + // Uniform rows settle in a probe or two; only the first page, or a run + // whose rows change size, pays for a search. + if hint > 0 && hint <= rows.len() && fits(hint) { + let mut take = hint; + while take < rows.len() && fits(take + 1) { + take += 1; + } + return take; + } + + // No usable hint, or the rows grew. One sample gives bytes per row, and + // the estimate from it bounds the search to about two pages of rows. + let sample = rows.len().min(64); + let sampled = rows[..sample].to_vec().encode().len(); + let estimate = (BODY_SIZE * sample) + .checked_div(sampled) + .map_or(rows.len(), |estimate| estimate.max(1)); + let mut low = 1usize; + let mut high = rows.len().min(estimate.saturating_mul(2)).max(1); + while low < high { + let mid = low + (high - low).div_ceil(2); + if fits(mid) { + low = mid; + } else { + high = mid - 1; + } + } + low +} + +/// Rows to pages, each page standing alone. +pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Vec +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + let mut out = Vec::new(); + let mut rest = rows; + let mut hint = 0usize; + + // An empty table still writes one page. A zero byte file is + // indistinguishable from a missing one, and a load has to tell "no rows" + // from "nothing landed". + loop { + // Zero for an empty table, which still writes its one page and stops. + // At least one for anything else, so this always makes progress. + let take = rows_per_page(rest, hint); + hint = take; + let archive = rest[..take].to_vec().encode(); + let body = archive.as_ref(); + Header { + version: PAGE_VERSION, + schema, + rows: u32::try_from(take).expect("a row count inside u32"), + body: u32::try_from(body.len()).expect("a body inside u32"), + crc: crc32(body), + flags: 0, + } + .write(&mut out); + out.extend_from_slice(body); + out.resize(out.len().next_multiple_of(PAGE_SIZE), 0); + + rest = &rest[take..]; + if rest.is_empty() { + break; + } + } + out +} + +/// One page back into rows, with every header field checked. +pub(crate) fn page_rows( + raw: &[u8], + index: usize, + schema: &mut Option, +) -> Result, LoadError> +where + Vec<(K, V)>: Codec, +{ + let header = Header::read(&raw[..HEADER_SIZE]); + if header.version != PAGE_VERSION { + return Err(LoadError::ForeignPages { + page: index, + version: header.version, + }); + } + match schema { + None => *schema = Some(header.schema), + // Every page names the row type, so a file spliced onto another is + // caught where they stop agreeing rather than concatenated. + Some(first) if *first != header.schema => { + return Err(LoadError::Inconsistent { page: index }); + } + Some(_) => {} + } + + let take = header.body as usize; + if take > BODY_SIZE { + return Err(LoadError::Overlong { + page: index, + claimed: take, + }); + } + let body = &raw[HEADER_SIZE..HEADER_SIZE + take]; + let found = crc32(body); + if found != header.crc { + return Err(LoadError::Corrupt { + page: index, + expected: header.crc, + found, + }); + } + + // Copied into an AlignedVec because rkyv reads an archive in place and + // needs it aligned. A page body sits at offset 24 in a Vec, which is + // aligned to nothing in particular. + let mut aligned = AlignedVec::<16>::with_capacity(take); + aligned.extend_from_slice(body); + let rows = + Vec::<(K, V)>::decode(&aligned).map_err(|NotAnArchive| LoadError::Rows { page: index })?; + if rows.len() != header.rows as usize { + return Err(LoadError::RowCount { + page: index, + expected: header.rows as usize, + found: rows.len(), + }); + } + Ok(rows) +} + +/// Every page back into one row vector. +fn from_pages(bytes: &[u8]) -> Result, LoadError> +where + Vec<(K, V)>: Codec, +{ + if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { + return Err(LoadError::NotWholePages { found: bytes.len() }); + } + + let mut schema = None; + let mut rows = Vec::new(); + for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { + rows.append_to(&mut page_rows(raw, index, &mut schema)?); + } + let expected = fingerprint::>(); + match schema { + Some(found) if found != expected => Err(LoadError::ForeignRows { found, expected }), + _ => Ok(rows), + } +} + +impl LinearTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + /// Every row, as pages. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages(&self.rows, fingerprint::>()) + } + + /// The rows from `first` on, as pages, ready to append to a file that + /// already holds the ones before it. + /// + /// Appending is possible at all because pages stand alone: the existing + /// file is untouched and these pages are simply more of them. + #[must_use] + pub fn unload_appending(&self, first: usize) -> Vec { + let first = first.min(self.rows.len()); + to_pages(&self.rows[first..], fingerprint::>()) + } + + /// Rows back from pages. + /// + /// # Errors + /// + /// Refuses bytes that are not whole pages, a page from another version, a + /// header claiming more body than a page holds, a body that fails its + /// checksum, pages that disagree about the row type, another row type, or + /// rows that do not deserialize. + pub fn load(bytes: &[u8]) -> Result { + Ok(Self { + rows: from_pages(bytes)?, + }) + } +} + +impl IndexedTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, + K: Ord + Clone, +{ + /// Every row, as pages. + /// + /// The index is not written. It is derived from the rows, so rebuilding it + /// on load costs one pass, where storing it would cost bytes at rest and a + /// second thing that can disagree with the rows. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages(&self.rows, fingerprint::>()) + } + + /// The rows from `first` on, as pages. + #[must_use] + pub fn unload_appending(&self, first: usize) -> Vec { + let first = first.min(self.rows.len()); + to_pages(&self.rows[first..], fingerprint::>()) + } + + /// Rows back from pages, with the index rebuilt. + /// + /// # Errors + /// + /// As [`LinearTable::load`]. + pub fn load(bytes: &[u8]) -> Result { + let rows = from_pages(bytes)?; + let mut table = Self::with_capacity(rows.len()); + for (key, value) in rows { + let _ = table.insert(key, value); + } + Ok(table) + } +} + +mod io; +pub use io::HydrateError; + +#[cfg(test)] +mod tests; diff --git a/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/journal.json b/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/journal.json new file mode 100644 index 0000000..30c9612 --- /dev/null +++ b/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/journal.json @@ -0,0 +1 @@ +{"transaction_id":"9e77492e-3cd7-4bcc-92ee-aa925ce1247c","state":"committed","entries":[{"relative_path":"src/hydrate/mod.rs","new_revision":"abb9b81da7b7956833dd20b6d3a23f588b9a2f0c4b9e87385574b8961a1fb722"}]} \ No newline at end of file diff --git a/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/backup/src/hydrate/mod.rs b/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/backup/src/hydrate/mod.rs new file mode 100644 index 0000000..72ae621 --- /dev/null +++ b/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/backup/src/hydrate/mod.rs @@ -0,0 +1,592 @@ +//! Load and unload a table as pages, and put those pages on a disk. +//! +//! # The interface +//! +//! ```ignore +//! table.flush("rows.wtv")?; // write it +//! let table = LinearTable::open("rows.wtv")?; // read it back +//! table.append("rows.wtv", from_row)?; // add rows without a rewrite +//! ``` +//! +//! and the same thing without a filesystem, for callers that already hold the +//! bytes or do not have one: +//! +//! ```ignore +//! let bytes = table.unload(); +//! let table = LinearTable::load(&bytes)?; +//! ``` +//! +//! Rows live in a `Vec` while the table is in use and are pages only at rest. +//! Everything between an open and a flush runs at `Vec` speed because it *is* a +//! `Vec`. This is a codec plus a file, not a storage engine underneath. +//! +//! # Page based, and each page stands alone +//! +//! A page is 16 KiB: a 24 byte header, then an rkyv archive of **the rows that +//! fit in that page**, and nothing spanning the boundary. +//! +//! That last part is the whole design. An archive split across pages means one +//! damaged page destroys every row in the file, and it means appending a row +//! rewrites everything. Self contained pages make damage local and appends +//! O(new rows), and cost only the few bytes of archive overhead repeated per +//! page. +//! +//! # What is checked +//! +//! Every page carries a CRC-32 of its body, and every field in the header is +//! validated rather than merely written. rkyv's own validation checks that an +//! archive is structurally sound, which is not the same as checking that these +//! are the bytes that were written: a flipped bit inside a `u64` passes +//! structural validation and reads back as a different number. The checksum is +//! what catches that. +//! +//! **These are not WorkTable space files.** A WorkTable space opens with a page +//! carrying a name, a schema and a primary key list, and a `Vec<(K, V)>` +//! declares no schema to put there. + +use alloc::vec::Vec; + +use rkyv::api::high::{HighDeserializer, HighValidator}; +use rkyv::bytecheck::CheckBytes; +use rkyv::rancor::{Error as RkyvError, Strategy}; +use rkyv::ser::Serializer; +use rkyv::ser::allocator::ArenaHandle; +use rkyv::ser::sharing::Share; +use rkyv::util::AlignedVec; +use rkyv::{Archive, Deserialize, Serialize}; + +use crate::{IndexedTable, LinearTable}; + +/// One page, header included. +pub const PAGE_SIZE: usize = 4096 * 4; + +/// The fixed header every page opens with. +pub const HEADER_SIZE: usize = 24; + +/// How much of a page is body. +pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; + +/// Bumped when the page layout changes, so an older file is refused rather than +/// read through the new shape. +pub const PAGE_VERSION: u32 = 1; + +/// What a load can refuse on. +/// +/// Every variant is a statement about the bytes rather than about the caller, +/// and every one of them names the page, because a file that will not load is +/// a question about which page went wrong. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum LoadError { + /// The byte length is not a whole number of pages. + NotWholePages { + /// How many bytes arrived. + found: usize, + }, + /// A page carries a version this build does not write. + /// + /// Also what a page of zeroes looks like, which is the shape a torn write + /// leaves behind. + ForeignPages { + /// Which page, counting from zero. + page: usize, + /// The version that page claims. + version: u32, + }, + /// A header claimed a body longer than a page holds. + Overlong { + /// Which page, counting from zero. + page: usize, + /// What its header claimed. + claimed: usize, + }, + /// The body does not match the checksum written with it. + /// + /// This is the one rkyv cannot find. A flipped bit inside an integer is a + /// structurally perfect archive of the wrong number. + Corrupt { + /// Which page, counting from zero. + page: usize, + /// The checksum in the header. + expected: u32, + /// The checksum of the bytes actually there. + found: u32, + }, + /// The pages disagree with each other about the row type. + Inconsistent { + /// Which page disagreed. + page: usize, + }, + /// These are a different row type's bytes. + /// + /// Caught by a fingerprint rather than by deserialization, because + /// deserialization does not catch it: rkyv validates a `(u64, String)` + /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s + /// relative pointer as an integer. Keys look right, values are debris, and + /// nothing errors. + ForeignRows { + /// The fingerprint these bytes were written with. + found: u32, + /// The fingerprint this row type expects. + expected: u32, + }, + /// A page's rows did not deserialize. + Rows { + /// Which page, counting from zero. + page: usize, + }, + /// A page's header promised a row count its body did not contain. + RowCount { + /// Which page, counting from zero. + page: usize, + /// What the header promised. + expected: usize, + /// What the body held. + found: usize, + }, +} + +impl core::fmt::Display for LoadError { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::NotWholePages { found } => write!( + formatter, + "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" + ), + Self::ForeignPages { page, version } => write!( + formatter, + "page {page} is version {version}, and this build writes {PAGE_VERSION}" + ), + Self::Overlong { page, claimed } => write!( + formatter, + "page {page} claims a {claimed} byte body, more than a page holds" + ), + Self::Corrupt { + page, + expected, + found, + } => write!( + formatter, + "page {page} checksums {found:#010x} and its header says {expected:#010x}" + ), + Self::Inconsistent { page } => { + write!(formatter, "page {page} names a different row type") + } + Self::ForeignRows { found, expected } => write!( + formatter, + "these are row type {found:#010x}, and this is row type {expected:#010x}" + ), + Self::Rows { page } => write!(formatter, "page {page} did not deserialize"), + Self::RowCount { + page, + expected, + found, + } => write!( + formatter, + "page {page} promised {expected} rows and held {found}" + ), + } + } +} + +impl core::error::Error for LoadError {} + +/// What a row set has to be able to do to make the trip. +/// +/// The bounds are rkyv's and there are five lines of them, so they are stated +/// once here and every signature below asks only for `Codec`. The blanket impl +/// means a caller never names this trait either: any row pair whose key and +/// value already derive rkyv's traits satisfies it. +/// The one thing [`Codec::decode`] can say. Which page it happened on is the +/// caller's to add, because a codec does not know it is reading a page. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct NotAnArchive; + +pub trait Codec: Sized { + /// Rows to bytes. + fn encode(&self) -> AlignedVec<16>; + /// Bytes back to rows. + /// + /// # Errors + /// + /// Fails when the bytes are not this type's archive. + fn decode(bytes: &[u8]) -> Result; +} + +impl Codec for T +where + T: Archive + + for<'a> Serialize, ArenaHandle<'a>, Share>, RkyvError>>, + ::Archived: Deserialize> + + for<'a> CheckBytes>, +{ + fn encode(&self) -> AlignedVec<16> { + // Infallible in practice: the only failure rkyv reports here is an + // allocator refusing, which on this path means the process is already + // out of memory. + rkyv::to_bytes::(self).expect("rows serialize") + } + + fn decode(bytes: &[u8]) -> Result { + rkyv::from_bytes::(bytes).map_err(|_| NotAnArchive) + } +} + +/// What row type wrote these bytes. +/// +/// FNV-1a over `core::any::type_name`, which is neither stable across compiler +/// versions nor guaranteed unique. That is fine for what it is for: refusing an +/// obvious mismatch, not authenticating a schema. A false match is possible and +/// a false mismatch is a rebuild, so it fails toward refusing to load rather +/// than toward reinterpreting. +pub(crate) fn fingerprint() -> u32 { + let mut hash: u32 = 0x811c_9dc5; + for byte in core::any::type_name::().as_bytes() { + hash ^= u32::from(*byte); + hash = hash.wrapping_mul(0x0100_0193); + } + hash +} + +/// CRC-32, the usual reversed polynomial, computed a nibble at a time. +/// +/// Sixteen entries rather than a 256 entry table: this runs once per 16 KiB +/// page, so the table is cache noise and the loop is not the cost of anything. +fn crc32(bytes: &[u8]) -> u32 { + const NIBBLE: [u32; 16] = [ + 0x0000_0000, + 0x1db7_1064, + 0x3b6e_20c8, + 0x26d9_30ac, + 0x76dc_4190, + 0x6b6b_51f4, + 0x4db2_6158, + 0x5005_713c, + 0xedb8_8320, + 0xf00f_9344, + 0xd6d6_a3e8, + 0xcb61_b38c, + 0x9b64_c2b0, + 0x86d3_d2d4, + 0xa00a_e278, + 0xbdbd_f21c, + ]; + let mut crc = 0xffff_ffffu32; + for byte in bytes { + crc ^= u32::from(*byte); + crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; + crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; + } + !crc +} + +/// A page header: six little-endian `u32`s, in this order. +/// +/// Every one of them is checked on the way back in. A field that is written and +/// never validated is worse than a field that does not exist, because it reads +/// like a guarantee. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(crate) struct Header { + version: u32, + /// The row type every page in a run carries. + schema: u32, + /// How many rows this page's archive holds. + rows: u32, + /// How many bytes of that archive are in this page. + body: u32, + /// CRC-32 over exactly `body` bytes. + crc: u32, + /// Zero for now. A layout change that needs a flag has somewhere to put it + /// without moving anything else. + flags: u32, +} + +impl Header { + fn write(self, out: &mut Vec) { + for field in [ + self.version, + self.schema, + self.rows, + self.body, + self.crc, + self.flags, + ] { + out.extend_from_slice(&field.to_le_bytes()); + } + } + + fn read(raw: &[u8]) -> Self { + let at = |n: usize| { + let mut word = [0u8; 4]; + word.copy_from_slice(&raw[n * 4..n * 4 + 4]); + u32::from_le_bytes(word) + }; + Self { + version: at(0), + schema: at(1), + rows: at(2), + body: at(3), + crc: at(4), + flags: at(5), + } + } +} + +/// The most rows of `rows` whose archive fits one page body. +/// +/// **Bounded probes.** The obvious version binary searches over the whole +/// remaining slice, which re-serializes every row still to be written on +/// every probe, for every page. That measured 2.3 seconds to write what rkyv +/// alone encodes in 6.7 ms, because the work is quadratic in the row count. +/// +/// So the search is bounded to roughly two pages of rows: one sample encode +/// gives bytes per row, the estimate from that sets the ceiling, and the +/// binary search runs under it. Every probe serializes about a page, never a +/// file. Uniform rows land in a probe or two and wildly variable rows still +/// terminate, because the ceiling is only a ceiling. +/// +/// Always returns at least one for a non-empty slice, so the caller always +/// makes progress. A single row too large for a page is written as an +/// oversized page rather than looping forever. +fn rows_per_page(rows: &[(K, V)], hint: usize) -> usize +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + if rows.is_empty() { + return 0; + } + + let fits = |take: usize| rows[..take].to_vec().encode().len() <= BODY_SIZE; + + // A page holds about what the last one held, so start there and walk. + // Uniform rows settle in a probe or two; only the first page, or a run + // whose rows change size, pays for a search. + if hint > 0 && hint <= rows.len() && fits(hint) { + let mut take = hint; + while take < rows.len() && fits(take + 1) { + take += 1; + } + return take; + } + + // No usable hint, or the rows grew. One sample gives bytes per row, and + // the estimate from it bounds the search to about two pages of rows. + let sample = rows.len().min(64); + let sampled = rows[..sample].to_vec().encode().len(); + let estimate = (BODY_SIZE * sample) + .checked_div(sampled) + .map_or(rows.len(), |estimate| estimate.max(1)); + let mut low = 1usize; + let mut high = rows.len().min(estimate.saturating_mul(2)).max(1); + while low < high { + let mid = low + (high - low).div_ceil(2); + if fits(mid) { + low = mid; + } else { + high = mid - 1; + } + } + low +} + +/// Rows to pages, each page standing alone. +pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Vec +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + let mut out = Vec::new(); + let mut rest = rows; + let mut hint = 0usize; + + // An empty table still writes one page. A zero byte file is + // indistinguishable from a missing one, and a load has to tell "no rows" + // from "nothing landed". + loop { + // Zero for an empty table, which still writes its one page and stops. + // At least one for anything else, so this always makes progress. + let take = rows_per_page(rest, hint); + hint = take; + let archive = rest[..take].to_vec().encode(); + let body = archive.as_ref(); + Header { + version: PAGE_VERSION, + schema, + rows: u32::try_from(take).expect("a row count inside u32"), + body: u32::try_from(body.len()).expect("a body inside u32"), + crc: crc32(body), + flags: 0, + } + .unload_to(&mut out); + out.extend_from_slice(body); + out.resize(out.len().next_multiple_of(PAGE_SIZE), 0); + + rest = &rest[take..]; + if rest.is_empty() { + break; + } + } + out +} + +/// One page back into rows, with every header field checked. +pub(crate) fn page_rows( + raw: &[u8], + index: usize, + schema: &mut Option, +) -> Result, LoadError> +where + Vec<(K, V)>: Codec, +{ + let header = Header::read(&raw[..HEADER_SIZE]); + if header.version != PAGE_VERSION { + return Err(LoadError::ForeignPages { + page: index, + version: header.version, + }); + } + match schema { + None => *schema = Some(header.schema), + // Every page names the row type, so a file spliced onto another is + // caught where they stop agreeing rather than concatenated. + Some(first) if *first != header.schema => { + return Err(LoadError::Inconsistent { page: index }); + } + Some(_) => {} + } + + let take = header.body as usize; + if take > BODY_SIZE { + return Err(LoadError::Overlong { + page: index, + claimed: take, + }); + } + let body = &raw[HEADER_SIZE..HEADER_SIZE + take]; + let found = crc32(body); + if found != header.crc { + return Err(LoadError::Corrupt { + page: index, + expected: header.crc, + found, + }); + } + + // Copied into an AlignedVec because rkyv reads an archive in place and + // needs it aligned. A page body sits at offset 24 in a Vec, which is + // aligned to nothing in particular. + let mut aligned = AlignedVec::<16>::with_capacity(take); + aligned.extend_from_slice(body); + let rows = + Vec::<(K, V)>::decode(&aligned).map_err(|NotAnArchive| LoadError::Rows { page: index })?; + if rows.len() != header.rows as usize { + return Err(LoadError::RowCount { + page: index, + expected: header.rows as usize, + found: rows.len(), + }); + } + Ok(rows) +} + +/// Every page back into one row vector. +fn from_pages(bytes: &[u8]) -> Result, LoadError> +where + Vec<(K, V)>: Codec, +{ + if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { + return Err(LoadError::NotWholePages { found: bytes.len() }); + } + + let mut schema = None; + let mut rows = Vec::new(); + for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { + rows.append_to(&mut page_rows(raw, index, &mut schema)?); + } + let expected = fingerprint::>(); + match schema { + Some(found) if found != expected => Err(LoadError::ForeignRows { found, expected }), + _ => Ok(rows), + } +} + +impl LinearTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + /// Every row, as pages. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages(&self.rows, fingerprint::>()) + } + + /// The rows from `first` on, as pages, ready to append to a file that + /// already holds the ones before it. + /// + /// Appending is possible at all because pages stand alone: the existing + /// file is untouched and these pages are simply more of them. + #[must_use] + pub fn unload_appending(&self, first: usize) -> Vec { + let first = first.min(self.rows.len()); + to_pages(&self.rows[first..], fingerprint::>()) + } + + /// Rows back from pages. + /// + /// # Errors + /// + /// Refuses bytes that are not whole pages, a page from another version, a + /// header claiming more body than a page holds, a body that fails its + /// checksum, pages that disagree about the row type, another row type, or + /// rows that do not deserialize. + pub fn load(bytes: &[u8]) -> Result { + Ok(Self { + rows: from_pages(bytes)?, + }) + } +} + +impl IndexedTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, + K: Ord + Clone, +{ + /// Every row, as pages. + /// + /// The index is not written. It is derived from the rows, so rebuilding it + /// on load costs one pass, where storing it would cost bytes at rest and a + /// second thing that can disagree with the rows. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages(&self.rows, fingerprint::>()) + } + + /// The rows from `first` on, as pages. + #[must_use] + pub fn unload_appending(&self, first: usize) -> Vec { + let first = first.min(self.rows.len()); + to_pages(&self.rows[first..], fingerprint::>()) + } + + /// Rows back from pages, with the index rebuilt. + /// + /// # Errors + /// + /// As [`LinearTable::load`]. + pub fn load(bytes: &[u8]) -> Result { + let rows = from_pages(bytes)?; + let mut table = Self::with_capacity(rows.len()); + for (key, value) in rows { + let _ = table.insert(key, value); + } + Ok(table) + } +} + +mod io; +pub use io::HydrateError; + +#[cfg(test)] +mod tests; diff --git a/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/journal.json b/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/journal.json new file mode 100644 index 0000000..437a91b --- /dev/null +++ b/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/journal.json @@ -0,0 +1 @@ +{"transaction_id":"e45443a5-bf32-48ae-9e52-c62b7f4a2732","state":"committed","entries":[{"relative_path":"src/hydrate/mod.rs","new_revision":"1bc3d272a28040188c663bf061a1e5661df6eb555b2cd512e41b3b6917cde64d"}]} \ No newline at end of file diff --git a/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/backup/src/hydrate/mod.rs b/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/backup/src/hydrate/mod.rs new file mode 100644 index 0000000..0c733e3 --- /dev/null +++ b/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/backup/src/hydrate/mod.rs @@ -0,0 +1,592 @@ +//! Load and unload a table as pages, and put those pages on a disk. +//! +//! # The interface +//! +//! ```ignore +//! table.flush("rows.wtv")?; // write it +//! let table = LinearTable::open("rows.wtv")?; // read it back +//! table.append("rows.wtv", from_row)?; // add rows without a rewrite +//! ``` +//! +//! and the same thing without a filesystem, for callers that already hold the +//! bytes or do not have one: +//! +//! ```ignore +//! let bytes = table.unload(); +//! let table = LinearTable::load(&bytes)?; +//! ``` +//! +//! Rows live in a `Vec` while the table is in use and are pages only at rest. +//! Everything between an open and a flush runs at `Vec` speed because it *is* a +//! `Vec`. This is a codec plus a file, not a storage engine underneath. +//! +//! # Page based, and each page stands alone +//! +//! A page is 16 KiB: a 24 byte header, then an rkyv archive of **the rows that +//! fit in that page**, and nothing spanning the boundary. +//! +//! That last part is the whole design. An archive split across pages means one +//! damaged page destroys every row in the file, and it means appending a row +//! rewrites everything. Self contained pages make damage local and appends +//! O(new rows), and cost only the few bytes of archive overhead repeated per +//! page. +//! +//! # What is checked +//! +//! Every page carries a CRC-32 of its body, and every field in the header is +//! validated rather than merely written. rkyv's own validation checks that an +//! archive is structurally sound, which is not the same as checking that these +//! are the bytes that were written: a flipped bit inside a `u64` passes +//! structural validation and reads back as a different number. The checksum is +//! what catches that. +//! +//! **These are not WorkTable space files.** A WorkTable space opens with a page +//! carrying a name, a schema and a primary key list, and a `Vec<(K, V)>` +//! declares no schema to put there. + +use alloc::vec::Vec; + +use rkyv::api::high::{HighDeserializer, HighValidator}; +use rkyv::bytecheck::CheckBytes; +use rkyv::rancor::{Error as RkyvError, Strategy}; +use rkyv::ser::Serializer; +use rkyv::ser::allocator::ArenaHandle; +use rkyv::ser::sharing::Share; +use rkyv::util::AlignedVec; +use rkyv::{Archive, Deserialize, Serialize}; + +use crate::{IndexedTable, LinearTable}; + +/// One page, header included. +pub const PAGE_SIZE: usize = 4096 * 4; + +/// The fixed header every page opens with. +pub const HEADER_SIZE: usize = 24; + +/// How much of a page is body. +pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; + +/// Bumped when the page layout changes, so an older file is refused rather than +/// read through the new shape. +pub const PAGE_VERSION: u32 = 1; + +/// What a load can refuse on. +/// +/// Every variant is a statement about the bytes rather than about the caller, +/// and every one of them names the page, because a file that will not load is +/// a question about which page went wrong. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum LoadError { + /// The byte length is not a whole number of pages. + NotWholePages { + /// How many bytes arrived. + found: usize, + }, + /// A page carries a version this build does not write. + /// + /// Also what a page of zeroes looks like, which is the shape a torn write + /// leaves behind. + ForeignPages { + /// Which page, counting from zero. + page: usize, + /// The version that page claims. + version: u32, + }, + /// A header claimed a body longer than a page holds. + Overlong { + /// Which page, counting from zero. + page: usize, + /// What its header claimed. + claimed: usize, + }, + /// The body does not match the checksum written with it. + /// + /// This is the one rkyv cannot find. A flipped bit inside an integer is a + /// structurally perfect archive of the wrong number. + Corrupt { + /// Which page, counting from zero. + page: usize, + /// The checksum in the header. + expected: u32, + /// The checksum of the bytes actually there. + found: u32, + }, + /// The pages disagree with each other about the row type. + Inconsistent { + /// Which page disagreed. + page: usize, + }, + /// These are a different row type's bytes. + /// + /// Caught by a fingerprint rather than by deserialization, because + /// deserialization does not catch it: rkyv validates a `(u64, String)` + /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s + /// relative pointer as an integer. Keys look right, values are debris, and + /// nothing errors. + ForeignRows { + /// The fingerprint these bytes were written with. + found: u32, + /// The fingerprint this row type expects. + expected: u32, + }, + /// A page's rows did not deserialize. + Rows { + /// Which page, counting from zero. + page: usize, + }, + /// A page's header promised a row count its body did not contain. + RowCount { + /// Which page, counting from zero. + page: usize, + /// What the header promised. + expected: usize, + /// What the body held. + found: usize, + }, +} + +impl core::fmt::Display for LoadError { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::NotWholePages { found } => write!( + formatter, + "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" + ), + Self::ForeignPages { page, version } => write!( + formatter, + "page {page} is version {version}, and this build writes {PAGE_VERSION}" + ), + Self::Overlong { page, claimed } => write!( + formatter, + "page {page} claims a {claimed} byte body, more than a page holds" + ), + Self::Corrupt { + page, + expected, + found, + } => write!( + formatter, + "page {page} checksums {found:#010x} and its header says {expected:#010x}" + ), + Self::Inconsistent { page } => { + write!(formatter, "page {page} names a different row type") + } + Self::ForeignRows { found, expected } => write!( + formatter, + "these are row type {found:#010x}, and this is row type {expected:#010x}" + ), + Self::Rows { page } => write!(formatter, "page {page} did not deserialize"), + Self::RowCount { + page, + expected, + found, + } => write!( + formatter, + "page {page} promised {expected} rows and held {found}" + ), + } + } +} + +impl core::error::Error for LoadError {} + +/// What a row set has to be able to do to make the trip. +/// +/// The bounds are rkyv's and there are five lines of them, so they are stated +/// once here and every signature below asks only for `Codec`. The blanket impl +/// means a caller never names this trait either: any row pair whose key and +/// value already derive rkyv's traits satisfies it. +/// The one thing [`Codec::decode`] can say. Which page it happened on is the +/// caller's to add, because a codec does not know it is reading a page. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct NotAnArchive; + +pub trait Codec: Sized { + /// Rows to bytes. + fn encode(&self) -> AlignedVec<16>; + /// Bytes back to rows. + /// + /// # Errors + /// + /// Fails when the bytes are not this type's archive. + fn decode(bytes: &[u8]) -> Result; +} + +impl Codec for T +where + T: Archive + + for<'a> Serialize, ArenaHandle<'a>, Share>, RkyvError>>, + ::Archived: Deserialize> + + for<'a> CheckBytes>, +{ + fn encode(&self) -> AlignedVec<16> { + // Infallible in practice: the only failure rkyv reports here is an + // allocator refusing, which on this path means the process is already + // out of memory. + rkyv::to_bytes::(self).expect("rows serialize") + } + + fn decode(bytes: &[u8]) -> Result { + rkyv::from_bytes::(bytes).map_err(|_| NotAnArchive) + } +} + +/// What row type wrote these bytes. +/// +/// FNV-1a over `core::any::type_name`, which is neither stable across compiler +/// versions nor guaranteed unique. That is fine for what it is for: refusing an +/// obvious mismatch, not authenticating a schema. A false match is possible and +/// a false mismatch is a rebuild, so it fails toward refusing to load rather +/// than toward reinterpreting. +pub(crate) fn fingerprint() -> u32 { + let mut hash: u32 = 0x811c_9dc5; + for byte in core::any::type_name::().as_bytes() { + hash ^= u32::from(*byte); + hash = hash.wrapping_mul(0x0100_0193); + } + hash +} + +/// CRC-32, the usual reversed polynomial, computed a nibble at a time. +/// +/// Sixteen entries rather than a 256 entry table: this runs once per 16 KiB +/// page, so the table is cache noise and the loop is not the cost of anything. +fn crc32(bytes: &[u8]) -> u32 { + const NIBBLE: [u32; 16] = [ + 0x0000_0000, + 0x1db7_1064, + 0x3b6e_20c8, + 0x26d9_30ac, + 0x76dc_4190, + 0x6b6b_51f4, + 0x4db2_6158, + 0x5005_713c, + 0xedb8_8320, + 0xf00f_9344, + 0xd6d6_a3e8, + 0xcb61_b38c, + 0x9b64_c2b0, + 0x86d3_d2d4, + 0xa00a_e278, + 0xbdbd_f21c, + ]; + let mut crc = 0xffff_ffffu32; + for byte in bytes { + crc ^= u32::from(*byte); + crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; + crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; + } + !crc +} + +/// A page header: six little-endian `u32`s, in this order. +/// +/// Every one of them is checked on the way back in. A field that is written and +/// never validated is worse than a field that does not exist, because it reads +/// like a guarantee. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(crate) struct Header { + version: u32, + /// The row type every page in a run carries. + schema: u32, + /// How many rows this page's archive holds. + rows: u32, + /// How many bytes of that archive are in this page. + body: u32, + /// CRC-32 over exactly `body` bytes. + crc: u32, + /// Zero for now. A layout change that needs a flag has somewhere to put it + /// without moving anything else. + flags: u32, +} + +impl Header { + fn write(self, out: &mut Vec) { + for field in [ + self.version, + self.schema, + self.rows, + self.body, + self.crc, + self.flags, + ] { + out.extend_from_slice(&field.to_le_bytes()); + } + } + + fn read(raw: &[u8]) -> Self { + let at = |n: usize| { + let mut word = [0u8; 4]; + word.copy_from_slice(&raw[n * 4..n * 4 + 4]); + u32::from_le_bytes(word) + }; + Self { + version: at(0), + schema: at(1), + rows: at(2), + body: at(3), + crc: at(4), + flags: at(5), + } + } +} + +/// The most rows of `rows` whose archive fits one page body. +/// +/// **Bounded probes.** The obvious version binary searches over the whole +/// remaining slice, which re-serializes every row still to be written on +/// every probe, for every page. That measured 2.3 seconds to write what rkyv +/// alone encodes in 6.7 ms, because the work is quadratic in the row count. +/// +/// So the search is bounded to roughly two pages of rows: one sample encode +/// gives bytes per row, the estimate from that sets the ceiling, and the +/// binary search runs under it. Every probe serializes about a page, never a +/// file. Uniform rows land in a probe or two and wildly variable rows still +/// terminate, because the ceiling is only a ceiling. +/// +/// Always returns at least one for a non-empty slice, so the caller always +/// makes progress. A single row too large for a page is written as an +/// oversized page rather than looping forever. +fn rows_per_page(rows: &[(K, V)], hint: usize) -> usize +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + if rows.is_empty() { + return 0; + } + + let fits = |take: usize| rows[..take].to_vec().encode().len() <= BODY_SIZE; + + // A page holds about what the last one held, so start there and walk. + // Uniform rows settle in a probe or two; only the first page, or a run + // whose rows change size, pays for a search. + if hint > 0 && hint <= rows.len() && fits(hint) { + let mut take = hint; + while take < rows.len() && fits(take + 1) { + take += 1; + } + return take; + } + + // No usable hint, or the rows grew. One sample gives bytes per row, and + // the estimate from it bounds the search to about two pages of rows. + let sample = rows.len().min(64); + let sampled = rows[..sample].to_vec().encode().len(); + let estimate = (BODY_SIZE * sample) + .checked_div(sampled) + .map_or(rows.len(), |estimate| estimate.max(1)); + let mut low = 1usize; + let mut high = rows.len().min(estimate.saturating_mul(2)).max(1); + while low < high { + let mid = low + (high - low).div_ceil(2); + if fits(mid) { + low = mid; + } else { + high = mid - 1; + } + } + low +} + +/// Rows to pages, each page standing alone. +pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Vec +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + let mut out = Vec::new(); + let mut rest = rows; + let mut hint = 0usize; + + // An empty table still writes one page. A zero byte file is + // indistinguishable from a missing one, and a load has to tell "no rows" + // from "nothing landed". + loop { + // Zero for an empty table, which still writes its one page and stops. + // At least one for anything else, so this always makes progress. + let take = rows_per_page(rest, hint); + hint = take; + let archive = rest[..take].to_vec().encode(); + let body = archive.as_ref(); + Header { + version: PAGE_VERSION, + schema, + rows: u32::try_from(take).expect("a row count inside u32"), + body: u32::try_from(body.len()).expect("a body inside u32"), + crc: crc32(body), + flags: 0, + } + .write(&mut out); + out.extend_from_slice(body); + out.resize(out.len().next_multiple_of(PAGE_SIZE), 0); + + rest = &rest[take..]; + if rest.is_empty() { + break; + } + } + out +} + +/// One page back into rows, with every header field checked. +pub(crate) fn page_rows( + raw: &[u8], + index: usize, + schema: &mut Option, +) -> Result, LoadError> +where + Vec<(K, V)>: Codec, +{ + let header = Header::read(&raw[..HEADER_SIZE]); + if header.version != PAGE_VERSION { + return Err(LoadError::ForeignPages { + page: index, + version: header.version, + }); + } + match schema { + None => *schema = Some(header.schema), + // Every page names the row type, so a file spliced onto another is + // caught where they stop agreeing rather than concatenated. + Some(first) if *first != header.schema => { + return Err(LoadError::Inconsistent { page: index }); + } + Some(_) => {} + } + + let take = header.body as usize; + if take > BODY_SIZE { + return Err(LoadError::Overlong { + page: index, + claimed: take, + }); + } + let body = &raw[HEADER_SIZE..HEADER_SIZE + take]; + let found = crc32(body); + if found != header.crc { + return Err(LoadError::Corrupt { + page: index, + expected: header.crc, + found, + }); + } + + // Copied into an AlignedVec because rkyv reads an archive in place and + // needs it aligned. A page body sits at offset 24 in a Vec, which is + // aligned to nothing in particular. + let mut aligned = AlignedVec::<16>::with_capacity(take); + aligned.extend_from_slice(body); + let rows = + Vec::<(K, V)>::decode(&aligned).map_err(|NotAnArchive| LoadError::Rows { page: index })?; + if rows.len() != header.rows as usize { + return Err(LoadError::RowCount { + page: index, + expected: header.rows as usize, + found: rows.len(), + }); + } + Ok(rows) +} + +/// Every page back into one row vector. +fn from_pages(bytes: &[u8]) -> Result, LoadError> +where + Vec<(K, V)>: Codec, +{ + if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { + return Err(LoadError::NotWholePages { found: bytes.len() }); + } + + let mut schema = None; + let mut rows = Vec::new(); + for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { + rows.append(&mut page_rows(raw, index, &mut schema)?); + } + let expected = fingerprint::>(); + match schema { + Some(found) if found != expected => Err(LoadError::ForeignRows { found, expected }), + _ => Ok(rows), + } +} + +impl LinearTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, +{ + /// Every row, as pages. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages(&self.rows, fingerprint::>()) + } + + /// The rows from `first` on, as pages, ready to append to a file that + /// already holds the ones before it. + /// + /// Appending is possible at all because pages stand alone: the existing + /// file is untouched and these pages are simply more of them. + #[must_use] + pub fn unload_appending(&self, first: usize) -> Vec { + let first = first.min(self.rows.len()); + to_pages(&self.rows[first..], fingerprint::>()) + } + + /// Rows back from pages. + /// + /// # Errors + /// + /// Refuses bytes that are not whole pages, a page from another version, a + /// header claiming more body than a page holds, a body that fails its + /// checksum, pages that disagree about the row type, another row type, or + /// rows that do not deserialize. + pub fn load(bytes: &[u8]) -> Result { + Ok(Self { + rows: from_pages(bytes)?, + }) + } +} + +impl IndexedTable +where + Vec<(K, V)>: Codec, + (K, V): Clone, + K: Ord + Clone, +{ + /// Every row, as pages. + /// + /// The index is not written. It is derived from the rows, so rebuilding it + /// on load costs one pass, where storing it would cost bytes at rest and a + /// second thing that can disagree with the rows. + #[must_use] + pub fn unload(&self) -> Vec { + to_pages(&self.rows, fingerprint::>()) + } + + /// The rows from `first` on, as pages. + #[must_use] + pub fn unload_appending(&self, first: usize) -> Vec { + let first = first.min(self.rows.len()); + to_pages(&self.rows[first..], fingerprint::>()) + } + + /// Rows back from pages, with the index rebuilt. + /// + /// # Errors + /// + /// As [`LinearTable::load`]. + pub fn load(bytes: &[u8]) -> Result { + let rows = from_pages(bytes)?; + let mut table = Self::with_capacity(rows.len()); + for (key, value) in rows { + let _ = table.insert(key, value); + } + Ok(table) + } +} + +mod io; +pub use io::HydrateError; + +#[cfg(test)] +mod tests; diff --git a/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/journal.json b/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/journal.json new file mode 100644 index 0000000..6d997e0 --- /dev/null +++ b/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/journal.json @@ -0,0 +1 @@ +{"transaction_id":"f0675041-5387-4d5f-bbea-b03fd7d79425","state":"committed","entries":[{"relative_path":"src/hydrate/mod.rs","new_revision":"7d87cdb31dfa04a16f5790d8b61f3dff895b4da527f47e5e3ce11750cfeeff74"}]} \ No newline at end of file diff --git a/examples/disk_cost.rs b/examples/disk_cost.rs index 69468cf..e21b88c 100644 --- a/examples/disk_cost.rs +++ b/examples/disk_cost.rs @@ -29,14 +29,15 @@ fn main() -> Result<(), Box> { let now = Instant::now(); { let file = std::fs::File::create(&path)?; - table.write(&mut FromStd::new(file))?; + table.unload_to(&mut FromStd::new(file))?; } wrote.push(now.elapsed()); bytes = std::fs::metadata(&path)?.len(); let now = Instant::now(); - let back = LinearTable::::read(&mut FromStd::new(std::fs::File::open(&path)?)) - .map_err(|error| format!("{error}"))?; + let back = + LinearTable::::load_from(&mut FromStd::new(std::fs::File::open(&path)?)) + .map_err(|error| format!("{error}"))?; read.push(now.elapsed()); assert_eq!(back.len(), ROWS); } diff --git a/src/hydrate/io.rs b/src/hydrate/io.rs index 1232f57..ce12700 100644 --- a/src/hydrate/io.rs +++ b/src/hydrate/io.rs @@ -10,11 +10,11 @@ //! ```ignore //! // a real file, through the adapter //! let file = std::fs::File::create("rows.wtv")?; -//! table.write(&mut embedded_io_adapters::std::FromStd::new(file))?; +//! table.unload_to(&mut embedded_io_adapters::std::FromStd::new(file))?; //! //! // memory, because Vec and &[u8] already implement the traits //! let mut bytes = Vec::new(); -//! table.write(&mut bytes)?; +//! table.unload_to(&mut bytes)?; //! ``` //! //! # Streaming, one page at a time @@ -40,7 +40,7 @@ use crate::{IndexedTable, LinearTable}; /// that was not what it claimed are different problems with different fixes, /// and collapsing them into one string loses which one happened. #[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum ReadError { +pub enum HydrateError { /// The reader failed. Io(E), /// The reader worked and the bytes were wrong. @@ -57,13 +57,13 @@ pub enum ReadError { }, } -impl From for ReadError { +impl From for HydrateError { fn from(error: LoadError) -> Self { Self::Page(error) } } -impl core::fmt::Display for ReadError { +impl core::fmt::Display for HydrateError { fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { match self { Self::Io(error) => write!(formatter, "the reader failed: {error}"), @@ -76,7 +76,7 @@ impl core::fmt::Display for ReadError { } } -impl core::error::Error for ReadError {} +impl core::error::Error for HydrateError {} /// Fill `page` from `source`, or say how far it got. /// @@ -86,13 +86,13 @@ fn fill( source: &mut R, page: &mut [u8], index: usize, -) -> Result> { +) -> Result> { let mut filled = 0; while filled < page.len() { - match source.read(&mut page[filled..]).map_err(ReadError::Io)? { + match source.read(&mut page[filled..]).map_err(HydrateError::Io)? { 0 if filled == 0 => return Ok(false), 0 => { - return Err(ReadError::Torn { + return Err(HydrateError::Torn { page: index, found: filled, }); @@ -104,7 +104,7 @@ fn fill( } /// Every page from a reader, back into rows. -fn read_rows(source: &mut R) -> Result, ReadError> +fn read_rows(source: &mut R) -> Result, HydrateError> where Vec<(K, V)>: Codec, { @@ -119,13 +119,14 @@ where } if index == 0 { - return Err(ReadError::Page(LoadError::NotWholePages { found: 0 })); + return Err(HydrateError::Page(LoadError::NotWholePages { found: 0 })); } let expected = fingerprint::>(); match schema { - Some(found) if found != expected => { - Err(ReadError::Page(LoadError::ForeignRows { found, expected })) - } + Some(found) if found != expected => Err(HydrateError::Page(LoadError::ForeignRows { + found, + expected, + })), _ => Ok(rows), } } @@ -140,7 +141,7 @@ where /// # Errors /// /// Whatever the writer reports. - pub fn write(&self, sink: &mut W) -> Result<(), W::Error> { + pub fn unload_to(&self, sink: &mut W) -> Result<(), W::Error> { sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; sink.flush() } @@ -153,7 +154,7 @@ where /// # Errors /// /// Whatever the writer reports. - pub fn append(&self, sink: &mut W, first: usize) -> Result<(), W::Error> { + pub fn append_to(&self, sink: &mut W, first: usize) -> Result<(), W::Error> { let first = first.min(self.rows.len()); if first == self.rows.len() { return sink.flush(); @@ -168,7 +169,7 @@ where /// /// The reader's own errors, a page that stops part way, or any of the /// refusals in [`LoadError`]. - pub fn read(source: &mut R) -> Result> { + pub fn load_from(source: &mut R) -> Result> { Ok(Self { rows: read_rows(source)?, }) @@ -188,7 +189,7 @@ where /// # Errors /// /// Whatever the writer reports. - pub fn write(&self, sink: &mut W) -> Result<(), W::Error> { + pub fn unload_to(&self, sink: &mut W) -> Result<(), W::Error> { sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; sink.flush() } @@ -198,7 +199,7 @@ where /// # Errors /// /// As [`LinearTable::read`]. - pub fn read(source: &mut R) -> Result> { + pub fn load_from(source: &mut R) -> Result> { let rows = read_rows(source)?; let mut table = Self::with_capacity(rows.len()); for (key, value) in rows { diff --git a/src/hydrate/mod.rs b/src/hydrate/mod.rs index 6b672ec..0157421 100644 --- a/src/hydrate/mod.rs +++ b/src/hydrate/mod.rs @@ -2,14 +2,20 @@ //! //! # The interface //! +//! Everything is a load or an unload, because what the two ends do is hydrate +//! a `Vec` and dehydrate it. `_to` and `_from` say where. +//! //! ```ignore -//! table.flush("rows.wtv")?; // write it -//! let table = LinearTable::open("rows.wtv")?; // read it back -//! table.append("rows.wtv", from_row)?; // add rows without a rewrite +//! table.unload_to(&mut sink)?; // rows out, as pages +//! let table = LinearTable::load_from(&mut source)?; // rows back into a Vec +//! table.append_to(&mut sink, from_row)?; // more pages, no rewrite //! ``` //! -//! and the same thing without a filesystem, for callers that already hold the -//! bytes or do not have one: +//! The sink and the source are `embedded_io` traits, so a `std::fs::File` +//! wrapped in an adapter is a real file and a `Vec` is memory, with no +//! second code path and no `std` in this crate. +//! +//! When the bytes are already in hand there is no I/O to do: //! //! ```ignore //! let bytes = table.unload(); @@ -17,8 +23,8 @@ //! ``` //! //! Rows live in a `Vec` while the table is in use and are pages only at rest. -//! Everything between an open and a flush runs at `Vec` speed because it *is* a -//! `Vec`. This is a codec plus a file, not a storage engine underneath. +//! Everything between a load and an unload runs at `Vec` speed because it *is* +//! a `Vec`. This is a codec plus a byte sink, not a storage engine underneath. //! //! # Page based, and each page stands alone //! @@ -527,7 +533,7 @@ where /// Appending is possible at all because pages stand alone: the existing /// file is untouched and these pages are simply more of them. #[must_use] - pub fn unload_from(&self, first: usize) -> Vec { + pub fn unload_appending(&self, first: usize) -> Vec { let first = first.min(self.rows.len()); to_pages(&self.rows[first..], fingerprint::>()) } @@ -565,7 +571,7 @@ where /// The rows from `first` on, as pages. #[must_use] - pub fn unload_from(&self, first: usize) -> Vec { + pub fn unload_appending(&self, first: usize) -> Vec { let first = first.min(self.rows.len()); to_pages(&self.rows[first..], fingerprint::>()) } @@ -586,7 +592,7 @@ where } mod io; -pub use io::ReadError; +pub use io::HydrateError; #[cfg(test)] mod tests; diff --git a/src/hydrate/tests.rs b/src/hydrate/tests.rs index e362211..f28582c 100644 --- a/src/hydrate/tests.rs +++ b/src/hydrate/tests.rs @@ -154,7 +154,7 @@ fn appended_pages_read_back_as_one_table() { first.push(key, value); } bytes.extend_from_slice(&first.unload()); - bytes.extend_from_slice(&whole.unload_from(2_000)); + bytes.extend_from_slice(&whole.unload_appending(2_000)); let back = LinearTable::::load(&bytes).expect("a load"); assert_eq!(back.rows(), whole.rows()); @@ -170,8 +170,8 @@ mod through_a_reader_and_a_writer { fn memory_needs_no_adapter() { let before = table(3_000); let mut sink = Vec::new(); - before.write(&mut sink).expect("a write"); - let back = LinearTable::::read(&mut sink.as_slice()).expect("a read"); + before.unload_to(&mut sink).expect("a write"); + let back = LinearTable::::load_from(&mut sink.as_slice()).expect("a read"); assert_eq!(back.rows(), before.rows()); } @@ -190,7 +190,7 @@ mod through_a_reader_and_a_writer { { let file = std::fs::File::create(&path).expect("a file"); before - .write(&mut FromStd::new(file)) + .unload_to(&mut FromStd::new(file)) .expect("a write to disk"); } @@ -202,7 +202,7 @@ mod through_a_reader_and_a_writer { ); let file = std::fs::File::open(&path).expect("the file back"); - let back = LinearTable::::read(&mut FromStd::new(file)).expect("a read"); + let back = LinearTable::::load_from(&mut FromStd::new(file)).expect("a read"); assert_eq!(back.rows(), before.rows()); let _ = std::fs::remove_file(&path); } @@ -224,7 +224,7 @@ mod through_a_reader_and_a_writer { { let file = std::fs::File::create(&path).expect("a file"); - first.write(&mut FromStd::new(file)).expect("a write"); + first.unload_to(&mut FromStd::new(file)).expect("a write"); } let after_first = std::fs::metadata(&path).expect("a stat").len(); @@ -234,7 +234,7 @@ mod through_a_reader_and_a_writer { .open(&path) .expect("the file, to append"); whole - .append(&mut FromStd::new(file), 2_000) + .append_to(&mut FromStd::new(file), 2_000) .expect("an append"); } let after_append = std::fs::metadata(&path).expect("a stat").len(); @@ -244,7 +244,7 @@ mod through_a_reader_and_a_writer { ); let file = std::fs::File::open(&path).expect("the file back"); - let back = LinearTable::::read(&mut FromStd::new(file)).expect("a read"); + let back = LinearTable::::load_from(&mut FromStd::new(file)).expect("a read"); assert_eq!(back.rows(), whole.rows()); let _ = std::fs::remove_file(&path); } @@ -255,8 +255,8 @@ mod through_a_reader_and_a_writer { fn a_half_written_page_is_torn_rather_than_ignored() { let bytes = table(5_000).unload(); let cut = bytes.len() - (PAGE_SIZE / 2); - match LinearTable::::read(&mut &bytes[..cut]) { - Err(ReadError::Torn { .. }) => {} + match LinearTable::::load_from(&mut &bytes[..cut]) { + Err(HydrateError::Torn { .. }) => {} other => panic!("a half written page has to be caught: {other:?}"), } } diff --git a/src/lib.rs b/src/lib.rs index 636899f..02d5da8 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -17,7 +17,7 @@ extern crate alloc; #[cfg(feature = "hydrate")] mod hydrate; #[cfg(feature = "hydrate")] -pub use hydrate::{Codec, LoadError, ReadError}; +pub use hydrate::{Codec, HydrateError, LoadError}; use alloc::collections::BTreeMap; #[cfg(feature = "congee")] From 036295354f6ff5b5f0bce8cc15aa0e287ee3f842 Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 14:45:49 +0700 Subject: [PATCH 06/16] Stop gating no_std on a CPU The workflow installed a target and ran a check against it to prove the crate is no_std. That proves something narrower and less useful: it pins a word size, and a dependency using AtomicU64 fails it for a reason that has nothing to do with no_std. `#![no_std]` enforces itself. A crate carrying the attribute cannot compile a `std::` path on any target, including the host, so the existing `--no-default-features` clippy and test steps already prove it. --- .github/workflows/ci.yml | 2 -- 1 file changed, 2 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 18ee4cd..75ed45d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -13,12 +13,10 @@ jobs: - uses: dtolnay/rust-toolchain@stable with: components: rustfmt, clippy - targets: thumbv7em-none-eabihf - uses: Swatinem/rust-cache@v2 - run: cargo fmt --all -- --check - run: cargo clippy --all-targets --no-default-features -- -D warnings - run: cargo test --all-targets --no-default-features - - run: cargo check --lib --no-default-features --target thumbv7em-none-eabihf - run: cargo clippy --all-targets --features arctic -- -D warnings - run: cargo test --all-targets --features arctic - run: cargo clippy --all-targets --all-features -- -D warnings From a8781bf63d7eec3d0a0b4b15ed5d1918ac4c270a Mon Sep 17 00:00:00 2001 From: meh Date: Sun, 6 Sep 2026 14:48:50 +0700 Subject: [PATCH 07/16] Write DataBucket's page format at version 3, with a row directory I left this half-changed and the crate did not build. That is the first thing this fixes. The header is now DataBucket's `GeneralHeader` byte for byte: seven little-endian u32s in declaration order, which is what `rkyv::to_bytes` of that struct produces, since it has no relative pointers and pads `page_type` from u16 to four bytes. Verified against data_bucket 0.5.7 and pinned by `the_header_is_databuckets_layout`, because a layout reproduced in two places drifts and something has to notice. It is reproduced rather than imported because `data_bucket` is std, through tokio for file access and eyre in its signatures, and this crate is not. **Version 3 is the version that has a row directory.** WorkTable writes 2, and a 2 page carries no directory, which is exactly why a WorkTable data page cannot be read without its index: rows are bump allocated with `data_length` as a high water mark and no delimiters. Eight bytes at the tail of every page fix that here, a row count and a CRC-32, and `every_page_declares_its_own_rows` holds it: the pages account for every row with no index in sight. The count and checksum live in the directory rather than the header because the header is not this crate's to extend. That is the slotted page shape, and it is what makes the format readable by something that did not write it. The `space_id` field carries the row type fingerprint, since a `Vec<(K, V)>` belongs to no space. A WorkTable reader meets a space id it does not know, which is the honest outcome. Unchanged by any of it: flush 67.0 ms, open 52.0 ms over 816 pages. --- .agentcoder/transaction.lock | 0 .../backup/src/hydrate/io.rs | 210 ------- .../journal.json | 1 - .../backup/src/hydrate/mod.rs | 592 ------------------ .../journal.json | 1 - .../backup/src/hydrate/mod.rs | 592 ------------------ .../journal.json | 1 - .../backup/src/hydrate/mod.rs | 592 ------------------ .../journal.json | 1 - src/hydrate/mod.rs | 151 +++-- src/hydrate/tests.rs | 84 ++- 11 files changed, 196 insertions(+), 2029 deletions(-) delete mode 100644 .agentcoder/transaction.lock delete mode 100644 .agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/backup/src/hydrate/io.rs delete mode 100644 .agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/journal.json delete mode 100644 .agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/backup/src/hydrate/mod.rs delete mode 100644 .agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/journal.json delete mode 100644 .agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/backup/src/hydrate/mod.rs delete mode 100644 .agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/journal.json delete mode 100644 .agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/backup/src/hydrate/mod.rs delete mode 100644 .agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/journal.json diff --git a/.agentcoder/transaction.lock b/.agentcoder/transaction.lock deleted file mode 100644 index e69de29..0000000 diff --git a/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/backup/src/hydrate/io.rs b/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/backup/src/hydrate/io.rs deleted file mode 100644 index e1eb143..0000000 --- a/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/backup/src/hydrate/io.rs +++ /dev/null @@ -1,210 +0,0 @@ -//! Reading and writing pages, without knowing what they are stored on. -//! -//! # no_std and real I/O at the same time -//! -//! The I/O is a pair of traits, not a filesystem. A caller with an operating -//! system wraps a `std::fs::File` and gets real files; a caller without one -//! implements two methods over whatever it has. Neither costs this crate `std`, -//! and there is no second code path for the two cases. -//! -//! ```ignore -//! // a real file, through the adapter -//! let file = std::fs::File::create("rows.wtv")?; -//! table.unload_to(&mut embedded_io_adapters::std::FromStd::new(file))?; -//! -//! // memory, because Vec and &[u8] already implement the traits -//! let mut bytes = Vec::new(); -//! table.unload_to(&mut bytes)?; -//! ``` -//! -//! # Streaming, one page at a time -//! -//! A write emits a page and moves on; a read consumes a page and moves on. -//! Neither holds the whole file, which is the other half of why pages stand -//! alone: a reader that had to see the last page before trusting the first -//! could not stream at all. -//! -//! Appending needs no support here. Pages are self contained, so appending is -//! opening the sink in append mode and writing more of them. - -use alloc::vec::Vec; - -use embedded_io::{Read, Write}; - -use super::{Codec, LoadError, PAGE_SIZE, fingerprint, page_rows, to_pages}; -use crate::{IndexedTable, LinearTable}; - -/// A read that failed, either at the transport or at the page. -/// -/// The two are kept apart on purpose. A disk that would not answer and a page -/// that was not what it claimed are different problems with different fixes, -/// and collapsing them into one string loses which one happened. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum HydrateError { - /// The reader failed. - Io(E), - /// The reader worked and the bytes were wrong. - Page(LoadError), - /// The last page stopped part way through. - /// - /// Distinct from a bad page: this is a write that did not finish, not a - /// page that was damaged after it did. - Torn { - /// Which page, counting from zero. - page: usize, - /// How many bytes of it arrived. - found: usize, - }, -} - -impl From for HydrateError { - fn from(error: LoadError) -> Self { - Self::Page(error) - } -} - -impl core::fmt::Display for HydrateError { - fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::Io(error) => write!(formatter, "the reader failed: {error}"), - Self::Page(error) => error.fmt(formatter), - Self::Torn { page, found } => write!( - formatter, - "page {page} stops after {found} of {PAGE_SIZE} bytes" - ), - } - } -} - -impl core::error::Error for HydrateError {} - -/// Fill `page` from `source`, or say how far it got. -/// -/// Returns `Ok(false)` at a clean end of input, which is the only case where a -/// short read is not a problem. -fn fill( - source: &mut R, - page: &mut [u8], - index: usize, -) -> Result> { - let mut filled = 0; - while filled < page.len() { - match source.read(&mut page[filled..]).map_err(HydrateError::Io)? { - 0 if filled == 0 => return Ok(false), - 0 => { - return Err(HydrateError::Torn { - page: index, - found: filled, - }); - } - read => filled += read, - } - } - Ok(true) -} - -/// Every page from a reader, back into rows. -fn read_rows(source: &mut R) -> Result, HydrateError> -where - Vec<(K, V)>: Codec, -{ - let mut page = alloc::vec![0u8; PAGE_SIZE]; - let mut schema = None; - let mut rows = Vec::new(); - let mut index = 0; - - while fill(source, &mut page, index)? { - rows.append_to(&mut page_rows(&page, index, &mut schema)?); - index += 1; - } - - if index == 0 { - return Err(HydrateError::Page(LoadError::NotWholePages { found: 0 })); - } - let expected = fingerprint::>(); - match schema { - Some(found) if found != expected => Err(HydrateError::Page(LoadError::ForeignRows { - found, - expected, - })), - _ => Ok(rows), - } -} - -impl LinearTable -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - /// Write every row, as pages. - /// - /// # Errors - /// - /// Whatever the writer reports. - pub fn unload_to(&self, sink: &mut W) -> Result<(), W::Error> { - sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; - sink.flush() - } - - /// Write the rows from `first` on, for a sink already holding the rest. - /// - /// Appending works because pages stand alone: what is already written is - /// untouched and these are simply more pages. - /// - /// # Errors - /// - /// Whatever the writer reports. - pub fn append_to(&self, sink: &mut W, first: usize) -> Result<(), W::Error> { - let first = first.min(self.rows.len()); - if first == self.rows.len() { - return sink.flush(); - } - sink.write_all(&to_pages(&self.rows[first..], fingerprint::>()))?; - sink.flush() - } - - /// Read a table back from a reader. - /// - /// # Errors - /// - /// The reader's own errors, a page that stops part way, or any of the - /// refusals in [`LoadError`]. - pub fn load_from(source: &mut R) -> Result> { - Ok(Self { - rows: read_rows(source)?, - }) - } -} - -impl IndexedTable -where - Vec<(K, V)>: Codec, - (K, V): Clone, - K: Ord + Clone, -{ - /// Write every row, as pages. - /// - /// The index is not written, because it is derived from the rows. - /// - /// # Errors - /// - /// Whatever the writer reports. - pub fn unload_to(&self, sink: &mut W) -> Result<(), W::Error> { - sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; - sink.flush() - } - - /// Read a table back from a reader, rebuilding the index. - /// - /// # Errors - /// - /// As [`LinearTable::read`]. - pub fn load_from(source: &mut R) -> Result> { - let rows = read_rows(source)?; - let mut table = Self::with_capacity(rows.len()); - for (key, value) in rows { - let _ = table.insert(key, value); - } - Ok(table) - } -} diff --git a/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/journal.json b/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/journal.json deleted file mode 100644 index 6d31c63..0000000 --- a/.agentcoder/transactions/6a9b2319-8d46-4b53-b221-87c513e8a640/journal.json +++ /dev/null @@ -1 +0,0 @@ -{"transaction_id":"6a9b2319-8d46-4b53-b221-87c513e8a640","state":"committed","entries":[{"relative_path":"src/hydrate/io.rs","new_revision":"dd320a71612b2d6d507d0bdd2639be2121e4ed0b7d6a777e7d3935849b4b93c8"}]} \ No newline at end of file diff --git a/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/backup/src/hydrate/mod.rs b/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/backup/src/hydrate/mod.rs deleted file mode 100644 index 6926736..0000000 --- a/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/backup/src/hydrate/mod.rs +++ /dev/null @@ -1,592 +0,0 @@ -//! Load and unload a table as pages, and put those pages on a disk. -//! -//! # The interface -//! -//! ```ignore -//! table.flush("rows.wtv")?; // write it -//! let table = LinearTable::open("rows.wtv")?; // read it back -//! table.append("rows.wtv", from_row)?; // add rows without a rewrite -//! ``` -//! -//! and the same thing without a filesystem, for callers that already hold the -//! bytes or do not have one: -//! -//! ```ignore -//! let bytes = table.unload(); -//! let table = LinearTable::load(&bytes)?; -//! ``` -//! -//! Rows live in a `Vec` while the table is in use and are pages only at rest. -//! Everything between an open and a flush runs at `Vec` speed because it *is* a -//! `Vec`. This is a codec plus a file, not a storage engine underneath. -//! -//! # Page based, and each page stands alone -//! -//! A page is 16 KiB: a 24 byte header, then an rkyv archive of **the rows that -//! fit in that page**, and nothing spanning the boundary. -//! -//! That last part is the whole design. An archive split across pages means one -//! damaged page destroys every row in the file, and it means appending a row -//! rewrites everything. Self contained pages make damage local and appends -//! O(new rows), and cost only the few bytes of archive overhead repeated per -//! page. -//! -//! # What is checked -//! -//! Every page carries a CRC-32 of its body, and every field in the header is -//! validated rather than merely written. rkyv's own validation checks that an -//! archive is structurally sound, which is not the same as checking that these -//! are the bytes that were written: a flipped bit inside a `u64` passes -//! structural validation and reads back as a different number. The checksum is -//! what catches that. -//! -//! **These are not WorkTable space files.** A WorkTable space opens with a page -//! carrying a name, a schema and a primary key list, and a `Vec<(K, V)>` -//! declares no schema to put there. - -use alloc::vec::Vec; - -use rkyv::api::high::{HighDeserializer, HighValidator}; -use rkyv::bytecheck::CheckBytes; -use rkyv::rancor::{Error as RkyvError, Strategy}; -use rkyv::ser::Serializer; -use rkyv::ser::allocator::ArenaHandle; -use rkyv::ser::sharing::Share; -use rkyv::util::AlignedVec; -use rkyv::{Archive, Deserialize, Serialize}; - -use crate::{IndexedTable, LinearTable}; - -/// One page, header included. -pub const PAGE_SIZE: usize = 4096 * 4; - -/// The fixed header every page opens with. -pub const HEADER_SIZE: usize = 24; - -/// How much of a page is body. -pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; - -/// Bumped when the page layout changes, so an older file is refused rather than -/// read through the new shape. -pub const PAGE_VERSION: u32 = 1; - -/// What a load can refuse on. -/// -/// Every variant is a statement about the bytes rather than about the caller, -/// and every one of them names the page, because a file that will not load is -/// a question about which page went wrong. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum LoadError { - /// The byte length is not a whole number of pages. - NotWholePages { - /// How many bytes arrived. - found: usize, - }, - /// A page carries a version this build does not write. - /// - /// Also what a page of zeroes looks like, which is the shape a torn write - /// leaves behind. - ForeignPages { - /// Which page, counting from zero. - page: usize, - /// The version that page claims. - version: u32, - }, - /// A header claimed a body longer than a page holds. - Overlong { - /// Which page, counting from zero. - page: usize, - /// What its header claimed. - claimed: usize, - }, - /// The body does not match the checksum written with it. - /// - /// This is the one rkyv cannot find. A flipped bit inside an integer is a - /// structurally perfect archive of the wrong number. - Corrupt { - /// Which page, counting from zero. - page: usize, - /// The checksum in the header. - expected: u32, - /// The checksum of the bytes actually there. - found: u32, - }, - /// The pages disagree with each other about the row type. - Inconsistent { - /// Which page disagreed. - page: usize, - }, - /// These are a different row type's bytes. - /// - /// Caught by a fingerprint rather than by deserialization, because - /// deserialization does not catch it: rkyv validates a `(u64, String)` - /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s - /// relative pointer as an integer. Keys look right, values are debris, and - /// nothing errors. - ForeignRows { - /// The fingerprint these bytes were written with. - found: u32, - /// The fingerprint this row type expects. - expected: u32, - }, - /// A page's rows did not deserialize. - Rows { - /// Which page, counting from zero. - page: usize, - }, - /// A page's header promised a row count its body did not contain. - RowCount { - /// Which page, counting from zero. - page: usize, - /// What the header promised. - expected: usize, - /// What the body held. - found: usize, - }, -} - -impl core::fmt::Display for LoadError { - fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::NotWholePages { found } => write!( - formatter, - "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" - ), - Self::ForeignPages { page, version } => write!( - formatter, - "page {page} is version {version}, and this build writes {PAGE_VERSION}" - ), - Self::Overlong { page, claimed } => write!( - formatter, - "page {page} claims a {claimed} byte body, more than a page holds" - ), - Self::Corrupt { - page, - expected, - found, - } => write!( - formatter, - "page {page} checksums {found:#010x} and its header says {expected:#010x}" - ), - Self::Inconsistent { page } => { - write!(formatter, "page {page} names a different row type") - } - Self::ForeignRows { found, expected } => write!( - formatter, - "these are row type {found:#010x}, and this is row type {expected:#010x}" - ), - Self::Rows { page } => write!(formatter, "page {page} did not deserialize"), - Self::RowCount { - page, - expected, - found, - } => write!( - formatter, - "page {page} promised {expected} rows and held {found}" - ), - } - } -} - -impl core::error::Error for LoadError {} - -/// What a row set has to be able to do to make the trip. -/// -/// The bounds are rkyv's and there are five lines of them, so they are stated -/// once here and every signature below asks only for `Codec`. The blanket impl -/// means a caller never names this trait either: any row pair whose key and -/// value already derive rkyv's traits satisfies it. -/// The one thing [`Codec::decode`] can say. Which page it happened on is the -/// caller's to add, because a codec does not know it is reading a page. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub struct NotAnArchive; - -pub trait Codec: Sized { - /// Rows to bytes. - fn encode(&self) -> AlignedVec<16>; - /// Bytes back to rows. - /// - /// # Errors - /// - /// Fails when the bytes are not this type's archive. - fn decode(bytes: &[u8]) -> Result; -} - -impl Codec for T -where - T: Archive - + for<'a> Serialize, ArenaHandle<'a>, Share>, RkyvError>>, - ::Archived: Deserialize> - + for<'a> CheckBytes>, -{ - fn encode(&self) -> AlignedVec<16> { - // Infallible in practice: the only failure rkyv reports here is an - // allocator refusing, which on this path means the process is already - // out of memory. - rkyv::to_bytes::(self).expect("rows serialize") - } - - fn decode(bytes: &[u8]) -> Result { - rkyv::from_bytes::(bytes).map_err(|_| NotAnArchive) - } -} - -/// What row type wrote these bytes. -/// -/// FNV-1a over `core::any::type_name`, which is neither stable across compiler -/// versions nor guaranteed unique. That is fine for what it is for: refusing an -/// obvious mismatch, not authenticating a schema. A false match is possible and -/// a false mismatch is a rebuild, so it fails toward refusing to load rather -/// than toward reinterpreting. -pub(crate) fn fingerprint() -> u32 { - let mut hash: u32 = 0x811c_9dc5; - for byte in core::any::type_name::().as_bytes() { - hash ^= u32::from(*byte); - hash = hash.wrapping_mul(0x0100_0193); - } - hash -} - -/// CRC-32, the usual reversed polynomial, computed a nibble at a time. -/// -/// Sixteen entries rather than a 256 entry table: this runs once per 16 KiB -/// page, so the table is cache noise and the loop is not the cost of anything. -fn crc32(bytes: &[u8]) -> u32 { - const NIBBLE: [u32; 16] = [ - 0x0000_0000, - 0x1db7_1064, - 0x3b6e_20c8, - 0x26d9_30ac, - 0x76dc_4190, - 0x6b6b_51f4, - 0x4db2_6158, - 0x5005_713c, - 0xedb8_8320, - 0xf00f_9344, - 0xd6d6_a3e8, - 0xcb61_b38c, - 0x9b64_c2b0, - 0x86d3_d2d4, - 0xa00a_e278, - 0xbdbd_f21c, - ]; - let mut crc = 0xffff_ffffu32; - for byte in bytes { - crc ^= u32::from(*byte); - crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; - crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; - } - !crc -} - -/// A page header: six little-endian `u32`s, in this order. -/// -/// Every one of them is checked on the way back in. A field that is written and -/// never validated is worse than a field that does not exist, because it reads -/// like a guarantee. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub(crate) struct Header { - version: u32, - /// The row type every page in a run carries. - schema: u32, - /// How many rows this page's archive holds. - rows: u32, - /// How many bytes of that archive are in this page. - body: u32, - /// CRC-32 over exactly `body` bytes. - crc: u32, - /// Zero for now. A layout change that needs a flag has somewhere to put it - /// without moving anything else. - flags: u32, -} - -impl Header { - fn write(self, out: &mut Vec) { - for field in [ - self.version, - self.schema, - self.rows, - self.body, - self.crc, - self.flags, - ] { - out.extend_from_slice(&field.to_le_bytes()); - } - } - - fn read(raw: &[u8]) -> Self { - let at = |n: usize| { - let mut word = [0u8; 4]; - word.copy_from_slice(&raw[n * 4..n * 4 + 4]); - u32::from_le_bytes(word) - }; - Self { - version: at(0), - schema: at(1), - rows: at(2), - body: at(3), - crc: at(4), - flags: at(5), - } - } -} - -/// The most rows of `rows` whose archive fits one page body. -/// -/// **Bounded probes.** The obvious version binary searches over the whole -/// remaining slice, which re-serializes every row still to be written on -/// every probe, for every page. That measured 2.3 seconds to write what rkyv -/// alone encodes in 6.7 ms, because the work is quadratic in the row count. -/// -/// So the search is bounded to roughly two pages of rows: one sample encode -/// gives bytes per row, the estimate from that sets the ceiling, and the -/// binary search runs under it. Every probe serializes about a page, never a -/// file. Uniform rows land in a probe or two and wildly variable rows still -/// terminate, because the ceiling is only a ceiling. -/// -/// Always returns at least one for a non-empty slice, so the caller always -/// makes progress. A single row too large for a page is written as an -/// oversized page rather than looping forever. -fn rows_per_page(rows: &[(K, V)], hint: usize) -> usize -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - if rows.is_empty() { - return 0; - } - - let fits = |take: usize| rows[..take].to_vec().encode().len() <= BODY_SIZE; - - // A page holds about what the last one held, so start there and walk. - // Uniform rows settle in a probe or two; only the first page, or a run - // whose rows change size, pays for a search. - if hint > 0 && hint <= rows.len() && fits(hint) { - let mut take = hint; - while take < rows.len() && fits(take + 1) { - take += 1; - } - return take; - } - - // No usable hint, or the rows grew. One sample gives bytes per row, and - // the estimate from it bounds the search to about two pages of rows. - let sample = rows.len().min(64); - let sampled = rows[..sample].to_vec().encode().len(); - let estimate = (BODY_SIZE * sample) - .checked_div(sampled) - .map_or(rows.len(), |estimate| estimate.max(1)); - let mut low = 1usize; - let mut high = rows.len().min(estimate.saturating_mul(2)).max(1); - while low < high { - let mid = low + (high - low).div_ceil(2); - if fits(mid) { - low = mid; - } else { - high = mid - 1; - } - } - low -} - -/// Rows to pages, each page standing alone. -pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Vec -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - let mut out = Vec::new(); - let mut rest = rows; - let mut hint = 0usize; - - // An empty table still writes one page. A zero byte file is - // indistinguishable from a missing one, and a load has to tell "no rows" - // from "nothing landed". - loop { - // Zero for an empty table, which still writes its one page and stops. - // At least one for anything else, so this always makes progress. - let take = rows_per_page(rest, hint); - hint = take; - let archive = rest[..take].to_vec().encode(); - let body = archive.as_ref(); - Header { - version: PAGE_VERSION, - schema, - rows: u32::try_from(take).expect("a row count inside u32"), - body: u32::try_from(body.len()).expect("a body inside u32"), - crc: crc32(body), - flags: 0, - } - .write(&mut out); - out.extend_from_slice(body); - out.resize(out.len().next_multiple_of(PAGE_SIZE), 0); - - rest = &rest[take..]; - if rest.is_empty() { - break; - } - } - out -} - -/// One page back into rows, with every header field checked. -pub(crate) fn page_rows( - raw: &[u8], - index: usize, - schema: &mut Option, -) -> Result, LoadError> -where - Vec<(K, V)>: Codec, -{ - let header = Header::read(&raw[..HEADER_SIZE]); - if header.version != PAGE_VERSION { - return Err(LoadError::ForeignPages { - page: index, - version: header.version, - }); - } - match schema { - None => *schema = Some(header.schema), - // Every page names the row type, so a file spliced onto another is - // caught where they stop agreeing rather than concatenated. - Some(first) if *first != header.schema => { - return Err(LoadError::Inconsistent { page: index }); - } - Some(_) => {} - } - - let take = header.body as usize; - if take > BODY_SIZE { - return Err(LoadError::Overlong { - page: index, - claimed: take, - }); - } - let body = &raw[HEADER_SIZE..HEADER_SIZE + take]; - let found = crc32(body); - if found != header.crc { - return Err(LoadError::Corrupt { - page: index, - expected: header.crc, - found, - }); - } - - // Copied into an AlignedVec because rkyv reads an archive in place and - // needs it aligned. A page body sits at offset 24 in a Vec, which is - // aligned to nothing in particular. - let mut aligned = AlignedVec::<16>::with_capacity(take); - aligned.extend_from_slice(body); - let rows = - Vec::<(K, V)>::decode(&aligned).map_err(|NotAnArchive| LoadError::Rows { page: index })?; - if rows.len() != header.rows as usize { - return Err(LoadError::RowCount { - page: index, - expected: header.rows as usize, - found: rows.len(), - }); - } - Ok(rows) -} - -/// Every page back into one row vector. -fn from_pages(bytes: &[u8]) -> Result, LoadError> -where - Vec<(K, V)>: Codec, -{ - if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { - return Err(LoadError::NotWholePages { found: bytes.len() }); - } - - let mut schema = None; - let mut rows = Vec::new(); - for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { - rows.append_to(&mut page_rows(raw, index, &mut schema)?); - } - let expected = fingerprint::>(); - match schema { - Some(found) if found != expected => Err(LoadError::ForeignRows { found, expected }), - _ => Ok(rows), - } -} - -impl LinearTable -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - /// Every row, as pages. - #[must_use] - pub fn unload(&self) -> Vec { - to_pages(&self.rows, fingerprint::>()) - } - - /// The rows from `first` on, as pages, ready to append to a file that - /// already holds the ones before it. - /// - /// Appending is possible at all because pages stand alone: the existing - /// file is untouched and these pages are simply more of them. - #[must_use] - pub fn unload_appending(&self, first: usize) -> Vec { - let first = first.min(self.rows.len()); - to_pages(&self.rows[first..], fingerprint::>()) - } - - /// Rows back from pages. - /// - /// # Errors - /// - /// Refuses bytes that are not whole pages, a page from another version, a - /// header claiming more body than a page holds, a body that fails its - /// checksum, pages that disagree about the row type, another row type, or - /// rows that do not deserialize. - pub fn load(bytes: &[u8]) -> Result { - Ok(Self { - rows: from_pages(bytes)?, - }) - } -} - -impl IndexedTable -where - Vec<(K, V)>: Codec, - (K, V): Clone, - K: Ord + Clone, -{ - /// Every row, as pages. - /// - /// The index is not written. It is derived from the rows, so rebuilding it - /// on load costs one pass, where storing it would cost bytes at rest and a - /// second thing that can disagree with the rows. - #[must_use] - pub fn unload(&self) -> Vec { - to_pages(&self.rows, fingerprint::>()) - } - - /// The rows from `first` on, as pages. - #[must_use] - pub fn unload_appending(&self, first: usize) -> Vec { - let first = first.min(self.rows.len()); - to_pages(&self.rows[first..], fingerprint::>()) - } - - /// Rows back from pages, with the index rebuilt. - /// - /// # Errors - /// - /// As [`LinearTable::load`]. - pub fn load(bytes: &[u8]) -> Result { - let rows = from_pages(bytes)?; - let mut table = Self::with_capacity(rows.len()); - for (key, value) in rows { - let _ = table.insert(key, value); - } - Ok(table) - } -} - -mod io; -pub use io::HydrateError; - -#[cfg(test)] -mod tests; diff --git a/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/journal.json b/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/journal.json deleted file mode 100644 index 30c9612..0000000 --- a/.agentcoder/transactions/9e77492e-3cd7-4bcc-92ee-aa925ce1247c/journal.json +++ /dev/null @@ -1 +0,0 @@ -{"transaction_id":"9e77492e-3cd7-4bcc-92ee-aa925ce1247c","state":"committed","entries":[{"relative_path":"src/hydrate/mod.rs","new_revision":"abb9b81da7b7956833dd20b6d3a23f588b9a2f0c4b9e87385574b8961a1fb722"}]} \ No newline at end of file diff --git a/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/backup/src/hydrate/mod.rs b/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/backup/src/hydrate/mod.rs deleted file mode 100644 index 72ae621..0000000 --- a/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/backup/src/hydrate/mod.rs +++ /dev/null @@ -1,592 +0,0 @@ -//! Load and unload a table as pages, and put those pages on a disk. -//! -//! # The interface -//! -//! ```ignore -//! table.flush("rows.wtv")?; // write it -//! let table = LinearTable::open("rows.wtv")?; // read it back -//! table.append("rows.wtv", from_row)?; // add rows without a rewrite -//! ``` -//! -//! and the same thing without a filesystem, for callers that already hold the -//! bytes or do not have one: -//! -//! ```ignore -//! let bytes = table.unload(); -//! let table = LinearTable::load(&bytes)?; -//! ``` -//! -//! Rows live in a `Vec` while the table is in use and are pages only at rest. -//! Everything between an open and a flush runs at `Vec` speed because it *is* a -//! `Vec`. This is a codec plus a file, not a storage engine underneath. -//! -//! # Page based, and each page stands alone -//! -//! A page is 16 KiB: a 24 byte header, then an rkyv archive of **the rows that -//! fit in that page**, and nothing spanning the boundary. -//! -//! That last part is the whole design. An archive split across pages means one -//! damaged page destroys every row in the file, and it means appending a row -//! rewrites everything. Self contained pages make damage local and appends -//! O(new rows), and cost only the few bytes of archive overhead repeated per -//! page. -//! -//! # What is checked -//! -//! Every page carries a CRC-32 of its body, and every field in the header is -//! validated rather than merely written. rkyv's own validation checks that an -//! archive is structurally sound, which is not the same as checking that these -//! are the bytes that were written: a flipped bit inside a `u64` passes -//! structural validation and reads back as a different number. The checksum is -//! what catches that. -//! -//! **These are not WorkTable space files.** A WorkTable space opens with a page -//! carrying a name, a schema and a primary key list, and a `Vec<(K, V)>` -//! declares no schema to put there. - -use alloc::vec::Vec; - -use rkyv::api::high::{HighDeserializer, HighValidator}; -use rkyv::bytecheck::CheckBytes; -use rkyv::rancor::{Error as RkyvError, Strategy}; -use rkyv::ser::Serializer; -use rkyv::ser::allocator::ArenaHandle; -use rkyv::ser::sharing::Share; -use rkyv::util::AlignedVec; -use rkyv::{Archive, Deserialize, Serialize}; - -use crate::{IndexedTable, LinearTable}; - -/// One page, header included. -pub const PAGE_SIZE: usize = 4096 * 4; - -/// The fixed header every page opens with. -pub const HEADER_SIZE: usize = 24; - -/// How much of a page is body. -pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; - -/// Bumped when the page layout changes, so an older file is refused rather than -/// read through the new shape. -pub const PAGE_VERSION: u32 = 1; - -/// What a load can refuse on. -/// -/// Every variant is a statement about the bytes rather than about the caller, -/// and every one of them names the page, because a file that will not load is -/// a question about which page went wrong. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum LoadError { - /// The byte length is not a whole number of pages. - NotWholePages { - /// How many bytes arrived. - found: usize, - }, - /// A page carries a version this build does not write. - /// - /// Also what a page of zeroes looks like, which is the shape a torn write - /// leaves behind. - ForeignPages { - /// Which page, counting from zero. - page: usize, - /// The version that page claims. - version: u32, - }, - /// A header claimed a body longer than a page holds. - Overlong { - /// Which page, counting from zero. - page: usize, - /// What its header claimed. - claimed: usize, - }, - /// The body does not match the checksum written with it. - /// - /// This is the one rkyv cannot find. A flipped bit inside an integer is a - /// structurally perfect archive of the wrong number. - Corrupt { - /// Which page, counting from zero. - page: usize, - /// The checksum in the header. - expected: u32, - /// The checksum of the bytes actually there. - found: u32, - }, - /// The pages disagree with each other about the row type. - Inconsistent { - /// Which page disagreed. - page: usize, - }, - /// These are a different row type's bytes. - /// - /// Caught by a fingerprint rather than by deserialization, because - /// deserialization does not catch it: rkyv validates a `(u64, String)` - /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s - /// relative pointer as an integer. Keys look right, values are debris, and - /// nothing errors. - ForeignRows { - /// The fingerprint these bytes were written with. - found: u32, - /// The fingerprint this row type expects. - expected: u32, - }, - /// A page's rows did not deserialize. - Rows { - /// Which page, counting from zero. - page: usize, - }, - /// A page's header promised a row count its body did not contain. - RowCount { - /// Which page, counting from zero. - page: usize, - /// What the header promised. - expected: usize, - /// What the body held. - found: usize, - }, -} - -impl core::fmt::Display for LoadError { - fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::NotWholePages { found } => write!( - formatter, - "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" - ), - Self::ForeignPages { page, version } => write!( - formatter, - "page {page} is version {version}, and this build writes {PAGE_VERSION}" - ), - Self::Overlong { page, claimed } => write!( - formatter, - "page {page} claims a {claimed} byte body, more than a page holds" - ), - Self::Corrupt { - page, - expected, - found, - } => write!( - formatter, - "page {page} checksums {found:#010x} and its header says {expected:#010x}" - ), - Self::Inconsistent { page } => { - write!(formatter, "page {page} names a different row type") - } - Self::ForeignRows { found, expected } => write!( - formatter, - "these are row type {found:#010x}, and this is row type {expected:#010x}" - ), - Self::Rows { page } => write!(formatter, "page {page} did not deserialize"), - Self::RowCount { - page, - expected, - found, - } => write!( - formatter, - "page {page} promised {expected} rows and held {found}" - ), - } - } -} - -impl core::error::Error for LoadError {} - -/// What a row set has to be able to do to make the trip. -/// -/// The bounds are rkyv's and there are five lines of them, so they are stated -/// once here and every signature below asks only for `Codec`. The blanket impl -/// means a caller never names this trait either: any row pair whose key and -/// value already derive rkyv's traits satisfies it. -/// The one thing [`Codec::decode`] can say. Which page it happened on is the -/// caller's to add, because a codec does not know it is reading a page. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub struct NotAnArchive; - -pub trait Codec: Sized { - /// Rows to bytes. - fn encode(&self) -> AlignedVec<16>; - /// Bytes back to rows. - /// - /// # Errors - /// - /// Fails when the bytes are not this type's archive. - fn decode(bytes: &[u8]) -> Result; -} - -impl Codec for T -where - T: Archive - + for<'a> Serialize, ArenaHandle<'a>, Share>, RkyvError>>, - ::Archived: Deserialize> - + for<'a> CheckBytes>, -{ - fn encode(&self) -> AlignedVec<16> { - // Infallible in practice: the only failure rkyv reports here is an - // allocator refusing, which on this path means the process is already - // out of memory. - rkyv::to_bytes::(self).expect("rows serialize") - } - - fn decode(bytes: &[u8]) -> Result { - rkyv::from_bytes::(bytes).map_err(|_| NotAnArchive) - } -} - -/// What row type wrote these bytes. -/// -/// FNV-1a over `core::any::type_name`, which is neither stable across compiler -/// versions nor guaranteed unique. That is fine for what it is for: refusing an -/// obvious mismatch, not authenticating a schema. A false match is possible and -/// a false mismatch is a rebuild, so it fails toward refusing to load rather -/// than toward reinterpreting. -pub(crate) fn fingerprint() -> u32 { - let mut hash: u32 = 0x811c_9dc5; - for byte in core::any::type_name::().as_bytes() { - hash ^= u32::from(*byte); - hash = hash.wrapping_mul(0x0100_0193); - } - hash -} - -/// CRC-32, the usual reversed polynomial, computed a nibble at a time. -/// -/// Sixteen entries rather than a 256 entry table: this runs once per 16 KiB -/// page, so the table is cache noise and the loop is not the cost of anything. -fn crc32(bytes: &[u8]) -> u32 { - const NIBBLE: [u32; 16] = [ - 0x0000_0000, - 0x1db7_1064, - 0x3b6e_20c8, - 0x26d9_30ac, - 0x76dc_4190, - 0x6b6b_51f4, - 0x4db2_6158, - 0x5005_713c, - 0xedb8_8320, - 0xf00f_9344, - 0xd6d6_a3e8, - 0xcb61_b38c, - 0x9b64_c2b0, - 0x86d3_d2d4, - 0xa00a_e278, - 0xbdbd_f21c, - ]; - let mut crc = 0xffff_ffffu32; - for byte in bytes { - crc ^= u32::from(*byte); - crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; - crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; - } - !crc -} - -/// A page header: six little-endian `u32`s, in this order. -/// -/// Every one of them is checked on the way back in. A field that is written and -/// never validated is worse than a field that does not exist, because it reads -/// like a guarantee. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub(crate) struct Header { - version: u32, - /// The row type every page in a run carries. - schema: u32, - /// How many rows this page's archive holds. - rows: u32, - /// How many bytes of that archive are in this page. - body: u32, - /// CRC-32 over exactly `body` bytes. - crc: u32, - /// Zero for now. A layout change that needs a flag has somewhere to put it - /// without moving anything else. - flags: u32, -} - -impl Header { - fn write(self, out: &mut Vec) { - for field in [ - self.version, - self.schema, - self.rows, - self.body, - self.crc, - self.flags, - ] { - out.extend_from_slice(&field.to_le_bytes()); - } - } - - fn read(raw: &[u8]) -> Self { - let at = |n: usize| { - let mut word = [0u8; 4]; - word.copy_from_slice(&raw[n * 4..n * 4 + 4]); - u32::from_le_bytes(word) - }; - Self { - version: at(0), - schema: at(1), - rows: at(2), - body: at(3), - crc: at(4), - flags: at(5), - } - } -} - -/// The most rows of `rows` whose archive fits one page body. -/// -/// **Bounded probes.** The obvious version binary searches over the whole -/// remaining slice, which re-serializes every row still to be written on -/// every probe, for every page. That measured 2.3 seconds to write what rkyv -/// alone encodes in 6.7 ms, because the work is quadratic in the row count. -/// -/// So the search is bounded to roughly two pages of rows: one sample encode -/// gives bytes per row, the estimate from that sets the ceiling, and the -/// binary search runs under it. Every probe serializes about a page, never a -/// file. Uniform rows land in a probe or two and wildly variable rows still -/// terminate, because the ceiling is only a ceiling. -/// -/// Always returns at least one for a non-empty slice, so the caller always -/// makes progress. A single row too large for a page is written as an -/// oversized page rather than looping forever. -fn rows_per_page(rows: &[(K, V)], hint: usize) -> usize -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - if rows.is_empty() { - return 0; - } - - let fits = |take: usize| rows[..take].to_vec().encode().len() <= BODY_SIZE; - - // A page holds about what the last one held, so start there and walk. - // Uniform rows settle in a probe or two; only the first page, or a run - // whose rows change size, pays for a search. - if hint > 0 && hint <= rows.len() && fits(hint) { - let mut take = hint; - while take < rows.len() && fits(take + 1) { - take += 1; - } - return take; - } - - // No usable hint, or the rows grew. One sample gives bytes per row, and - // the estimate from it bounds the search to about two pages of rows. - let sample = rows.len().min(64); - let sampled = rows[..sample].to_vec().encode().len(); - let estimate = (BODY_SIZE * sample) - .checked_div(sampled) - .map_or(rows.len(), |estimate| estimate.max(1)); - let mut low = 1usize; - let mut high = rows.len().min(estimate.saturating_mul(2)).max(1); - while low < high { - let mid = low + (high - low).div_ceil(2); - if fits(mid) { - low = mid; - } else { - high = mid - 1; - } - } - low -} - -/// Rows to pages, each page standing alone. -pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Vec -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - let mut out = Vec::new(); - let mut rest = rows; - let mut hint = 0usize; - - // An empty table still writes one page. A zero byte file is - // indistinguishable from a missing one, and a load has to tell "no rows" - // from "nothing landed". - loop { - // Zero for an empty table, which still writes its one page and stops. - // At least one for anything else, so this always makes progress. - let take = rows_per_page(rest, hint); - hint = take; - let archive = rest[..take].to_vec().encode(); - let body = archive.as_ref(); - Header { - version: PAGE_VERSION, - schema, - rows: u32::try_from(take).expect("a row count inside u32"), - body: u32::try_from(body.len()).expect("a body inside u32"), - crc: crc32(body), - flags: 0, - } - .unload_to(&mut out); - out.extend_from_slice(body); - out.resize(out.len().next_multiple_of(PAGE_SIZE), 0); - - rest = &rest[take..]; - if rest.is_empty() { - break; - } - } - out -} - -/// One page back into rows, with every header field checked. -pub(crate) fn page_rows( - raw: &[u8], - index: usize, - schema: &mut Option, -) -> Result, LoadError> -where - Vec<(K, V)>: Codec, -{ - let header = Header::read(&raw[..HEADER_SIZE]); - if header.version != PAGE_VERSION { - return Err(LoadError::ForeignPages { - page: index, - version: header.version, - }); - } - match schema { - None => *schema = Some(header.schema), - // Every page names the row type, so a file spliced onto another is - // caught where they stop agreeing rather than concatenated. - Some(first) if *first != header.schema => { - return Err(LoadError::Inconsistent { page: index }); - } - Some(_) => {} - } - - let take = header.body as usize; - if take > BODY_SIZE { - return Err(LoadError::Overlong { - page: index, - claimed: take, - }); - } - let body = &raw[HEADER_SIZE..HEADER_SIZE + take]; - let found = crc32(body); - if found != header.crc { - return Err(LoadError::Corrupt { - page: index, - expected: header.crc, - found, - }); - } - - // Copied into an AlignedVec because rkyv reads an archive in place and - // needs it aligned. A page body sits at offset 24 in a Vec, which is - // aligned to nothing in particular. - let mut aligned = AlignedVec::<16>::with_capacity(take); - aligned.extend_from_slice(body); - let rows = - Vec::<(K, V)>::decode(&aligned).map_err(|NotAnArchive| LoadError::Rows { page: index })?; - if rows.len() != header.rows as usize { - return Err(LoadError::RowCount { - page: index, - expected: header.rows as usize, - found: rows.len(), - }); - } - Ok(rows) -} - -/// Every page back into one row vector. -fn from_pages(bytes: &[u8]) -> Result, LoadError> -where - Vec<(K, V)>: Codec, -{ - if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { - return Err(LoadError::NotWholePages { found: bytes.len() }); - } - - let mut schema = None; - let mut rows = Vec::new(); - for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { - rows.append_to(&mut page_rows(raw, index, &mut schema)?); - } - let expected = fingerprint::>(); - match schema { - Some(found) if found != expected => Err(LoadError::ForeignRows { found, expected }), - _ => Ok(rows), - } -} - -impl LinearTable -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - /// Every row, as pages. - #[must_use] - pub fn unload(&self) -> Vec { - to_pages(&self.rows, fingerprint::>()) - } - - /// The rows from `first` on, as pages, ready to append to a file that - /// already holds the ones before it. - /// - /// Appending is possible at all because pages stand alone: the existing - /// file is untouched and these pages are simply more of them. - #[must_use] - pub fn unload_appending(&self, first: usize) -> Vec { - let first = first.min(self.rows.len()); - to_pages(&self.rows[first..], fingerprint::>()) - } - - /// Rows back from pages. - /// - /// # Errors - /// - /// Refuses bytes that are not whole pages, a page from another version, a - /// header claiming more body than a page holds, a body that fails its - /// checksum, pages that disagree about the row type, another row type, or - /// rows that do not deserialize. - pub fn load(bytes: &[u8]) -> Result { - Ok(Self { - rows: from_pages(bytes)?, - }) - } -} - -impl IndexedTable -where - Vec<(K, V)>: Codec, - (K, V): Clone, - K: Ord + Clone, -{ - /// Every row, as pages. - /// - /// The index is not written. It is derived from the rows, so rebuilding it - /// on load costs one pass, where storing it would cost bytes at rest and a - /// second thing that can disagree with the rows. - #[must_use] - pub fn unload(&self) -> Vec { - to_pages(&self.rows, fingerprint::>()) - } - - /// The rows from `first` on, as pages. - #[must_use] - pub fn unload_appending(&self, first: usize) -> Vec { - let first = first.min(self.rows.len()); - to_pages(&self.rows[first..], fingerprint::>()) - } - - /// Rows back from pages, with the index rebuilt. - /// - /// # Errors - /// - /// As [`LinearTable::load`]. - pub fn load(bytes: &[u8]) -> Result { - let rows = from_pages(bytes)?; - let mut table = Self::with_capacity(rows.len()); - for (key, value) in rows { - let _ = table.insert(key, value); - } - Ok(table) - } -} - -mod io; -pub use io::HydrateError; - -#[cfg(test)] -mod tests; diff --git a/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/journal.json b/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/journal.json deleted file mode 100644 index 437a91b..0000000 --- a/.agentcoder/transactions/e45443a5-bf32-48ae-9e52-c62b7f4a2732/journal.json +++ /dev/null @@ -1 +0,0 @@ -{"transaction_id":"e45443a5-bf32-48ae-9e52-c62b7f4a2732","state":"committed","entries":[{"relative_path":"src/hydrate/mod.rs","new_revision":"1bc3d272a28040188c663bf061a1e5661df6eb555b2cd512e41b3b6917cde64d"}]} \ No newline at end of file diff --git a/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/backup/src/hydrate/mod.rs b/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/backup/src/hydrate/mod.rs deleted file mode 100644 index 0c733e3..0000000 --- a/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/backup/src/hydrate/mod.rs +++ /dev/null @@ -1,592 +0,0 @@ -//! Load and unload a table as pages, and put those pages on a disk. -//! -//! # The interface -//! -//! ```ignore -//! table.flush("rows.wtv")?; // write it -//! let table = LinearTable::open("rows.wtv")?; // read it back -//! table.append("rows.wtv", from_row)?; // add rows without a rewrite -//! ``` -//! -//! and the same thing without a filesystem, for callers that already hold the -//! bytes or do not have one: -//! -//! ```ignore -//! let bytes = table.unload(); -//! let table = LinearTable::load(&bytes)?; -//! ``` -//! -//! Rows live in a `Vec` while the table is in use and are pages only at rest. -//! Everything between an open and a flush runs at `Vec` speed because it *is* a -//! `Vec`. This is a codec plus a file, not a storage engine underneath. -//! -//! # Page based, and each page stands alone -//! -//! A page is 16 KiB: a 24 byte header, then an rkyv archive of **the rows that -//! fit in that page**, and nothing spanning the boundary. -//! -//! That last part is the whole design. An archive split across pages means one -//! damaged page destroys every row in the file, and it means appending a row -//! rewrites everything. Self contained pages make damage local and appends -//! O(new rows), and cost only the few bytes of archive overhead repeated per -//! page. -//! -//! # What is checked -//! -//! Every page carries a CRC-32 of its body, and every field in the header is -//! validated rather than merely written. rkyv's own validation checks that an -//! archive is structurally sound, which is not the same as checking that these -//! are the bytes that were written: a flipped bit inside a `u64` passes -//! structural validation and reads back as a different number. The checksum is -//! what catches that. -//! -//! **These are not WorkTable space files.** A WorkTable space opens with a page -//! carrying a name, a schema and a primary key list, and a `Vec<(K, V)>` -//! declares no schema to put there. - -use alloc::vec::Vec; - -use rkyv::api::high::{HighDeserializer, HighValidator}; -use rkyv::bytecheck::CheckBytes; -use rkyv::rancor::{Error as RkyvError, Strategy}; -use rkyv::ser::Serializer; -use rkyv::ser::allocator::ArenaHandle; -use rkyv::ser::sharing::Share; -use rkyv::util::AlignedVec; -use rkyv::{Archive, Deserialize, Serialize}; - -use crate::{IndexedTable, LinearTable}; - -/// One page, header included. -pub const PAGE_SIZE: usize = 4096 * 4; - -/// The fixed header every page opens with. -pub const HEADER_SIZE: usize = 24; - -/// How much of a page is body. -pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; - -/// Bumped when the page layout changes, so an older file is refused rather than -/// read through the new shape. -pub const PAGE_VERSION: u32 = 1; - -/// What a load can refuse on. -/// -/// Every variant is a statement about the bytes rather than about the caller, -/// and every one of them names the page, because a file that will not load is -/// a question about which page went wrong. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum LoadError { - /// The byte length is not a whole number of pages. - NotWholePages { - /// How many bytes arrived. - found: usize, - }, - /// A page carries a version this build does not write. - /// - /// Also what a page of zeroes looks like, which is the shape a torn write - /// leaves behind. - ForeignPages { - /// Which page, counting from zero. - page: usize, - /// The version that page claims. - version: u32, - }, - /// A header claimed a body longer than a page holds. - Overlong { - /// Which page, counting from zero. - page: usize, - /// What its header claimed. - claimed: usize, - }, - /// The body does not match the checksum written with it. - /// - /// This is the one rkyv cannot find. A flipped bit inside an integer is a - /// structurally perfect archive of the wrong number. - Corrupt { - /// Which page, counting from zero. - page: usize, - /// The checksum in the header. - expected: u32, - /// The checksum of the bytes actually there. - found: u32, - }, - /// The pages disagree with each other about the row type. - Inconsistent { - /// Which page disagreed. - page: usize, - }, - /// These are a different row type's bytes. - /// - /// Caught by a fingerprint rather than by deserialization, because - /// deserialization does not catch it: rkyv validates a `(u64, String)` - /// archive as a perfectly good `(u64, u64)` and hands back a `String`'s - /// relative pointer as an integer. Keys look right, values are debris, and - /// nothing errors. - ForeignRows { - /// The fingerprint these bytes were written with. - found: u32, - /// The fingerprint this row type expects. - expected: u32, - }, - /// A page's rows did not deserialize. - Rows { - /// Which page, counting from zero. - page: usize, - }, - /// A page's header promised a row count its body did not contain. - RowCount { - /// Which page, counting from zero. - page: usize, - /// What the header promised. - expected: usize, - /// What the body held. - found: usize, - }, -} - -impl core::fmt::Display for LoadError { - fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::NotWholePages { found } => write!( - formatter, - "{found} bytes is not a whole number of {PAGE_SIZE} byte pages" - ), - Self::ForeignPages { page, version } => write!( - formatter, - "page {page} is version {version}, and this build writes {PAGE_VERSION}" - ), - Self::Overlong { page, claimed } => write!( - formatter, - "page {page} claims a {claimed} byte body, more than a page holds" - ), - Self::Corrupt { - page, - expected, - found, - } => write!( - formatter, - "page {page} checksums {found:#010x} and its header says {expected:#010x}" - ), - Self::Inconsistent { page } => { - write!(formatter, "page {page} names a different row type") - } - Self::ForeignRows { found, expected } => write!( - formatter, - "these are row type {found:#010x}, and this is row type {expected:#010x}" - ), - Self::Rows { page } => write!(formatter, "page {page} did not deserialize"), - Self::RowCount { - page, - expected, - found, - } => write!( - formatter, - "page {page} promised {expected} rows and held {found}" - ), - } - } -} - -impl core::error::Error for LoadError {} - -/// What a row set has to be able to do to make the trip. -/// -/// The bounds are rkyv's and there are five lines of them, so they are stated -/// once here and every signature below asks only for `Codec`. The blanket impl -/// means a caller never names this trait either: any row pair whose key and -/// value already derive rkyv's traits satisfies it. -/// The one thing [`Codec::decode`] can say. Which page it happened on is the -/// caller's to add, because a codec does not know it is reading a page. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub struct NotAnArchive; - -pub trait Codec: Sized { - /// Rows to bytes. - fn encode(&self) -> AlignedVec<16>; - /// Bytes back to rows. - /// - /// # Errors - /// - /// Fails when the bytes are not this type's archive. - fn decode(bytes: &[u8]) -> Result; -} - -impl Codec for T -where - T: Archive - + for<'a> Serialize, ArenaHandle<'a>, Share>, RkyvError>>, - ::Archived: Deserialize> - + for<'a> CheckBytes>, -{ - fn encode(&self) -> AlignedVec<16> { - // Infallible in practice: the only failure rkyv reports here is an - // allocator refusing, which on this path means the process is already - // out of memory. - rkyv::to_bytes::(self).expect("rows serialize") - } - - fn decode(bytes: &[u8]) -> Result { - rkyv::from_bytes::(bytes).map_err(|_| NotAnArchive) - } -} - -/// What row type wrote these bytes. -/// -/// FNV-1a over `core::any::type_name`, which is neither stable across compiler -/// versions nor guaranteed unique. That is fine for what it is for: refusing an -/// obvious mismatch, not authenticating a schema. A false match is possible and -/// a false mismatch is a rebuild, so it fails toward refusing to load rather -/// than toward reinterpreting. -pub(crate) fn fingerprint() -> u32 { - let mut hash: u32 = 0x811c_9dc5; - for byte in core::any::type_name::().as_bytes() { - hash ^= u32::from(*byte); - hash = hash.wrapping_mul(0x0100_0193); - } - hash -} - -/// CRC-32, the usual reversed polynomial, computed a nibble at a time. -/// -/// Sixteen entries rather than a 256 entry table: this runs once per 16 KiB -/// page, so the table is cache noise and the loop is not the cost of anything. -fn crc32(bytes: &[u8]) -> u32 { - const NIBBLE: [u32; 16] = [ - 0x0000_0000, - 0x1db7_1064, - 0x3b6e_20c8, - 0x26d9_30ac, - 0x76dc_4190, - 0x6b6b_51f4, - 0x4db2_6158, - 0x5005_713c, - 0xedb8_8320, - 0xf00f_9344, - 0xd6d6_a3e8, - 0xcb61_b38c, - 0x9b64_c2b0, - 0x86d3_d2d4, - 0xa00a_e278, - 0xbdbd_f21c, - ]; - let mut crc = 0xffff_ffffu32; - for byte in bytes { - crc ^= u32::from(*byte); - crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; - crc = (crc >> 4) ^ NIBBLE[(crc & 0x0f) as usize]; - } - !crc -} - -/// A page header: six little-endian `u32`s, in this order. -/// -/// Every one of them is checked on the way back in. A field that is written and -/// never validated is worse than a field that does not exist, because it reads -/// like a guarantee. -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub(crate) struct Header { - version: u32, - /// The row type every page in a run carries. - schema: u32, - /// How many rows this page's archive holds. - rows: u32, - /// How many bytes of that archive are in this page. - body: u32, - /// CRC-32 over exactly `body` bytes. - crc: u32, - /// Zero for now. A layout change that needs a flag has somewhere to put it - /// without moving anything else. - flags: u32, -} - -impl Header { - fn write(self, out: &mut Vec) { - for field in [ - self.version, - self.schema, - self.rows, - self.body, - self.crc, - self.flags, - ] { - out.extend_from_slice(&field.to_le_bytes()); - } - } - - fn read(raw: &[u8]) -> Self { - let at = |n: usize| { - let mut word = [0u8; 4]; - word.copy_from_slice(&raw[n * 4..n * 4 + 4]); - u32::from_le_bytes(word) - }; - Self { - version: at(0), - schema: at(1), - rows: at(2), - body: at(3), - crc: at(4), - flags: at(5), - } - } -} - -/// The most rows of `rows` whose archive fits one page body. -/// -/// **Bounded probes.** The obvious version binary searches over the whole -/// remaining slice, which re-serializes every row still to be written on -/// every probe, for every page. That measured 2.3 seconds to write what rkyv -/// alone encodes in 6.7 ms, because the work is quadratic in the row count. -/// -/// So the search is bounded to roughly two pages of rows: one sample encode -/// gives bytes per row, the estimate from that sets the ceiling, and the -/// binary search runs under it. Every probe serializes about a page, never a -/// file. Uniform rows land in a probe or two and wildly variable rows still -/// terminate, because the ceiling is only a ceiling. -/// -/// Always returns at least one for a non-empty slice, so the caller always -/// makes progress. A single row too large for a page is written as an -/// oversized page rather than looping forever. -fn rows_per_page(rows: &[(K, V)], hint: usize) -> usize -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - if rows.is_empty() { - return 0; - } - - let fits = |take: usize| rows[..take].to_vec().encode().len() <= BODY_SIZE; - - // A page holds about what the last one held, so start there and walk. - // Uniform rows settle in a probe or two; only the first page, or a run - // whose rows change size, pays for a search. - if hint > 0 && hint <= rows.len() && fits(hint) { - let mut take = hint; - while take < rows.len() && fits(take + 1) { - take += 1; - } - return take; - } - - // No usable hint, or the rows grew. One sample gives bytes per row, and - // the estimate from it bounds the search to about two pages of rows. - let sample = rows.len().min(64); - let sampled = rows[..sample].to_vec().encode().len(); - let estimate = (BODY_SIZE * sample) - .checked_div(sampled) - .map_or(rows.len(), |estimate| estimate.max(1)); - let mut low = 1usize; - let mut high = rows.len().min(estimate.saturating_mul(2)).max(1); - while low < high { - let mid = low + (high - low).div_ceil(2); - if fits(mid) { - low = mid; - } else { - high = mid - 1; - } - } - low -} - -/// Rows to pages, each page standing alone. -pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Vec -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - let mut out = Vec::new(); - let mut rest = rows; - let mut hint = 0usize; - - // An empty table still writes one page. A zero byte file is - // indistinguishable from a missing one, and a load has to tell "no rows" - // from "nothing landed". - loop { - // Zero for an empty table, which still writes its one page and stops. - // At least one for anything else, so this always makes progress. - let take = rows_per_page(rest, hint); - hint = take; - let archive = rest[..take].to_vec().encode(); - let body = archive.as_ref(); - Header { - version: PAGE_VERSION, - schema, - rows: u32::try_from(take).expect("a row count inside u32"), - body: u32::try_from(body.len()).expect("a body inside u32"), - crc: crc32(body), - flags: 0, - } - .write(&mut out); - out.extend_from_slice(body); - out.resize(out.len().next_multiple_of(PAGE_SIZE), 0); - - rest = &rest[take..]; - if rest.is_empty() { - break; - } - } - out -} - -/// One page back into rows, with every header field checked. -pub(crate) fn page_rows( - raw: &[u8], - index: usize, - schema: &mut Option, -) -> Result, LoadError> -where - Vec<(K, V)>: Codec, -{ - let header = Header::read(&raw[..HEADER_SIZE]); - if header.version != PAGE_VERSION { - return Err(LoadError::ForeignPages { - page: index, - version: header.version, - }); - } - match schema { - None => *schema = Some(header.schema), - // Every page names the row type, so a file spliced onto another is - // caught where they stop agreeing rather than concatenated. - Some(first) if *first != header.schema => { - return Err(LoadError::Inconsistent { page: index }); - } - Some(_) => {} - } - - let take = header.body as usize; - if take > BODY_SIZE { - return Err(LoadError::Overlong { - page: index, - claimed: take, - }); - } - let body = &raw[HEADER_SIZE..HEADER_SIZE + take]; - let found = crc32(body); - if found != header.crc { - return Err(LoadError::Corrupt { - page: index, - expected: header.crc, - found, - }); - } - - // Copied into an AlignedVec because rkyv reads an archive in place and - // needs it aligned. A page body sits at offset 24 in a Vec, which is - // aligned to nothing in particular. - let mut aligned = AlignedVec::<16>::with_capacity(take); - aligned.extend_from_slice(body); - let rows = - Vec::<(K, V)>::decode(&aligned).map_err(|NotAnArchive| LoadError::Rows { page: index })?; - if rows.len() != header.rows as usize { - return Err(LoadError::RowCount { - page: index, - expected: header.rows as usize, - found: rows.len(), - }); - } - Ok(rows) -} - -/// Every page back into one row vector. -fn from_pages(bytes: &[u8]) -> Result, LoadError> -where - Vec<(K, V)>: Codec, -{ - if bytes.is_empty() || bytes.len() % PAGE_SIZE != 0 { - return Err(LoadError::NotWholePages { found: bytes.len() }); - } - - let mut schema = None; - let mut rows = Vec::new(); - for (index, raw) in bytes.chunks_exact(PAGE_SIZE).enumerate() { - rows.append(&mut page_rows(raw, index, &mut schema)?); - } - let expected = fingerprint::>(); - match schema { - Some(found) if found != expected => Err(LoadError::ForeignRows { found, expected }), - _ => Ok(rows), - } -} - -impl LinearTable -where - Vec<(K, V)>: Codec, - (K, V): Clone, -{ - /// Every row, as pages. - #[must_use] - pub fn unload(&self) -> Vec { - to_pages(&self.rows, fingerprint::>()) - } - - /// The rows from `first` on, as pages, ready to append to a file that - /// already holds the ones before it. - /// - /// Appending is possible at all because pages stand alone: the existing - /// file is untouched and these pages are simply more of them. - #[must_use] - pub fn unload_appending(&self, first: usize) -> Vec { - let first = first.min(self.rows.len()); - to_pages(&self.rows[first..], fingerprint::>()) - } - - /// Rows back from pages. - /// - /// # Errors - /// - /// Refuses bytes that are not whole pages, a page from another version, a - /// header claiming more body than a page holds, a body that fails its - /// checksum, pages that disagree about the row type, another row type, or - /// rows that do not deserialize. - pub fn load(bytes: &[u8]) -> Result { - Ok(Self { - rows: from_pages(bytes)?, - }) - } -} - -impl IndexedTable -where - Vec<(K, V)>: Codec, - (K, V): Clone, - K: Ord + Clone, -{ - /// Every row, as pages. - /// - /// The index is not written. It is derived from the rows, so rebuilding it - /// on load costs one pass, where storing it would cost bytes at rest and a - /// second thing that can disagree with the rows. - #[must_use] - pub fn unload(&self) -> Vec { - to_pages(&self.rows, fingerprint::>()) - } - - /// The rows from `first` on, as pages. - #[must_use] - pub fn unload_appending(&self, first: usize) -> Vec { - let first = first.min(self.rows.len()); - to_pages(&self.rows[first..], fingerprint::>()) - } - - /// Rows back from pages, with the index rebuilt. - /// - /// # Errors - /// - /// As [`LinearTable::load`]. - pub fn load(bytes: &[u8]) -> Result { - let rows = from_pages(bytes)?; - let mut table = Self::with_capacity(rows.len()); - for (key, value) in rows { - let _ = table.insert(key, value); - } - Ok(table) - } -} - -mod io; -pub use io::HydrateError; - -#[cfg(test)] -mod tests; diff --git a/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/journal.json b/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/journal.json deleted file mode 100644 index 6d997e0..0000000 --- a/.agentcoder/transactions/f0675041-5387-4d5f-bbea-b03fd7d79425/journal.json +++ /dev/null @@ -1 +0,0 @@ -{"transaction_id":"f0675041-5387-4d5f-bbea-b03fd7d79425","state":"committed","entries":[{"relative_path":"src/hydrate/mod.rs","new_revision":"7d87cdb31dfa04a16f5790d8b61f3dff895b4da527f47e5e3ce11750cfeeff74"}]} \ No newline at end of file diff --git a/src/hydrate/mod.rs b/src/hydrate/mod.rs index 0157421..4e10080 100644 --- a/src/hydrate/mod.rs +++ b/src/hydrate/mod.rs @@ -66,15 +66,23 @@ use crate::{IndexedTable, LinearTable}; /// One page, header included. pub const PAGE_SIZE: usize = 4096 * 4; -/// The fixed header every page opens with. -pub const HEADER_SIZE: usize = 24; +/// DataBucket's `GENERAL_HEADER_SIZE`, which this page opens with. +pub const HEADER_SIZE: usize = 28; -/// How much of a page is body. -pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE; +/// The row directory at the page tail: a row count and a CRC-32. +pub const DIRECTORY_SIZE: usize = 8; -/// Bumped when the page layout changes, so an older file is refused rather than -/// read through the new shape. -pub const PAGE_VERSION: u32 = 1; +/// How much of a page is body, between the header and the directory. +pub const BODY_SIZE: usize = PAGE_SIZE - HEADER_SIZE - DIRECTORY_SIZE; + +/// `DATA_VERSION` 3: DataBucket's page framing, plus a row directory. +/// +/// 2 is what WorkTable writes today, and a 2 page has no directory, so a reader +/// cannot find its rows without the index. 3 says the directory is there. +pub const PAGE_VERSION: u32 = 3; + +/// `PageType::Data` in DataBucket's enum. +const PAGE_TYPE_DATA: u32 = 2; /// What a load can refuse on. /// @@ -285,25 +293,42 @@ fn crc32(bytes: &[u8]) -> u32 { !crc } -/// A page header: six little-endian `u32`s, in this order. +/// DataBucket's `GeneralHeader`, byte for byte. +/// +/// Seven little-endian `u32`s in declaration order, which is what +/// `rkyv::to_bytes` of that struct actually produces: no relative pointers, and +/// `page_type` padded from `u16` to four bytes. Verified against +/// `data_bucket 0.5.7`, which for a `Data` page of space 3, id 7, previous 6, +/// next 8, length `0x11223344` emits: /// -/// Every one of them is checked on the way back in. A field that is written and -/// never validated is worse than a field that does not exist, because it reads -/// like a guarantee. +/// ```text +/// 02000000 03000000 07000000 06000000 08000000 02000000 44332211 +/// version space page previous next type length +/// ``` +/// +/// It is reproduced here rather than imported because `data_bucket` is `std` +/// (tokio for file access, eyre through its signatures) and this crate is not. +/// **That is a real duplication and the risk that comes with it is a layout +/// drifting apart in two places**, which is why the bytes above are written +/// down and `the_header_matches_databuckets_layout` checks them. #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub(crate) struct Header { + /// `DATA_VERSION`. See [`PAGE_VERSION`]. version: u32, - /// The row type every page in a run carries. + /// Carries the row type fingerprint rather than a space id. + /// + /// A `Vec<(K, V)>` belongs to no space, and the field is a `u32` sitting in + /// the right place, so it holds the one identity these pages do have. A + /// WorkTable reader will see a space id it does not recognise, which is the + /// honest outcome: these are not its rows. schema: u32, - /// How many rows this page's archive holds. - rows: u32, - /// How many bytes of that archive are in this page. + page: u32, + previous: u32, + next: u32, + /// `PageType::Data`, which is 2. + page_type: u32, + /// Bytes of row archive in this page, before the directory. body: u32, - /// CRC-32 over exactly `body` bytes. - crc: u32, - /// Zero for now. A layout change that needs a flag has somewhere to put it - /// without moving anything else. - flags: u32, } impl Header { @@ -311,10 +336,11 @@ impl Header { for field in [ self.version, self.schema, - self.rows, + self.page, + self.previous, + self.next, + self.page_type, self.body, - self.crc, - self.flags, ] { out.extend_from_slice(&field.to_le_bytes()); } @@ -329,10 +355,49 @@ impl Header { Self { version: at(0), schema: at(1), - rows: at(2), - body: at(3), - crc: at(4), - flags: at(5), + page: at(2), + previous: at(3), + next: at(4), + page_type: at(5), + body: at(6), + } + } +} + +/// The row directory, at the tail of every page. +/// +/// **This is the slotted part.** The header is DataBucket's and has nowhere to +/// say how many rows a page holds, which is exactly the gap that makes a +/// WorkTable data page unreadable without its index. Putting the count in the +/// page means the page describes itself. +/// +/// Eight bytes at the very end: the row count, then a CRC-32 of the body. The +/// checksum lives here rather than in the header for the same reason the count +/// does, and because a header this crate did not design has no spare field for +/// it. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(crate) struct Directory { + rows: u32, + crc: u32, +} + +impl Directory { + fn write(self, page: &mut [u8]) { + let at = page.len() - DIRECTORY_SIZE; + page[at..at + 4].copy_from_slice(&self.rows.to_le_bytes()); + page[at + 4..].copy_from_slice(&self.crc.to_le_bytes()); + } + + fn read(page: &[u8]) -> Self { + let at = page.len() - DIRECTORY_SIZE; + let word = |n: usize| { + let mut bytes = [0u8; 4]; + bytes.copy_from_slice(&page[n..n + 4]); + u32::from_le_bytes(bytes) + }; + Self { + rows: word(at), + crc: word(at + 4), } } } @@ -415,18 +480,31 @@ where hint = take; let archive = rest[..take].to_vec().encode(); let body = archive.as_ref(); + let page = u32::try_from(out.len() / PAGE_SIZE).expect("a page index inside u32"); + let last = rest.len() == take; Header { version: PAGE_VERSION, schema, - rows: u32::try_from(take).expect("a row count inside u32"), + page, + previous: page.saturating_sub(1), + // A last page points at itself, so a chain walker stops rather than + // running off the end. + next: if last { page } else { page + 1 }, + page_type: PAGE_TYPE_DATA, body: u32::try_from(body.len()).expect("a body inside u32"), - crc: crc32(body), - flags: 0, } .write(&mut out); out.extend_from_slice(body); out.resize(out.len().next_multiple_of(PAGE_SIZE), 0); + // The directory goes in last, into the tail of the page just written. + let start = out.len() - PAGE_SIZE; + Directory { + rows: u32::try_from(take).expect("a row count inside u32"), + crc: crc32(body), + } + .write(&mut out[start..]); + rest = &rest[take..]; if rest.is_empty() { break; @@ -468,27 +546,28 @@ where claimed: take, }); } + let directory = Directory::read(raw); let body = &raw[HEADER_SIZE..HEADER_SIZE + take]; let found = crc32(body); - if found != header.crc { + if found != directory.crc { return Err(LoadError::Corrupt { page: index, - expected: header.crc, + expected: directory.crc, found, }); } // Copied into an AlignedVec because rkyv reads an archive in place and - // needs it aligned. A page body sits at offset 24 in a Vec, which is - // aligned to nothing in particular. + // needs it aligned. A page body sits at a header's offset into a Vec, + // which is aligned to nothing in particular. let mut aligned = AlignedVec::<16>::with_capacity(take); aligned.extend_from_slice(body); let rows = Vec::<(K, V)>::decode(&aligned).map_err(|NotAnArchive| LoadError::Rows { page: index })?; - if rows.len() != header.rows as usize { + if rows.len() != directory.rows as usize { return Err(LoadError::RowCount { page: index, - expected: header.rows as usize, + expected: directory.rows as usize, found: rows.len(), }); } diff --git a/src/hydrate/tests.rs b/src/hydrate/tests.rs index f28582c..65a1603 100644 --- a/src/hydrate/tests.rs +++ b/src/hydrate/tests.rs @@ -135,9 +135,12 @@ fn damage_stays_inside_the_page_it_happened_to() { #[test] fn a_row_count_that_disagrees_with_the_body_is_refused() { let mut bytes = table(64).unload(); - // Claim one more row than the body holds, and fix nothing else. - let rows = u32::from_le_bytes([bytes[8], bytes[9], bytes[10], bytes[11]]); - bytes[8..12].copy_from_slice(&(rows + 1).to_le_bytes()); + // Claim one more row than the body holds, and fix nothing else. The count + // is in the directory at the page tail, not in the header, because the + // header is DataBucket's and has no field for it. + let at = PAGE_SIZE - DIRECTORY_SIZE; + let rows = u32::from_le_bytes([bytes[at], bytes[at + 1], bytes[at + 2], bytes[at + 3]]); + bytes[at..at + 4].copy_from_slice(&(rows + 1).to_le_bytes()); match LinearTable::::load(&bytes) { Err(LoadError::RowCount { page: 0, .. }) => {} other => panic!("a lying row count has to be caught: {other:?}"), @@ -261,3 +264,78 @@ mod through_a_reader_and_a_writer { } } } + +/// The header is DataBucket's `GeneralHeader`, byte for byte. +/// +/// Verified against `data_bucket 0.5.7`, which for a `Data` page of space 3, +/// id 7, previous 6, next 8, length `0x11223344` emits exactly these 28 bytes. +/// This crate reproduces that layout rather than importing it, because +/// `data_bucket` is `std`. **A duplicated layout drifts**, and this is what +/// notices when it does. +#[test] +fn the_header_is_databuckets_layout() { + let mut out = Vec::new(); + Header { + version: 2, + schema: 3, + page: 7, + previous: 6, + next: 8, + page_type: PAGE_TYPE_DATA, + body: 0x1122_3344, + } + .write(&mut out); + + assert_eq!(out.len(), HEADER_SIZE, "GENERAL_HEADER_SIZE is 28"); + assert_eq!( + out, + alloc::vec![ + 0x02, 0x00, 0x00, 0x00, // data_version + 0x03, 0x00, 0x00, 0x00, // space_id, here the row fingerprint + 0x07, 0x00, 0x00, 0x00, // page_id + 0x06, 0x00, 0x00, 0x00, // previous_id + 0x08, 0x00, 0x00, 0x00, // next_id + 0x02, 0x00, 0x00, 0x00, // page_type: Data + 0x44, 0x33, 0x22, 0x11, // data_length + ], + "the layout drifted from data_bucket 0.5.7" + ); +} + +/// A page says how many rows it holds, which is what a WorkTable data page +/// cannot do and why one cannot be read without its index. +#[test] +fn every_page_declares_its_own_rows() { + let before = table(20_000); + let bytes = before.unload(); + let pages = bytes.len() / PAGE_SIZE; + assert!(pages > 2, "need several pages: {pages}"); + + let mut counted = 0usize; + for page in bytes.chunks_exact(PAGE_SIZE) { + let at = PAGE_SIZE - DIRECTORY_SIZE; + let mut word = [0u8; 4]; + word.copy_from_slice(&page[at..at + 4]); + counted += u32::from_le_bytes(word) as usize; + } + assert_eq!( + counted, + before.len(), + "the pages account for every row without an index" + ); +} + +/// Version 3 is the version that has a directory. A page claiming 2 is a +/// WorkTable page, and its rows are not where this reader would look. +#[test] +fn a_version_two_page_is_refused() { + let mut bytes = table(4).unload(); + bytes[0..4].copy_from_slice(&2u32.to_le_bytes()); + assert_eq!( + LinearTable::::load(&bytes), + Err(LoadError::ForeignPages { + page: 0, + version: 2 + }) + ); +} From 1803f22c0bb0453811014537d9992641b3885d23 Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 16:40:33 +0700 Subject: [PATCH 08/16] Show where an open spends its time, and export the page size to ask The example needed the page size to split a file into pages, and reached for a `hydrate_page_size()` that does not exist. `PAGE_SIZE` is already a `pub const`; the module is private, so nothing outside could see it. Export it rather than adding a function that returns it. Page decode is the only embarrassingly parallel part of a load, so this is the measurement that says whether threading it is worth anything at all. --- Cargo.toml | 4 +++ examples/where_time_goes.rs | 61 +++++++++++++++++++++++++++++++++++++ src/lib.rs | 2 +- 3 files changed, 66 insertions(+), 1 deletion(-) create mode 100644 examples/where_time_goes.rs diff --git a/Cargo.toml b/Cargo.toml index ac654d5..f09e8cb 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -39,3 +39,7 @@ rkyv = { version = "0.8.17", features = ["alloc", "bytecheck"] } name = "disk_cost" # It measures the page codec, so it needs it. required-features = ["hydrate"] + +[[example]] +name = "where_time_goes" +required-features = ["hydrate"] diff --git a/examples/where_time_goes.rs b/examples/where_time_goes.rs new file mode 100644 index 0000000..93cd22d --- /dev/null +++ b/examples/where_time_goes.rs @@ -0,0 +1,61 @@ +//! What an open actually spends its time on, before anyone parallelises it. +//! +//! A speedup on a synthetic loop says nothing about this path. Page decode is +//! the only embarrassingly parallel part, and it is only worth threads if it is +//! most of the time. + +use std::hint::black_box; +use std::time::Instant; + +use worktable_vec::{LinearTable, PAGE_SIZE}; + +const ROWS: usize = 200_000; +const REPS: usize = 5; + +fn main() -> Result<(), Box> { + let mut table = LinearTable::new(); + for n in 0..ROWS as u64 { + table.push( + n, + format!("row {n} with enough text to be worth serializing"), + ); + } + let bytes = table.unload(); + let pages = bytes.len() / PAGE_SIZE; + + let mut whole = Vec::new(); + let mut per_page = Vec::new(); + for _ in 0..REPS { + let now = Instant::now(); + let back = LinearTable::::load(black_box(&bytes))?; + whole.push(now.elapsed()); + black_box(back.len()); + + // Every page decoded on its own, which is what a worker would do, with + // no vector to append into and no ordering to preserve. + let now = Instant::now(); + let mut n = 0usize; + for page in bytes.chunks_exact(PAGE_SIZE) { + n += LinearTable::::load(black_box(page))?.len(); + } + per_page.push(now.elapsed()); + black_box(n); + } + whole.sort(); + per_page.sort(); + let (w, p) = (whole[REPS / 2], per_page[REPS / 2]); + println!("\n{ROWS} rows, {pages} pages, median of {REPS}\n"); + println!( + " load, whole file {:>7.1} ms", + w.as_secs_f64() * 1e3 + ); + println!( + " page decode alone {:>7.1} ms {:.0}% of it", + p.as_secs_f64() * 1e3, + 100.0 * p.as_secs_f64() / w.as_secs_f64() + ); + println!("\n Amdahl ceiling at 16 cores, if only decode parallelises:"); + let s = 1.0 - p.as_secs_f64() / w.as_secs_f64(); + println!(" {:.2}x", 1.0 / (s + (1.0 - s) / 16.0)); + Ok(()) +} diff --git a/src/lib.rs b/src/lib.rs index 02d5da8..0d7232b 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -17,7 +17,7 @@ extern crate alloc; #[cfg(feature = "hydrate")] mod hydrate; #[cfg(feature = "hydrate")] -pub use hydrate::{Codec, HydrateError, LoadError}; +pub use hydrate::{Codec, HydrateError, LoadError, PAGE_SIZE}; use alloc::collections::BTreeMap; #[cfg(feature = "congee")] From ecfcc7d407fff193af1a527a8ff8717ac5fefc1e Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 16:44:11 +0700 Subject: [PATCH 09/16] Stop the arctic backend from dragging std in, and check that it does not Enabling `arctic` enabled arctic's default features, which are `std` plus an SMR backend, so a crate that declares `#![no_std]` quietly linked one anyway. arctic 0.1.11 is the first release where this is fixable: before it, `smr-ps-reclaim` forced `std` on by itself. --features arctic arctic-wt [smr-ps-reclaim] ps-reclaim [libc,spin] --features arctic,std arctic-wt [smr-ps-reclaim,std] ps-reclaim [libc,spin,std] `congee` and `wti` now state `std` in the manifest, because neither crate is no_std upstream and pretending otherwise only moves the failure later. CI grows three steps in the existing job, no new runner: - the core and `hydrate` build for x86_64-unknown-none, which has no `std` to find, so the no_std claim is checked against a target rather than a host - the arctic graph is asserted to carry no `std` feature; the assertion was run against `--features arctic,std` first to prove it can fail The arctic backend itself cannot target bare metal: ps-reclaim stores its participant slot in thread-local storage and its no_std path is pthread keys, so it needs an OS. no_std here means no standard library, not no operating system, and that distinction is now written down instead of assumed. --- .github/workflows/ci.yml | 14 ++++++++++++++ Cargo.toml | 13 ++++++++++--- 2 files changed, 24 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 75ed45d..31d00df 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -13,6 +13,7 @@ jobs: - uses: dtolnay/rust-toolchain@stable with: components: rustfmt, clippy + targets: x86_64-unknown-none - uses: Swatinem/rust-cache@v2 - run: cargo fmt --all -- --check - run: cargo clippy --all-targets --no-default-features -- -D warnings @@ -21,6 +22,19 @@ jobs: - run: cargo test --all-targets --features arctic - run: cargo clippy --all-targets --all-features -- -D warnings - run: cargo test --all-targets --all-features + # A crate that says `#![no_std]` and is only ever built on a host with one + # proves nothing. This target has no `std` to find. + - run: cargo build --target x86_64-unknown-none --no-default-features + - run: cargo build --target x86_64-unknown-none --no-default-features --features hydrate + # The arctic backend needs an OS for its thread-local storage, so it + # cannot be built for a bare target - but it must still not pull `std` in. + - name: The arctic backend does not enable std + run: | + graph=$(cargo tree --no-default-features --features arctic -f '{p} [{f}]') + echo "$graph" | grep -E 'arctic-wt|ps-reclaim' + if echo "$graph" | grep -E '(arctic-wt|ps-reclaim) v[^[]*\[[^]]*std'; then + echo "std leaked into the arctic backend"; exit 1 + fi package: name: Release dry run diff --git a/Cargo.toml b/Cargo.toml index f09e8cb..1414307 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,12 +12,16 @@ categories = ["data-structures"] [features] default = [] +# On by default for nobody: this crate is `#![no_std]` and the core never needed +# `std`. The feature exists so a backend that does need one can say so, and so a +# bare-metal caller can prove it is not getting one. +std = ["arctic?/std"] # Load and unload rows as pages. Optional because it is the only thing here # that needs a serializer. hydrate = ["dep:rkyv", "dep:embedded-io"] arctic = ["dep:arctic"] -congee = ["dep:congee"] -wti = ["dep:wti"] +congee = ["dep:congee", "std"] +wti = ["dep:wti", "std"] [dependencies] rkyv = { version = "0.8.17", default-features = false, features = ["alloc", "bytecheck"], optional = true } @@ -26,7 +30,10 @@ rkyv = { version = "0.8.17", default-features = false, features = ["alloc", "byt # a caller with an OS gets real files and a caller without one plugs in whatever # it has. Nothing here needs `std` either way. embedded-io = { version = "0.7", default-features = false, features = ["alloc"], optional = true } -arctic = { package = "arctic-wt", version = "^0.1, >=0.1.9", optional = true } +# `default-features = false` and an explicit SMR backend, so enabling `arctic` +# does not silently enable `std`. 0.1.11 is the first release where that is +# possible: before it, `smr-ps-reclaim` forced `std` on. +arctic = { package = "arctic-wt", version = "^0.1, >=0.1.11", optional = true, default-features = false, features = ["smr-ps-reclaim"] } congee = { package = "congee-wt", version = "^0.4, >=0.4.4", optional = true } wti = { package = "WorkTablesIndex", version = "^0.0, >=0.0.11", optional = true, default-features = false, features = ["concurrent"] } From 29d6a0c86dbc2821a47878add5d1e6897c54286a Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 16:44:11 +0700 Subject: [PATCH 10/16] chore: 0.1.5 --- Cargo.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Cargo.toml b/Cargo.toml index 1414307..74c088c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "worktable-vec" -version = "0.1.4" +version = "0.1.5" edition = "2024" rust-version = "1.85" license = "MIT OR Apache-2.0" From 0e079a86a6332c393dd6c04aa1fa6893a77cedda Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 17:06:28 +0700 Subject: [PATCH 11/16] Require WorkTablesIndex 0.0.12, which is where the topology lives now 0.0.12 replaced the skiplist topology with ordered routing, took the shared reader bottleneck out of lookups, and hardened publication invariants. It is on WTI master and published, and WorkTable already resolves it. This crate was the last consumer still floored at 0.0.11, so it was the only one running the old topology. --- Cargo.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Cargo.toml b/Cargo.toml index 74c088c..fadf711 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -35,7 +35,7 @@ embedded-io = { version = "0.7", default-features = false, features = ["alloc"], # possible: before it, `smr-ps-reclaim` forced `std` on. arctic = { package = "arctic-wt", version = "^0.1, >=0.1.11", optional = true, default-features = false, features = ["smr-ps-reclaim"] } congee = { package = "congee-wt", version = "^0.4, >=0.4.4", optional = true } -wti = { package = "WorkTablesIndex", version = "^0.0, >=0.0.11", optional = true, default-features = false, features = ["concurrent"] } +wti = { package = "WorkTablesIndex", version = "^0.0, >=0.0.12", optional = true, default-features = false, features = ["concurrent"] } [dev-dependencies] # Only the tests need a real file, and this is what turns one into the trait. From 8a94c01520e935cca986859abf812e561558851b Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 17:11:47 +0700 Subject: [PATCH 12/16] Stop justifying a type by a target we do not ship The doc said keys are `u64` and that `usize` was chosen because `AtomicU64` does not exist on 32-bit bare-metal targets such as `thumbv7em-none-eabi`. Both halves are wrong. The API takes `key: usize` and always has, and nothing here is built or tested for a 32-bit target, so the justification cited support that does not exist to explain a choice that needs no excuse: the key is the same width as the slot arithmetic that indexes it. no_std is a statement about the standard library, not about a class of hardware. The wording no longer implies otherwise. --- Cargo.toml | 2 +- src/lib.rs | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index fadf711..064b684 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -14,7 +14,7 @@ categories = ["data-structures"] default = [] # On by default for nobody: this crate is `#![no_std]` and the core never needed # `std`. The feature exists so a backend that does need one can say so, and so a -# bare-metal caller can prove it is not getting one. +# caller without a standard library can prove it is not getting one. std = ["arctic?/std"] # Load and unload rows as pages. Optional because it is the only thing here # that needs a serializer. diff --git a/src/lib.rs b/src/lib.rs index 0d7232b..7874614 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -596,9 +596,9 @@ mod tests { /// /// No removal, no resize, and no iteration order beyond slot order. A full table refuses rather /// than growing, and [`AtomicKeyTable::len`] says how many slots are taken so a caller can -/// see it coming. Keys are `u64` and zero is the empty sentinel, so a caller whose key is a -/// pointer or a hash maps it in. `usize` rather than `u64` because `AtomicU64` does not exist -/// on 32-bit bare-metal targets such as `thumbv7em-none-eabi`, and this crate builds for them. +/// see it coming. Keys are `usize`, the same width as the slot arithmetic that indexes them, +/// and zero is the empty sentinel, so a caller whose key is a pointer, a hash or a `u64` maps +/// it in. /// Scatter a key across the table. /// /// # Why not the low bits, and why not a modulo From a197f6c3062e118d3cf600b63099b7bda54fb4a3 Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 17:30:40 +0700 Subject: [PATCH 13/16] Refuse a narrow target instead of carrying a path nobody runs `scatter` kept a second golden-ratio constant for a 32-bit `usize`, arriving with the same commit as the `thumbv7em-none-eabi` claim in the doc above it. Nothing here is built or tested for a narrow target, so that branch was code written for support that does not exist. It cannot simply be deleted: truncating the 64-bit constant leaves `0x7F4A_7C15`, which is even, so the multiply stops being invertible and keys quietly collapse onto the same slot. Silent wrong hashing is a worse answer than no support, so the crate now says which it is at compile time, the way WorkTable's codegen already does for a `u64` congee key. Verified both ways: a 32-bit target stops with "worktable-vec's AtomicKeyTable requires a 64-bit target", and x86_64-unknown-none still builds clean. --- src/lib.rs | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/src/lib.rs b/src/lib.rs index 7874614..65c4bcc 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -599,6 +599,14 @@ mod tests { /// see it coming. Keys are `usize`, the same width as the slot arithmetic that indexes them, /// and zero is the empty sentinel, so a caller whose key is a pointer, a hash or a `u64` maps /// it in. +// A 64-bit `usize`, and it refuses rather than assuming one. `GOLDEN` below is +// a 64-bit constant, and truncating it to 32 bits leaves an even number, which +// is not invertible and quietly collapses keys onto the same slot. Nothing here +// is built or tested for a narrower target, so the honest answer is to say so +// at compile time instead of carrying a second constant nobody exercises. +#[cfg(not(target_pointer_width = "64"))] +compile_error!("worktable-vec's AtomicKeyTable requires a 64-bit target"); + /// Scatter a key across the table. /// /// # Why not the low bits, and why not a modulo @@ -614,12 +622,8 @@ mod tests { /// power-of-two capacity fixes the second: the index is then a mask. #[inline(always)] const fn scatter(key: usize, shift: u32, mask: usize) -> usize { - // 2^BITS / phi, odd so the multiply is invertible and no input is lost. - const GOLDEN: usize = if usize::BITS == 64 { - 0x9E37_79B9_7F4A_7C15u64 as usize - } else { - 0x9E37_79B9u32 as usize - }; + // 2^64 / phi, odd so the multiply is invertible and no input is lost. + const GOLDEN: usize = 0x9E37_79B9_7F4A_7C15u64 as usize; (key.wrapping_mul(GOLDEN) >> shift) & mask } #[derive(Debug)] From b6cd8140566bd01f376cbe813c21c33a49a08b58 Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 17:41:29 +0700 Subject: [PATCH 14/16] Refuse a row too large for a page, instead of writing a file that will not load `unload()` returned `Vec` and could not fail. A row whose archive did not fit a page body got a page of its own anyway, spilling past the page boundary; `load` then stopped at `Overlong` and every row in the file was unreachable. The write reported success. Measured before the fix: one 20 KB row wrote 32,768 bytes and lost the row, and a 30 KB row among a hundred ordinary ones lost all hundred and one. The limit is 16,348 bytes of archive per row, which is a page minus the header and the directory, and it is a limit rather than a spill because a page holding part of a row stops standing alone. That property is the reason the format exists. unload, unload_appending -> Result, RowTooLarge> unload_to, append_to -> Result<(), UnloadError> `RowTooLarge` names the row, the size it needed and the size available. `UnloadError` mirrors `HydrateError`, splitting a sink that would not take the bytes from rows that cannot be written at all. Four tests, and all four fail with the check disabled: the refusal itself, the row it names, the boundary asserted from both sides, and a refused write leaving the sink untouched. cargo test --all-features 35 passed, 0 failed clippy, fmt, x86_64-unknown-none clean --- examples/where_time_goes.rs | 2 +- src/hydrate/io.rs | 65 ++++++++++++++++++----- src/hydrate/mod.rs | 74 ++++++++++++++++++++++---- src/hydrate/tests.rs | 101 ++++++++++++++++++++++++++++++------ src/lib.rs | 2 +- 5 files changed, 202 insertions(+), 42 deletions(-) diff --git a/examples/where_time_goes.rs b/examples/where_time_goes.rs index 93cd22d..f62b390 100644 --- a/examples/where_time_goes.rs +++ b/examples/where_time_goes.rs @@ -20,7 +20,7 @@ fn main() -> Result<(), Box> { format!("row {n} with enough text to be worth serializing"), ); } - let bytes = table.unload(); + let bytes = table.unload()?; let pages = bytes.len() / PAGE_SIZE; let mut whole = Vec::new(); diff --git a/src/hydrate/io.rs b/src/hydrate/io.rs index ce12700..4f8a3c2 100644 --- a/src/hydrate/io.rs +++ b/src/hydrate/io.rs @@ -31,7 +31,7 @@ use alloc::vec::Vec; use embedded_io::{Read, Write}; -use super::{Codec, LoadError, PAGE_SIZE, fingerprint, page_rows, to_pages}; +use super::{Codec, LoadError, PAGE_SIZE, RowTooLarge, fingerprint, page_rows, to_pages}; use crate::{IndexedTable, LinearTable}; /// A read that failed, either at the transport or at the page. @@ -78,6 +78,36 @@ impl core::fmt::Display for HydrateError { impl core::error::Error for HydrateError {} +/// A write that failed, either at the transport or at the rows. +/// +/// The mirror of [`HydrateError`], and split for the same reason: a sink that +/// would not take the bytes and a row that cannot be written at all are +/// different problems. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum UnloadError { + /// The writer failed. + Io(E), + /// The rows could not be made into pages. + Row(RowTooLarge), +} + +impl From for UnloadError { + fn from(error: RowTooLarge) -> Self { + Self::Row(error) + } +} + +impl core::fmt::Display for UnloadError { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Io(error) => write!(formatter, "the writer failed: {error}"), + Self::Row(error) => error.fmt(formatter), + } + } +} + +impl core::error::Error for UnloadError {} + /// Fill `page` from `source`, or say how far it got. /// /// Returns `Ok(false)` at a clean end of input, which is the only case where a @@ -140,10 +170,11 @@ where /// /// # Errors /// - /// Whatever the writer reports. - pub fn unload_to(&self, sink: &mut W) -> Result<(), W::Error> { - sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; - sink.flush() + /// Whatever the writer reports, or a row too large for a page. + pub fn unload_to(&self, sink: &mut W) -> Result<(), UnloadError> { + let pages = to_pages(&self.rows, fingerprint::>())?; + sink.write_all(&pages).map_err(UnloadError::Io)?; + sink.flush().map_err(UnloadError::Io) } /// Write the rows from `first` on, for a sink already holding the rest. @@ -153,14 +184,19 @@ where /// /// # Errors /// - /// Whatever the writer reports. - pub fn append_to(&self, sink: &mut W, first: usize) -> Result<(), W::Error> { + /// Whatever the writer reports, or a row too large for a page. + pub fn append_to( + &self, + sink: &mut W, + first: usize, + ) -> Result<(), UnloadError> { let first = first.min(self.rows.len()); if first == self.rows.len() { - return sink.flush(); + return sink.flush().map_err(UnloadError::Io); } - sink.write_all(&to_pages(&self.rows[first..], fingerprint::>()))?; - sink.flush() + let pages = to_pages(&self.rows[first..], fingerprint::>())?; + sink.write_all(&pages).map_err(UnloadError::Io)?; + sink.flush().map_err(UnloadError::Io) } /// Read a table back from a reader. @@ -188,10 +224,11 @@ where /// /// # Errors /// - /// Whatever the writer reports. - pub fn unload_to(&self, sink: &mut W) -> Result<(), W::Error> { - sink.write_all(&to_pages(&self.rows, fingerprint::>()))?; - sink.flush() + /// Whatever the writer reports, or a row too large for a page. + pub fn unload_to(&self, sink: &mut W) -> Result<(), UnloadError> { + let pages = to_pages(&self.rows, fingerprint::>())?; + sink.write_all(&pages).map_err(UnloadError::Io)?; + sink.flush().map_err(UnloadError::Io) } /// Read a table back from a reader, rebuilding the index. diff --git a/src/hydrate/mod.rs b/src/hydrate/mod.rs index 4e10080..5365945 100644 --- a/src/hydrate/mod.rs +++ b/src/hydrate/mod.rs @@ -204,6 +204,34 @@ impl core::fmt::Display for LoadError { impl core::error::Error for LoadError {} +/// The one thing an unload can refuse on. +/// +/// A page body is [`BODY_SIZE`] bytes and a row is written whole, so a row +/// whose archive does not fit one cannot be written at all. It is a refusal +/// rather than a spill because a page that carries part of a row stops +/// standing alone, which is the property the format exists for. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct RowTooLarge { + /// Which row, counting from the first one written. + pub row: usize, + /// How many bytes its archive needed. + pub bytes: usize, + /// How many a page body holds. + pub limit: usize, +} + +impl core::fmt::Display for RowTooLarge { + fn fmt(&self, formatter: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + let Self { row, bytes, limit } = self; + write!( + formatter, + "row {row} archives to {bytes} bytes and a page body holds {limit}" + ) + } +} + +impl core::error::Error for RowTooLarge {} + /// What a row set has to be able to do to make the trip. /// /// The bounds are rkyv's and there are five lines of them, so they are stated @@ -461,7 +489,7 @@ where } /// Rows to pages, each page standing alone. -pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Vec +pub(crate) fn to_pages(rows: &[(K, V)], schema: u32) -> Result, RowTooLarge> where Vec<(K, V)>: Codec, (K, V): Clone, @@ -480,6 +508,18 @@ where hint = take; let archive = rest[..take].to_vec().encode(); let body = archive.as_ref(); + // `rows_per_page` returns at least one so the loop always advances, so + // a body over the limit means that one row does not fit a page. It is + // caught here rather than written, because the writer used to produce + // a file `load` then refused: `unload` reported success and the rows + // were gone. + if body.len() > BODY_SIZE { + return Err(RowTooLarge { + row: rows.len() - rest.len(), + bytes: body.len(), + limit: BODY_SIZE, + }); + } let page = u32::try_from(out.len() / PAGE_SIZE).expect("a page index inside u32"); let last = rest.len() == take; Header { @@ -510,7 +550,7 @@ where break; } } - out + Ok(out) } /// One page back into rows, with every header field checked. @@ -601,8 +641,11 @@ where (K, V): Clone, { /// Every row, as pages. - #[must_use] - pub fn unload(&self) -> Vec { + /// + /// # Errors + /// + /// Refuses a row whose archive does not fit one page body. + pub fn unload(&self) -> Result, RowTooLarge> { to_pages(&self.rows, fingerprint::>()) } @@ -611,8 +654,11 @@ where /// /// Appending is possible at all because pages stand alone: the existing /// file is untouched and these pages are simply more of them. - #[must_use] - pub fn unload_appending(&self, first: usize) -> Vec { + /// + /// # Errors + /// + /// Refuses a row whose archive does not fit one page body. + pub fn unload_appending(&self, first: usize) -> Result, RowTooLarge> { let first = first.min(self.rows.len()); to_pages(&self.rows[first..], fingerprint::>()) } @@ -643,14 +689,20 @@ where /// The index is not written. It is derived from the rows, so rebuilding it /// on load costs one pass, where storing it would cost bytes at rest and a /// second thing that can disagree with the rows. - #[must_use] - pub fn unload(&self) -> Vec { + /// + /// # Errors + /// + /// Refuses a row whose archive does not fit one page body. + pub fn unload(&self) -> Result, RowTooLarge> { to_pages(&self.rows, fingerprint::>()) } /// The rows from `first` on, as pages. - #[must_use] - pub fn unload_appending(&self, first: usize) -> Vec { + /// + /// # Errors + /// + /// Refuses a row whose archive does not fit one page body. + pub fn unload_appending(&self, first: usize) -> Result, RowTooLarge> { let first = first.min(self.rows.len()); to_pages(&self.rows[first..], fingerprint::>()) } @@ -671,7 +723,7 @@ where } mod io; -pub use io::HydrateError; +pub use io::{HydrateError, UnloadError}; #[cfg(test)] mod tests; diff --git a/src/hydrate/tests.rs b/src/hydrate/tests.rs index 65a1603..53ef187 100644 --- a/src/hydrate/tests.rs +++ b/src/hydrate/tests.rs @@ -12,7 +12,8 @@ fn table(rows: usize) -> LinearTable { #[test] fn rows_survive_the_round_trip() { let before = table(1_000); - let back = LinearTable::::load(&before.unload()).expect("a load"); + let back = LinearTable::::load(&before.unload().expect("rows that fit a page")) + .expect("a load"); assert_eq!(back.rows(), before.rows()); } @@ -20,7 +21,7 @@ fn rows_survive_the_round_trip() { #[test] fn rows_survive_spanning_many_pages() { let before = table(20_000); - let bytes = before.unload(); + let bytes = before.unload().expect("rows that fit a page"); assert!( bytes.len() / PAGE_SIZE > 1, "the fixture has to span pages: {} pages", @@ -32,7 +33,7 @@ fn rows_survive_spanning_many_pages() { #[test] fn an_empty_table_is_one_page_and_comes_back_empty() { - let bytes = table(0).unload(); + let bytes = table(0).unload().expect("rows that fit a page"); assert_eq!(bytes.len(), PAGE_SIZE); assert!( LinearTable::::load(&bytes) @@ -47,7 +48,8 @@ fn the_index_is_rebuilt_rather_than_stored() { for n in 0..500u64 { before.insert(n, n.to_string()).expect("a row"); } - let back = IndexedTable::::load(&before.unload()).expect("a load"); + let back = IndexedTable::::load(&before.unload().expect("rows that fit a page")) + .expect("a load"); assert_eq!(back.len(), 500); assert_eq!(back.select(&37), Some(&"37".to_string())); } @@ -78,7 +80,7 @@ fn a_zeroed_page_is_not_an_empty_table() { /// succeeds and hands back debris. #[test] fn a_different_row_type_is_refused_rather_than_reinterpreted() { - let bytes = table(10).unload(); + let bytes = table(10).unload().expect("rows that fit a page"); assert_eq!( LinearTable::::load(&bytes), Err(LoadError::ForeignRows { @@ -91,10 +93,10 @@ fn a_different_row_type_is_refused_rather_than_reinterpreted() { /// Two files spliced together are not one longer file. #[test] fn pages_from_two_runs_are_refused() { - let mut spliced = table(1).unload(); + let mut spliced = table(1).unload().expect("rows that fit a page"); let mut other: LinearTable = LinearTable::new(); other.push(1, 1); - spliced.extend_from_slice(&other.unload()); + spliced.extend_from_slice(&other.unload().expect("rows that fit a page")); assert_eq!( LinearTable::::load(&spliced), Err(LoadError::Inconsistent { page: 1 }) @@ -106,7 +108,7 @@ fn pages_from_two_runs_are_refused() { /// only the checksum notices. #[test] fn a_flipped_bit_in_a_body_is_caught_by_the_checksum() { - let mut bytes = table(64).unload(); + let mut bytes = table(64).unload().expect("rows that fit a page"); bytes[HEADER_SIZE + 40] ^= 0b0000_0100; match LinearTable::::load(&bytes) { Err(LoadError::Corrupt { page: 0, .. }) => {} @@ -119,7 +121,7 @@ fn a_flipped_bit_in_a_body_is_caught_by_the_checksum() { #[test] fn damage_stays_inside_the_page_it_happened_to() { let before = table(20_000); - let bytes = before.unload(); + let bytes = before.unload().expect("rows that fit a page"); let pages = bytes.len() / PAGE_SIZE; assert!(pages > 2, "need a middle page to damage: {pages}"); @@ -134,7 +136,7 @@ fn damage_stays_inside_the_page_it_happened_to() { #[test] fn a_row_count_that_disagrees_with_the_body_is_refused() { - let mut bytes = table(64).unload(); + let mut bytes = table(64).unload().expect("rows that fit a page"); // Claim one more row than the body holds, and fix nothing else. The count // is in the directory at the page tail, not in the header, because the // header is DataBucket's and has no field for it. @@ -156,8 +158,8 @@ fn appended_pages_read_back_as_one_table() { for (key, value) in whole.rows().iter().take(2_000).cloned() { first.push(key, value); } - bytes.extend_from_slice(&first.unload()); - bytes.extend_from_slice(&whole.unload_appending(2_000)); + bytes.extend_from_slice(&first.unload().expect("rows that fit a page")); + bytes.extend_from_slice(&whole.unload_appending(2_000).expect("rows that fit a page")); let back = LinearTable::::load(&bytes).expect("a load"); assert_eq!(back.rows(), whole.rows()); @@ -256,7 +258,7 @@ mod through_a_reader_and_a_writer { /// rather than quietly dropping the rows it did not finish. #[test] fn a_half_written_page_is_torn_rather_than_ignored() { - let bytes = table(5_000).unload(); + let bytes = table(5_000).unload().expect("rows that fit a page"); let cut = bytes.len() - (PAGE_SIZE / 2); match LinearTable::::load_from(&mut &bytes[..cut]) { Err(HydrateError::Torn { .. }) => {} @@ -307,7 +309,7 @@ fn the_header_is_databuckets_layout() { #[test] fn every_page_declares_its_own_rows() { let before = table(20_000); - let bytes = before.unload(); + let bytes = before.unload().expect("rows that fit a page"); let pages = bytes.len() / PAGE_SIZE; assert!(pages > 2, "need several pages: {pages}"); @@ -329,7 +331,7 @@ fn every_page_declares_its_own_rows() { /// WorkTable page, and its rows are not where this reader would look. #[test] fn a_version_two_page_is_refused() { - let mut bytes = table(4).unload(); + let mut bytes = table(4).unload().expect("rows that fit a page"); bytes[0..4].copy_from_slice(&2u32.to_le_bytes()); assert_eq!( LinearTable::::load(&bytes), @@ -339,3 +341,72 @@ fn a_version_two_page_is_refused() { }) ); } + +/// A row that does not fit a page is refused, rather than written into a file +/// that cannot be read back. +/// +/// This is a regression. The writer used to hand such a row a page of its own, +/// spilling past the page boundary; `load` then stopped at +/// `Overlong`, so `unload` reported success and every row in the file was +/// unreachable. A 20 KB row wrote 32,768 bytes and lost one row; a 30 KB row +/// among a hundred ordinary ones lost all hundred and one. +#[test] +fn a_row_too_large_for_a_page_is_refused() { + let mut table = LinearTable::::new(); + table.push(0, "x".repeat(20_000)); + let refusal = table + .unload() + .expect_err("a row that big cannot be written"); + assert_eq!(refusal.row, 0); + assert_eq!(refusal.limit, BODY_SIZE); + assert!( + refusal.bytes > BODY_SIZE, + "the refusal reports the archive size it could not place: {refusal}" + ); +} + +/// The refusal names the row, not just the fact of one. +#[test] +fn the_refusal_names_which_row_is_too_large() { + let mut table = table(50); + table.push(999, "y".repeat(30_000)); + for n in 1_000..1_050u64 { + table.push(n, "z".repeat(100)); + } + let refusal = table + .unload() + .expect_err("a row that big cannot be written"); + assert_eq!(refusal.row, 50, "{refusal}"); +} + +/// The limit is a page body, and a row just under it still writes. +/// +/// Both sides are asserted so the boundary is pinned from both directions: a +/// check that only ever refuses would pass with the limit set to zero. +#[test] +fn the_limit_is_a_page_body_and_not_less() { + let mut fits = LinearTable::::new(); + fits.push(0, "x".repeat(BODY_SIZE - 64)); + let bytes = fits.unload().expect("a row just under the limit fits"); + let back = LinearTable::::load(&bytes).expect("a load"); + assert_eq!(back.rows(), fits.rows()); + + let mut over = LinearTable::::new(); + over.push(0, "x".repeat(BODY_SIZE + 1)); + assert!(over.unload().is_err(), "a row over the limit is refused"); +} + +/// A refused unload writes nothing at all. +/// +/// The refusal happens while the pages are being built, before the sink is +/// touched, so a caller who ignores the error still does not end up with a +/// half-written file. +#[test] +fn a_refused_unload_leaves_the_sink_untouched() { + let mut table = LinearTable::::new(); + table.push(0, "x".repeat(20_000)); + let mut sink = alloc::vec::Vec::new(); + let refusal = table.unload_to(&mut sink).expect_err("nothing to write"); + assert!(matches!(refusal, UnloadError::Row(_)), "{refusal}"); + assert!(sink.is_empty(), "{} bytes reached the sink", sink.len()); +} diff --git a/src/lib.rs b/src/lib.rs index 65c4bcc..04ccb57 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -17,7 +17,7 @@ extern crate alloc; #[cfg(feature = "hydrate")] mod hydrate; #[cfg(feature = "hydrate")] -pub use hydrate::{Codec, HydrateError, LoadError, PAGE_SIZE}; +pub use hydrate::{Codec, HydrateError, LoadError, PAGE_SIZE, RowTooLarge, UnloadError}; use alloc::collections::BTreeMap; #[cfg(feature = "congee")] From 0315365aaefd7ce71e7a413f2735dde2554ff511 Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 18:07:37 +0700 Subject: [PATCH 15/16] Say the requirements as carets `"^0.1, >=0.1.11"` and `"^0.1.11"` describe the same set, and the second one says it once. A bare `"0.1.11"` is already a caret, which is the form the rest of the house uses. `wti` is the one that changes meaning. Caret on a `0.0.x` version pins the last number, so `"0.0.12"` will not take 0.0.13 the way `"^0.0, >=0.0.12"` would have. That is what a caret means for a crate below 0.1.0 - every release is breaking - so the bump becomes deliberate rather than automatic. cargo test --all-features 35 passed, 0 failed --- Cargo.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 064b684..5587310 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -33,9 +33,9 @@ embedded-io = { version = "0.7", default-features = false, features = ["alloc"], # `default-features = false` and an explicit SMR backend, so enabling `arctic` # does not silently enable `std`. 0.1.11 is the first release where that is # possible: before it, `smr-ps-reclaim` forced `std` on. -arctic = { package = "arctic-wt", version = "^0.1, >=0.1.11", optional = true, default-features = false, features = ["smr-ps-reclaim"] } -congee = { package = "congee-wt", version = "^0.4, >=0.4.4", optional = true } -wti = { package = "WorkTablesIndex", version = "^0.0, >=0.0.12", optional = true, default-features = false, features = ["concurrent"] } +arctic = { package = "arctic-wt", version = "0.1.11", optional = true, default-features = false, features = ["smr-ps-reclaim"] } +congee = { package = "congee-wt", version = "0.4.4", optional = true } +wti = { package = "WorkTablesIndex", version = "0.0.12", optional = true, default-features = false, features = ["concurrent"] } [dev-dependencies] # Only the tests need a real file, and this is what turns one into the trait. From da1991cbbee6b8a20e53f42b90be0162a6dd9f51 Mon Sep 17 00:00:00 2001 From: meh Date: Mon, 7 Sep 2026 18:33:35 +0700 Subject: [PATCH 16/16] Stop congee and wti implying std, now that neither does Both crates were std-only, so this crate declared it rather than pretending otherwise. congee-wt 0.4.5 and WorkTablesIndex 0.0.13 changed that, so the declaration is now wrong and comes out. --features arctic arctic-wt [smr-ps-reclaim] ps-reclaim [libc,spin] --features congee congee-wt [] ps-reclaim [libc,spin] --features wti WorkTablesIndex [concurrent] parking_lot_lite_hack [arc_lock,send_guard] ps-reclaim [libc,spin] with std added every one of them gains it, and nothing gains it without asking WorkTablesIndex moves by hand rather than by caret: `0.0.12` will not take 0.0.13, because a caret on a `0.0.x` version pins the last number. That is what a caret means below 0.1.0, so the bump is deliberate. The CI assertion covered arctic alone and now covers all three, together and separately. Run against `--features arctic,congee,wti,std` first, where it fails, because a check that cannot fail proves nothing. all features 35 passed no_std + arctic / congee / wti 10 passed each x86_64-unknown-none core and hydrate build clippy, fmt clean --- .github/workflows/ci.yml | 19 +++++++++++-------- Cargo.toml | 10 +++++----- 2 files changed, 16 insertions(+), 13 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 31d00df..8b3bb8b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -26,15 +26,18 @@ jobs: # proves nothing. This target has no `std` to find. - run: cargo build --target x86_64-unknown-none --no-default-features - run: cargo build --target x86_64-unknown-none --no-default-features --features hydrate - # The arctic backend needs an OS for its thread-local storage, so it - # cannot be built for a bare target - but it must still not pull `std` in. - - name: The arctic backend does not enable std + # The backends need an OS for their thread-local storage, so they cannot + # be built for a bare target - but none of them may pull `std` in. + - name: No backend enables std run: | - graph=$(cargo tree --no-default-features --features arctic -f '{p} [{f}]') - echo "$graph" | grep -E 'arctic-wt|ps-reclaim' - if echo "$graph" | grep -E '(arctic-wt|ps-reclaim) v[^[]*\[[^]]*std'; then - echo "std leaked into the arctic backend"; exit 1 - fi + for features in arctic congee wti arctic,congee,wti; do + graph=$(cargo tree -e normal --no-default-features --features "$features" -f '{p} [{f}]') + echo "--- $features" + echo "$graph" | grep -E 'arctic-wt|congee-wt|WorkTablesIndex|ps-reclaim|parking_lot' + if echo "$graph" | grep -E '(arctic-wt|congee-wt|WorkTablesIndex|ps-reclaim|parking_lot[a-z_]*) v[^[]*\[[^]]*std'; then + echo "std leaked into --features $features"; exit 1 + fi + done package: name: Release dry run diff --git a/Cargo.toml b/Cargo.toml index 5587310..ec850f3 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -15,13 +15,13 @@ default = [] # On by default for nobody: this crate is `#![no_std]` and the core never needed # `std`. The feature exists so a backend that does need one can say so, and so a # caller without a standard library can prove it is not getting one. -std = ["arctic?/std"] +std = ["arctic?/std", "congee?/std", "wti?/std"] # Load and unload rows as pages. Optional because it is the only thing here # that needs a serializer. hydrate = ["dep:rkyv", "dep:embedded-io"] arctic = ["dep:arctic"] -congee = ["dep:congee", "std"] -wti = ["dep:wti", "std"] +congee = ["dep:congee"] +wti = ["dep:wti"] [dependencies] rkyv = { version = "0.8.17", default-features = false, features = ["alloc", "bytecheck"], optional = true } @@ -34,8 +34,8 @@ embedded-io = { version = "0.7", default-features = false, features = ["alloc"], # does not silently enable `std`. 0.1.11 is the first release where that is # possible: before it, `smr-ps-reclaim` forced `std` on. arctic = { package = "arctic-wt", version = "0.1.11", optional = true, default-features = false, features = ["smr-ps-reclaim"] } -congee = { package = "congee-wt", version = "0.4.4", optional = true } -wti = { package = "WorkTablesIndex", version = "0.0.12", optional = true, default-features = false, features = ["concurrent"] } +congee = { package = "congee-wt", version = "0.4.5", optional = true, default-features = false } +wti = { package = "WorkTablesIndex", version = "0.0.13", optional = true, default-features = false, features = ["concurrent"] } [dev-dependencies] # Only the tests need a real file, and this is what turns one into the trait.