From 380ac24eb5d1f6a5a79e98a54c1b67fa04f95f9f Mon Sep 17 00:00:00 2001 From: MauroFab Date: Mon, 10 Aug 2026 15:06:38 -0300 Subject: [PATCH 1/2] fix(gpu): harden the downgrade download recovery Three fixes on the device-only downgrade path, all in the graceful degradation function whose whole point is to avoid a hard abort. - The aux branch of `materialize_lde_trace_host` sliced the downloaded slabs without checking their length, so a short download would panic inside the recovery instead of degrading. Both sibling download paths already validate (`download_main_lde_row_major` checks `col_major.len() != m * lde`, `materialize_aux_trace_host` checks `raw.len() != rows * cols * 3`); this adds the matching check. - Restore the `len/capacity % 3` guard the other two ext3 `from_raw_parts` sites carry, spelled `is_multiple_of` because clippy's `manual_is_multiple_of` rejects the older form here. - The failure error claimed "host aux trace is empty" on a path where that is false: when the aux download succeeded and the follow-up main-LDE download failed, the host aux trace had just been populated. Track which recovery step failed and name it. Control flow unchanged. --- crypto/stark/src/gpu_lde.rs | 10 ++++++++++ crypto/stark/src/prover.rs | 14 +++++++++++++- 2 files changed, 23 insertions(+), 1 deletion(-) diff --git a/crypto/stark/src/gpu_lde.rs b/crypto/stark/src/gpu_lde.rs index 2a9547153..6fef1d9de 100644 --- a/crypto/stark/src/gpu_lde.rs +++ b/crypto/stark/src/gpu_lde.rs @@ -1512,6 +1512,12 @@ where return false; } let (m, lde) = (h.m, h.lde_size); + // Short download: degrade like the sibling paths + // (`download_main_lde_row_major`, `materialize_aux_trace_host`) + // rather than panic on the slab slicing below. + if slabs.len() != m * lde * 3 { + return false; + } let mut interleaved = vec![0u64; m * lde * 3]; for c in 0..m { for k in 0..3 { @@ -1525,6 +1531,10 @@ where // is [u64; 3]. unsafe { let mut v = std::mem::ManuallyDrop::new(interleaved); + debug_assert!( + v.len().is_multiple_of(3) && v.capacity().is_multiple_of(3), + "interleaved len/capacity must be a multiple of 3 for Fp3 reinterpret" + ); Vec::from_raw_parts( v.as_mut_ptr() as *mut FieldElement, v.len() / 3, diff --git a/crypto/stark/src/prover.rs b/crypto/stark/src/prover.rs index d5cca05e1..b609786e7 100644 --- a/crypto/stark/src/prover.rs +++ b/crypto/stark/src/prover.rs @@ -3478,6 +3478,11 @@ pub trait IsStarkProver< // the main LDE if this table was device-only — and // continue fully host-backed on the arms below. let mut recovered = crate::gpu_lde::materialize_aux_trace_host(*trace); + // Once the aux download lands, the host aux trace is + // populated: a later failure is the main-LDE + // download's, and the error has to name that step + // instead of claiming an empty aux trace. + let aux_recovered = recovered; if recovered && device_only { let mut cell = main_lde_cells[idx].lock().unwrap(); if let Some((data, _)) = cell.as_mut() @@ -3506,7 +3511,14 @@ pub trait IsStarkProver< } if !recovered { return Err(ProvingError::Fft( - "resident aux LDE failed; host aux trace is empty".to_string(), + if aux_recovered { + "resident aux LDE declined; the aux trace was recovered \ + but the main-LDE download failed" + } else { + "resident aux LDE declined and the aux-trace download \ + recovery failed" + } + .to_string(), )); } eprintln!( From beebf2800d62d81d047c69b232336c26e62ff446 Mon Sep 17 00:00:00 2001 From: MauroFab Date: Mon, 10 Aug 2026 15:06:55 -0300 Subject: [PATCH 2/2] perf(gpu): parallelize the downgrade recovery transposes Both conversions in the recovery path were single-threaded nested loops over the full LDE: the col-major -> row-major main transpose in `download_main_lde_row_major`, and the de-interleaved-slabs -> row-major interleaved aux conversion in `materialize_lde_trace_host`. For MEMW at LDE 2^20 those are a 411 MB and a 327 MB buffer respectively, walked with a strided access on one core. Both now follow the existing idiom in `trace.rs` ("Parallel col-major -> row-major transpose"): parallelize over OUTPUT row chunks with `par_chunks_exact_mut`, so every element is still written exactly once and no unsafe is involved. The index math is unchanged -- chunk `r` of width `m` is `row_major[r * m + c]`, and chunk `r` of width `m * 3` sub-chunked by 3 is `interleaved[(r * m + c) * 3 + k]` -- because the layout was verified against the kernels. Gated on the `parallel` feature with the sequential loop kept for builds without it, and skipped when `m == 0` since `chunks_exact_mut(0)` panics. These loops run on a scheduler driver thread holding no locks, so rayon is safe here, unlike the pinned-staging unpack in math-cuda. --- crypto/stark/src/gpu_lde.rs | 55 +++++++++++++++++++++++++++++++------ 1 file changed, 47 insertions(+), 8 deletions(-) diff --git a/crypto/stark/src/gpu_lde.rs b/crypto/stark/src/gpu_lde.rs index 6fef1d9de..e4e85792e 100644 --- a/crypto/stark/src/gpu_lde.rs +++ b/crypto/stark/src/gpu_lde.rs @@ -26,6 +26,8 @@ use math::field::extensions_goldilocks::Degree3GoldilocksExtensionField; use math::field::goldilocks::GoldilocksField; use math::field::traits::{IsFFTField, IsField, IsSubFieldOf}; use math::traits::AsBytes; +#[cfg(feature = "parallel")] +use rayon::prelude::{IndexedParallelIterator, ParallelIterator, ParallelSliceMut}; use crate::config::{Commitment, FriLayerMerkleTreeBackend}; use crate::domain::Domain; @@ -1518,12 +1520,31 @@ where if slabs.len() != m * lde * 3 { return false; } + // Parallel de-interleaved slabs → row-major interleaved: each row + // chunk gathers from the source slabs independently. let mut interleaved = vec![0u64; m * lde * 3]; - for c in 0..m { - for k in 0..3 { - let slab = &slabs[(c * 3 + k) * lde..(c * 3 + k + 1) * lde]; - for r in 0..lde { - interleaved[(r * m + c) * 3 + k] = slab[r]; + if m > 0 { + #[cfg(feature = "parallel")] + { + interleaved + .par_chunks_exact_mut(m * 3) + .enumerate() + .for_each(|(r, dst)| { + for (c, dst_col) in dst.chunks_exact_mut(3).enumerate() { + for (k, d) in dst_col.iter_mut().enumerate() { + *d = slabs[(c * 3 + k) * lde + r]; + } + } + }); + } + #[cfg(not(feature = "parallel"))] + { + for (r, dst) in interleaved.chunks_exact_mut(m * 3).enumerate() { + for (c, dst_col) in dst.chunks_exact_mut(3).enumerate() { + for (k, d) in dst_col.iter_mut().enumerate() { + *d = slabs[(c * 3 + k) * lde + r]; + } + } } } } @@ -1567,10 +1588,28 @@ where if col_major.len() != m * lde { return None; } + // Parallel col-major → row-major transpose: each row chunk gathers from + // the source columns independently. let mut row_major = vec![0u64; m * lde]; - for c in 0..m { - for r in 0..lde { - row_major[r * m + c] = col_major[c * lde + r]; + if m > 0 { + #[cfg(feature = "parallel")] + { + row_major + .par_chunks_exact_mut(m) + .enumerate() + .for_each(|(r, dst)| { + for (c, d) in dst.iter_mut().enumerate() { + *d = col_major[c * lde + r]; + } + }); + } + #[cfg(not(feature = "parallel"))] + { + for (r, dst) in row_major.chunks_exact_mut(m).enumerate() { + for (c, d) in dst.iter_mut().enumerate() { + *d = col_major[c * lde + r]; + } + } } } // SAFETY: F == Goldilocks (gated above); FieldElement is