From a812ad4fde8d834d8cf0f1fc4ddc8d973e019e71 Mon Sep 17 00:00:00 2001 From: Joaquin Carletti Date: Wed, 12 Aug 2026 15:53:22 -0300 Subject: [PATCH 1/5] fix(gpu): recover device-only declines at the remaining cliff sites (R2 commit, R3 OOD, R4 DEEP) Under VRAM pressure a device dispatch can decline after the device-only gate already skipped the host drain, and the host fallbacks at the R2 comp-poly commit, the R3 parts/trace OOD and the R4 DEEP loop hard-abort on the empty host buffers. Download the resident data instead: the trace LDEs via materialize_lde_trace_host, the H part evaluations via a new download off the resident R2 parts handle. The asserts remain only for handles that cannot serve the data. The R4 DEEP host loop reads both the trace and the part evals, so it recovers both sides. Also adds sticky fault-injection hooks (test-faults) to the cuda barycentric, DEEP and comp-tree entries: the drain-and-retry absorbs one-shot faults, so the cliff paths need a fault that keeps firing. --- crypto/math-cuda/src/barycentric.rs | 8 ++ crypto/math-cuda/src/deep.rs | 8 ++ crypto/math-cuda/src/faults.rs | 39 +++++++ crypto/math-cuda/src/lib.rs | 2 + crypto/math-cuda/src/merkle.rs | 4 + crypto/stark/src/gpu_lde.rs | 153 +++++++++++++++++++++++-- crypto/stark/src/prover.rs | 124 ++++++++++++-------- crypto/stark/src/tests/prover_tests.rs | 4 +- crypto/stark/src/trace.rs | 43 ++++--- 9 files changed, 312 insertions(+), 73 deletions(-) create mode 100644 crypto/math-cuda/src/faults.rs diff --git a/crypto/math-cuda/src/barycentric.rs b/crypto/math-cuda/src/barycentric.rs index e9aceaea2..41df3119f 100644 --- a/crypto/math-cuda/src/barycentric.rs +++ b/crypto/math-cuda/src/barycentric.rs @@ -143,6 +143,8 @@ pub fn barycentric_base_on_device( inv_denoms_ext3: &[u64], n: usize, ) -> Result> { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_BARYCENTRIC_STICKY)?; assert_eq!(coset_points.len(), n); assert_eq!(inv_denoms_ext3.len(), 3 * n); let num_cols = main_handle.m; @@ -204,6 +206,8 @@ pub fn barycentric_base_on_device_with_dev_inv_denoms( inv_offset_u64: usize, n: usize, ) -> Result> { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_BARYCENTRIC_STICKY)?; main_handle.wait_ready_on(stream)?; assert!(coset_points_dev.len() >= n); let inv_end = inv_offset_u64 @@ -255,6 +259,8 @@ pub fn barycentric_ext3_on_device( inv_denoms_ext3: &[u64], n: usize, ) -> Result> { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_BARYCENTRIC_STICKY)?; assert_eq!(coset_points.len(), n); assert_eq!(inv_denoms_ext3.len(), 3 * n); let num_cols = aux_handle.m; @@ -308,6 +314,8 @@ pub fn barycentric_ext3_on_device_with_dev_inv_denoms( inv_offset_u64: usize, n: usize, ) -> Result> { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_BARYCENTRIC_STICKY)?; aux_handle.wait_ready_on(stream)?; assert!(coset_points_dev.len() >= n); let inv_end = inv_offset_u64 diff --git a/crypto/math-cuda/src/deep.rs b/crypto/math-cuda/src/deep.rs index 241ac5ad3..b0eefd61d 100644 --- a/crypto/math-cuda/src/deep.rs +++ b/crypto/math-cuda/src/deep.rs @@ -41,6 +41,8 @@ pub fn deep_composition_ext3( row_stride: usize, domain_size: usize, ) -> Result> { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_DEEP_STICKY)?; let be = backend()?; let stream = be.next_stream(); deep_composition_ext3_impl( @@ -86,6 +88,8 @@ pub fn deep_composition_ext3_with_dev_parts( row_stride: usize, domain_size: usize, ) -> Result> { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_DEEP_STICKY)?; let be = backend()?; let stream = be.next_stream(); deep_composition_ext3_impl( @@ -262,6 +266,8 @@ pub fn deep_composition_ext3_with_dev_parts_and_inv_denoms( row_stride: usize, domain_size: usize, ) -> Result> { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_DEEP_STICKY)?; let deep_out = deep_fully_resident_launch( stream, main_lde, @@ -324,6 +330,8 @@ pub fn deep_composition_ext3_fully_resident_keep( row_stride: usize, domain_size: usize, ) -> Result { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_DEEP_STICKY)?; assert!( domain_size.is_power_of_two() && domain_size >= 2, "bit-reverse needs a power-of-two codeword" diff --git a/crypto/math-cuda/src/faults.rs b/crypto/math-cuda/src/faults.rs new file mode 100644 index 000000000..5e9fe4dcf --- /dev/null +++ b/crypto/math-cuda/src/faults.rs @@ -0,0 +1,39 @@ +//! Sticky fault-injection hooks for the GPU error-path tests. +//! +//! Unlike the one-shot hooks in `fri` and `inverse` (which disarm after +//! firing, so a drain-and-retry absorbs the injected error before it can +//! surface), a sticky hook keeps failing once its armed call count is +//! reached, until explicitly disarmed. The device-decline recovery tests +//! need that: a stage falls through to its host path only when every device +//! arm of that stage declines in the same prove. + +use std::sync::atomic::{AtomicI64, Ordering}; + +use crate::Result; + +/// R3 barycentric entries (`barycentric_{base,ext3}_on_device{,_with_dev_inv_denoms}`). +pub static FAULT_BARYCENTRIC_STICKY: AtomicI64 = AtomicI64::new(-1); +/// R4 DEEP composition entries (`deep_composition_ext3*`). +pub static FAULT_DEEP_STICKY: AtomicI64 = AtomicI64::new(-1); +/// R2 comp-poly tree entries (`build_comp_poly_tree_from_{evals_ext3_keep,slabs_dev}`). +pub static FAULT_COMP_TREE_STICKY: AtomicI64 = AtomicI64::new(-1); + +/// Countdown check shared by the sticky hooks: negative = disarmed (the +/// production state); N > 0 counts down across calls and the Nth call — and +/// every call after it — returns Err (the counter parks at 0); 0 therefore +/// doubles as the "fired" marker. Disarm by storing -1. +pub fn check_sticky(counter: &AtomicI64) -> Result<()> { + let v = counter.load(Ordering::Relaxed); + if v < 0 { + return Ok(()); + } + if v > 0 { + counter.fetch_sub(1, Ordering::Relaxed); + } + if v <= 1 { + return Err(cudarc::driver::DriverError( + cudarc::driver::sys::CUresult::CUDA_ERROR_UNKNOWN, + )); + } + Ok(()) +} diff --git a/crypto/math-cuda/src/lib.rs b/crypto/math-cuda/src/lib.rs index 6b58d935b..d6f19b7c7 100644 --- a/crypto/math-cuda/src/lib.rs +++ b/crypto/math-cuda/src/lib.rs @@ -9,6 +9,8 @@ pub mod barycentric; pub mod constraint_interp; pub mod deep; pub mod device; +#[cfg(feature = "test-faults")] +pub mod faults; pub mod fri; pub mod inverse; pub mod lde; diff --git a/crypto/math-cuda/src/merkle.rs b/crypto/math-cuda/src/merkle.rs index c499df702..02532f6de 100644 --- a/crypto/math-cuda/src/merkle.rs +++ b/crypto/math-cuda/src/merkle.rs @@ -497,6 +497,8 @@ pub fn build_comp_poly_tree_from_slabs_dev( m: usize, lde_size: usize, ) -> Result { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_COMP_TREE_STICKY)?; assert!(m > 0); assert!(lde_size.is_power_of_two() && lde_size >= 2); assert_eq!(buf.len(), 3 * m * lde_size, "slab buffer shape"); @@ -544,6 +546,8 @@ pub fn build_comp_poly_tree_from_slabs_dev( pub fn build_comp_poly_tree_from_evals_ext3_keep( parts_interleaved: &[&[u64]], ) -> Result { + #[cfg(feature = "test-faults")] + crate::faults::check_sticky(&crate::faults::FAULT_COMP_TREE_STICKY)?; let (nodes_dev, num_leaves, stream) = build_comp_poly_tree_nodes_dev(parts_interleaved)?; let mut root = [0u8; 32]; stream.memcpy_dtoh(&nodes_dev.slice(0..32), &mut root)?; diff --git a/crypto/stark/src/gpu_lde.rs b/crypto/stark/src/gpu_lde.rs index 4aa756b25..166a20f7a 100644 --- a/crypto/stark/src/gpu_lde.rs +++ b/crypto/stark/src/gpu_lde.rs @@ -117,6 +117,7 @@ pub fn reset_all_gpu_call_counters() { GPU_DEVICE_ONLY_DOWNGRADES.store(0, Ordering::Relaxed); GPU_RESIDENT_AUX_RETRIES.store(0, Ordering::Relaxed); GPU_RESIDENT_AUX_DOWNGRADES.store(0, Ordering::Relaxed); + GPU_COMPOSITION_PARTS_DOWNLOADS.store(0, Ordering::Relaxed); } pub(crate) static GPU_EXTEND_HALVES_CALLS: AtomicU64 = AtomicU64::new(0); @@ -1464,16 +1465,17 @@ pub fn gpu_fri_calls() -> u64 { /// are counted here, so a single failed dispatch does not necessarily lower /// the total; R3's fallbacks are CPU-only, so a failure there does. pub(crate) static GPU_BATCH_INVERT_CALLS: AtomicU64 = AtomicU64::new(0); -/// R2 downgrades, and only those: times a device-only table fell back to the -/// host evaluator and had its resident LDEs downloaded into the host buffers -/// first ([`materialize_lde_trace_host`], the sole site that bumps this). -/// Nonzero means the device-only gate cleared a table whose R2 dispatch then -/// declined at runtime — the table continued host-backed, correct but slower — -/// so every count is a gate miss, and the fix is to mirror the missing -/// condition into the gate. The R1 resident-aux downgrade is counted by -/// [`GPU_RESIDENT_AUX_DOWNGRADES`] instead: it fires on tables the gate never -/// marked device-only, so summing the two would blame the gate for declines it -/// never made. +/// Device-only trace downgrades: times a device-only table fell back to a +/// host arm and had its resident LDEs downloaded into the host buffers first +/// ([`materialize_lde_trace_host`], the sole function that bumps this — +/// entered from the R2 host evaluator, the R3 barycentric arms and the R4 +/// DEEP host loop). Nonzero means the device-only gate cleared a table whose +/// downstream dispatch then declined at runtime — the table continued +/// host-backed, correct but slower — so every count is a gate miss, and the +/// fix is to mirror the missing condition into the gate. The R1 resident-aux +/// downgrade is counted by [`GPU_RESIDENT_AUX_DOWNGRADES`] instead: it fires +/// on tables the gate never marked device-only, so summing the two would +/// blame the gate for declines it never made. pub(crate) static GPU_DEVICE_ONLY_DOWNGRADES: AtomicU64 = AtomicU64::new(0); pub fn gpu_device_only_downgrades() -> u64 { GPU_DEVICE_ONLY_DOWNGRADES.load(Ordering::Relaxed) @@ -1493,6 +1495,18 @@ pub fn gpu_resident_aux_downgrades() -> u64 { GPU_RESIDENT_AUX_DOWNGRADES.load(Ordering::Relaxed) } +/// Times the composition-poly parts of a device-only table were downloaded +/// from the resident R2 handle so a host consumer could run +/// ([`download_composition_parts_host`], the sole site that bumps this). The +/// parts-side counterpart of [`GPU_DEVICE_ONLY_DOWNGRADES`]: that one covers +/// the trace LDEs, this one the H part evaluations whose R2 host drain was +/// skipped, when the R2 commit, the R3 parts OOD or the R4 DEEP H terms later +/// fall back to the host path. +pub(crate) static GPU_COMPOSITION_PARTS_DOWNLOADS: AtomicU64 = AtomicU64::new(0); +pub fn gpu_composition_parts_downloads() -> u64 { + GPU_COMPOSITION_PARTS_DOWNLOADS.load(Ordering::Relaxed) +} + /// Times the R1 resident-aux LDE declined and the prover drained the device to /// retry it (prover.rs). Nonzero means the device hit transient VRAM pressure — /// the retry is what keeps a decline from becoming a @@ -1725,6 +1739,82 @@ where true } +/// Parts counterpart of [`materialize_lde_trace_host`]: download the resident +/// composition-poly parts (de-interleaved ext3 slabs, natural evaluation +/// order) into per-part host Vecs. Serves the host consumers of the part +/// evaluations — the R2 Merkle commit, the R3 parts OOD and the R4 DEEP H +/// terms — when a device dispatch declines on a table whose R2 host drain was +/// skipped (device-only). Returns `None` when the handle cannot serve the +/// data: a non-ext3 field, a failed download or sync. +pub(crate) fn download_composition_parts_host( + h: &math_cuda::lde::GpuLdeExt3, + stream: &Arc, +) -> Option>>> +where + E: IsField + 'static, +{ + if TypeId::of::() != TypeId::of::() { + return None; + } + h.wait_ready_on(stream).ok()?; + let slabs = stream.clone_dtoh(h.buf.as_ref()).ok()?; + stream.synchronize().ok()?; + let (m, lde) = (h.m, h.lde_size); + if slabs.len() != m * lde * 3 { + return None; + } + let parts = (0..m) + .map(|p| { + let mut interleaved = vec![0u64; lde * 3]; + for k in 0..3 { + let slab = &slabs[(p * 3 + k) * lde..(p * 3 + k + 1) * lde]; + for (r, v) in slab.iter().enumerate() { + interleaved[r * 3 + k] = *v; + } + } + u64_to_ext3_vec::(&interleaved) + }) + .collect(); + GPU_COMPOSITION_PARTS_DOWNLOADS.fetch_add(1, Ordering::Relaxed); + Some(parts) +} + +/// Repopulate empty host part evaluations from the resident R2 parts handle +/// held by `lde_trace`. Already-populated evaluations are left untouched (the +/// R2 host drain ran, nothing is missing). Returns false only when the parts +/// are empty and the handle cannot serve them — a missing handle or bound +/// stream, a handle whose part count disagrees with the evaluations, or a +/// failed download — so the caller's abort carries the device-only contract's +/// message. +pub(crate) fn materialize_composition_parts_host( + lde_trace: &crate::trace::LDETraceTable, + evals: &mut [Vec>], +) -> bool +where + F: IsField + IsSubFieldOf + 'static, + E: IsField + 'static, +{ + if evals.first().is_none_or(|p| !p.is_empty()) { + return true; + } + let Some(h) = lde_trace.gpu_composition_parts() else { + return false; + }; + let Some(stream) = lde_trace.bound_stream() else { + return false; + }; + if h.m != evals.len() { + return false; + } + let Some(parts) = download_composition_parts_host::(h, &stream) else { + return false; + }; + for (dst, src) in evals.iter_mut().zip(parts) { + *dst = src; + } + true +} + pub fn gpu_batch_invert_calls() -> u64 { GPU_BATCH_INVERT_CALLS.load(Ordering::Relaxed) } @@ -1764,6 +1854,49 @@ pub fn inverse_fault_fired() -> bool { math_cuda::inverse::FAULT_INVERSE_REMAINING_UNTIL_ERR.load(Ordering::Relaxed) < 0 } +/// Test-only: make the Nth upcoming math-cuda barycentric dispatch — and +/// every one after it — return Err. Sticky, unlike the one-shot hooks above: +/// the retry arms would absorb a single-shot fault before the fall-through +/// could reach a device-only cliff site. Pass -1 to disarm (the production +/// state). Only available with the `test-cuda-faults` feature. +#[cfg(feature = "test-cuda-faults")] +pub fn schedule_barycentric_fault_sticky(n_calls_until_err: i64) { + math_cuda::faults::FAULT_BARYCENTRIC_STICKY.store(n_calls_until_err, Ordering::Relaxed); +} + +/// Test-only: whether the sticky barycentric fault reached its firing point +/// (the countdown parks at 0 once it fires and stays there until disarmed). +#[cfg(feature = "test-cuda-faults")] +pub fn barycentric_fault_fired() -> bool { + math_cuda::faults::FAULT_BARYCENTRIC_STICKY.load(Ordering::Relaxed) == 0 +} + +/// Sticky counterpart of [`schedule_barycentric_fault_sticky`] for the R4 +/// DEEP composition dispatches (`deep_composition_ext3*`). +#[cfg(feature = "test-cuda-faults")] +pub fn schedule_deep_fault_sticky(n_calls_until_err: i64) { + math_cuda::faults::FAULT_DEEP_STICKY.store(n_calls_until_err, Ordering::Relaxed); +} + +/// Test-only counterpart of [`barycentric_fault_fired`] for the DEEP hook. +#[cfg(feature = "test-cuda-faults")] +pub fn deep_fault_fired() -> bool { + math_cuda::faults::FAULT_DEEP_STICKY.load(Ordering::Relaxed) == 0 +} + +/// Sticky counterpart of [`schedule_barycentric_fault_sticky`] for the R2 +/// comp-poly tree builds (`build_comp_poly_tree_from_*`). +#[cfg(feature = "test-cuda-faults")] +pub fn schedule_comp_tree_fault_sticky(n_calls_until_err: i64) { + math_cuda::faults::FAULT_COMP_TREE_STICKY.store(n_calls_until_err, Ordering::Relaxed); +} + +/// Test-only counterpart of [`barycentric_fault_fired`] for the comp-tree hook. +#[cfg(feature = "test-cuda-faults")] +pub fn comp_tree_fault_fired() -> bool { + math_cuda::faults::FAULT_COMP_TREE_STICKY.load(Ordering::Relaxed) == 0 +} + /// R2 GPU dispatch: batched ext3 LDE over `parts_coefs` (composition-poly /// coefficient parts). Returns both the host LDE eval Vecs (needed for the /// R2 Merkle commit and R3 OOD path) and a device-resident `GpuLdeExt3` diff --git a/crypto/stark/src/prover.rs b/crypto/stark/src/prover.rs index 232e1faaf..1982f5b5c 100644 --- a/crypto/stark/src/prover.rs +++ b/crypto/stark/src/prover.rs @@ -1687,7 +1687,8 @@ pub trait IsStarkProver< ); } - let lde_composition_poly_parts_evaluations = if let Some(parts) = precomputed_parts { + #[cfg_attr(not(feature = "cuda"), allow(unused_mut))] + let mut lde_composition_poly_parts_evaluations = if let Some(parts) = precomputed_parts { parts } else if number_of_parts == 2 { // Direct quotient decomposition: avoid full-size iFFT by algebraically @@ -1772,6 +1773,15 @@ pub trait IsStarkProver< #[cfg(feature = "instruments")] let fft_dur = t_sub.elapsed(); + // Fold the R2 device composition parts handle into the session + // (resident R2 to R4) before the commit: the tree build below, its + // recovery, R3 OOD, R4 DEEP and the openings all read it from the + // trace. The host evaluations stay in `Round2` for the R4 openings. + #[cfg(feature = "cuda")] + if let Some(handle) = gpu_composition_parts { + round_1_result.lde_trace.set_gpu_composition_parts(handle); + } + #[cfg(feature = "instruments")] let t_sub = Instant::now(); // GPU fast path for the comp-poly Merkle commit: hash straight from @@ -1782,8 +1792,9 @@ pub trait IsStarkProver< // `Round2.gpu_composition_tree`. #[cfg(feature = "cuda")] let (composition_poly_merkle_tree, composition_poly_root, gpu_composition_tree) = - match gpu_composition_parts - .as_ref() + match round_1_result + .lde_trace + .gpu_composition_parts() .and_then(|h| { crate::gpu_lde::try_build_comp_poly_tree_gpu_from_dev::< FieldExtension, @@ -1802,19 +1813,23 @@ pub trait IsStarkProver< } None => { // The host part evals are empty under device-only (the R2 - // drain is skipped); abort with the device-only contract's - // message instead of a misleading EmptyCommitment. Gate on - // the parts the CPU fallback actually consumes, not on + // drain is skipped) — repopulate them from the resident + // parts handle rather than abort. Gate on the parts the + // CPU fallback actually consumes, not on // `host_trace_empty()`: the trace can stay device-resident // while these parts were downloaded to the host anyway (the // GPU decompose fell back to `decompose_and_extend_d2`), in - // which case this fallback is valid and must not panic. + // which case the materialize is a no-op. The assert fires + // only when the handle cannot serve the data. + let recovered = crate::gpu_lde::materialize_composition_parts_host( + &round_1_result.lde_trace, + &mut lde_composition_poly_parts_evaluations, + ); assert!( - lde_composition_poly_parts_evaluations - .first() - .is_none_or(|p| !p.is_empty()), - "R2 composition commit fell back to the host part evals, \ - but they are device-only (empty)" + recovered, + "R2 composition commit fell back to the host part evals \ + on a device-only table and the resident parts handle \ + could not be downloaded" ); let (tree, root) = crate::commitment::commit_bit_reversed( &lde_composition_poly_parts_evaluations, @@ -1837,13 +1852,6 @@ pub trait IsStarkProver< #[cfg(feature = "instruments")] crate::instruments::store_r2_sub(constraints_dur, fft_dur, merkle_dur); - // Fold the R2 device composition parts handle into the session (resident - // R2 to R4). The host evaluations stay in `Round2` for R4 openings. - #[cfg(feature = "cuda")] - if let Some(handle) = gpu_composition_parts { - round_1_result.lde_trace.set_gpu_composition_parts(handle); - } - Ok(Round2 { lde_composition_poly_evaluations: lde_composition_poly_parts_evaluations, composition_poly_merkle_tree, @@ -1857,8 +1865,8 @@ pub trait IsStarkProver< fn round_3_evaluate_polynomials_in_out_of_domain_element( air: &dyn AIR, domain: &Domain, - round_1_result: &Round1, - round_2_result: &Round2, + round_1_result: &mut Round1, + round_2_result: &mut Round2, z: &FieldElement, ) -> Round3 where @@ -1922,16 +1930,22 @@ pub trait IsStarkProver< Some(v) => v, None => { // The host part evals are empty under device-only (the R2 - // drain is skipped); reaching this arm there is a mis-gate. + // drain is skipped) — repopulate them from the resident parts + // handle rather than abort; the assert fires only when the + // handle cannot serve the data. #[cfg(feature = "cuda")] - assert!( - round_2_result - .lde_composition_poly_evaluations - .first() - .is_none_or(|p| !p.is_empty()), - "R3 parts OOD fell back to the host part evals, but they are \ - device-only (empty)" - ); + { + let recovered = crate::gpu_lde::materialize_composition_parts_host( + &round_1_result.lde_trace, + &mut round_2_result.lde_composition_poly_evaluations, + ); + assert!( + recovered, + "R3 parts OOD fell back to the host part evals on a \ + device-only table and the resident parts handle could \ + not be downloaded" + ); + } let comp_inv_denoms = math::polynomial::barycentric_inv_denoms(&z_power, &dc.points); round_2_result @@ -1958,7 +1972,7 @@ pub trait IsStarkProver< // === Trace polynomials: barycentric evaluation via LDE === let trace_ood_evaluations = crate::trace::get_trace_evaluations_from_lde( - &round_1_result.lde_trace, + &mut round_1_result.lde_trace, domain, z, &air.context().transition_offsets, @@ -1993,8 +2007,8 @@ pub trait IsStarkProver< fn round_4_compute_and_run_fri_on_the_deep_composition_polynomial( air: &dyn AIR, domain: &Domain, - round_1_result: &Round1, - round_2_result: &Round2, + round_1_result: &mut Round1, + round_2_result: &mut Round2, round_3_result: &Round3, z: &FieldElement, transcript: &mut (impl IsStarkTranscript + Clone), @@ -2086,7 +2100,7 @@ pub trait IsStarkProver< #[cfg(feature = "instruments")] let t_sub = Instant::now(); let deep_evals = Self::compute_deep_composition_poly_evaluations( - &round_1_result.lde_trace, + &mut round_1_result.lde_trace, round_2_result, round_3_result, z, @@ -2247,8 +2261,8 @@ pub trait IsStarkProver< #[allow(clippy::too_many_arguments)] fn compute_deep_composition_poly_evaluations( - lde_trace: &LDETraceTable, - round_2_result: &Round2, + lde_trace: &mut LDETraceTable, + round_2_result: &mut Round2, round_3_result: &Round3, z: &FieldElement, domain: &Domain, @@ -2360,14 +2374,32 @@ pub trait IsStarkProver< } // Reaching here means both GPU DEEP arms fell through to the host loop - // below (which reads `get_main`/`get_aux`). Under the device-only gate - // the host trace is empty, so a fall-through is a mis-gate or an - // unexpected GPU failure: hard-abort rather than read empty buffers. + // below, which reads the host trace (`get_main`/`get_aux`) AND the + // host part evals. Under the device-only gate either may be empty — + // download the resident data rather than abort; the asserts fire only + // when a resident handle cannot serve it. #[cfg(feature = "cuda")] - assert!( - !lde_trace.host_trace_empty(), - "R4 DEEP composition fell back to the host trace, but it is device-only (empty)" - ); + { + if lde_trace.host_trace_empty() { + let recovered = crate::gpu_lde::materialize_lde_trace_host(lde_trace); + assert!( + recovered, + "R4 DEEP composition fell back to the host trace on a \ + device-only table and the resident handles could not be \ + downloaded" + ); + } + let parts_recovered = crate::gpu_lde::materialize_composition_parts_host( + lde_trace, + &mut round_2_result.lde_composition_poly_evaluations, + ); + assert!( + parts_recovered, + "R4 DEEP composition fell back to the host part evals on a \ + device-only table and the resident parts handle could not be \ + downloaded" + ); + } // OOD column compression (Plonky3-style): precompute one value per eval point, // ood_compressed_k = Σ_j gamma[j][k] * ood[j][k]. @@ -3885,7 +3917,7 @@ pub trait IsStarkProver< coefficients.drain(..num_transition_constraints).collect(); let boundary_coefficients = coefficients; - let round_2_result = Self::round_2_compute_composition_polynomial( + let mut round_2_result = Self::round_2_compute_composition_polynomial( air, pub_inputs, domain, @@ -3914,7 +3946,7 @@ pub trait IsStarkProver< air, domain, round_1_result, - &round_2_result, + &mut round_2_result, &z, ); #[cfg(feature = "instruments")] @@ -3950,7 +3982,7 @@ pub trait IsStarkProver< air, domain, round_1_result, - &round_2_result, + &mut round_2_result, &round_3_result, &z, transcript, diff --git a/crypto/stark/src/tests/prover_tests.rs b/crypto/stark/src/tests/prover_tests.rs index ff4a0313c..ee41a4e4e 100644 --- a/crypto/stark/src/tests/prover_tests.rs +++ b/crypto/stark/src/tests/prover_tests.rs @@ -186,7 +186,7 @@ fn barycentric_trace_eval_matches_horner_trace_eval() { .collect(); // Build LDE trace table - let lde_trace = LDETraceTable::from_columns( + let mut lde_trace = LDETraceTable::from_columns( lde_evaluations, Vec::>::new(), air.step_size(), @@ -213,7 +213,7 @@ fn barycentric_trace_eval_matches_horner_trace_eval() { // Barycentric evaluation (new path) let result = - get_trace_evaluations_from_lde(&lde_trace, &domain, &z, &frame_offsets, step_size, &dc); + get_trace_evaluations_from_lde(&mut lde_trace, &domain, &z, &frame_offsets, step_size, &dc); assert_eq!(result.width, expected.width); assert_eq!(result.height, expected.height); diff --git a/crypto/stark/src/trace.rs b/crypto/stark/src/trace.rs index ccf35cca5..387fc0dfa 100644 --- a/crypto/stark/src/trace.rs +++ b/crypto/stark/src/trace.rs @@ -706,7 +706,7 @@ where /// has already derived these values (e.g., round_3 shares them with composition /// poly evaluation). pub fn get_trace_evaluations_from_lde( - lde_trace: &LDETraceTable, + lde_trace: &mut LDETraceTable, domain: &Domain, z: &FieldElement, frame_offsets: &[usize], @@ -813,15 +813,23 @@ where let main_evals: Vec> = if let Some(v) = main_gpu { v } else { - // Device-only tables have no host trace; a GPU fall-through here would - // read empty `main_data`. Hard-abort instead of a wrong OOD eval. The - // check is on the buffer itself, not the table-wide flag: a mixed - // state can leave a valid host copy on one side only. + // Device-only tables have no host trace; a GPU fall-through here + // would read empty `main_data` — download the resident LDEs rather + // than abort (the materialize fills both missing sides and clears + // the flag). The check is on the buffer itself, not the table-wide + // flag: a mixed state can leave a valid host copy on one side + // only. The assert fires only when the handles cannot serve the + // data. #[cfg(feature = "cuda")] - assert!( - lde_trace.num_main_cols() == 0 || !lde_trace.main_data.is_empty(), - "R3 barycentric (main) fell back to the host trace, but it is device-only (empty)" - ); + if lde_trace.num_main_cols() > 0 && lde_trace.main_data.is_empty() { + crate::gpu_lde::materialize_lde_trace_host(lde_trace); + assert!( + !lde_trace.main_data.is_empty(), + "R3 barycentric (main) fell back to the host trace on a \ + device-only table and the resident handles could not be \ + downloaded" + ); + } let inv_denoms_v = inv_denoms.get_or_insert_with(|| barycentric_inv_denoms(eval_point, &dc.points)); let col_scale = col_scale.get_or_insert_with(|| { @@ -873,14 +881,19 @@ where let aux_evals: Vec> = if let Some(v) = aux_gpu { v } else { - // Device-only tables have no host trace; a GPU fall-through here would - // read empty `aux_data`. Hard-abort instead of a wrong OOD eval. Same + // Device-only tables have no host trace; a GPU fall-through here + // would read empty `aux_data` — download rather than abort. Same // buffer-level check as the main arm: mixed states are valid here. #[cfg(feature = "cuda")] - assert!( - lde_trace.num_aux_cols() == 0 || !lde_trace.aux_data.is_empty(), - "R3 barycentric (aux) fell back to the host trace, but it is device-only (empty)" - ); + if lde_trace.num_aux_cols() > 0 && lde_trace.aux_data.is_empty() { + crate::gpu_lde::materialize_lde_trace_host(lde_trace); + assert!( + !lde_trace.aux_data.is_empty(), + "R3 barycentric (aux) fell back to the host trace on a \ + device-only table and the resident handles could not be \ + downloaded" + ); + } let inv_denoms_v = inv_denoms.get_or_insert_with(|| barycentric_inv_denoms(eval_point, &dc.points)); let col_scale = col_scale.get_or_insert_with(|| { From a8910205e86d19f8e56283147032ff723835533e Mon Sep 17 00:00:00 2001 From: Joaquin Carletti Date: Wed, 12 Aug 2026 15:53:36 -0300 Subject: [PATCH 2/5] test(gpu): exercise the cliff-site recoveries end to end Three prove+verify runs under sticky faults (comp-tree, barycentric, DEEP), each requiring the device-only path to fire on the warm-up and the recovery counters to move. --- prover/tests/cuda_fallback_tests.rs | 121 +++++++++++++++++++++++++++- 1 file changed, 120 insertions(+), 1 deletion(-) diff --git a/prover/tests/cuda_fallback_tests.rs b/prover/tests/cuda_fallback_tests.rs index 50eefc5ff..1c952d3c3 100644 --- a/prover/tests/cuda_fallback_tests.rs +++ b/prover/tests/cuda_fallback_tests.rs @@ -14,7 +14,10 @@ use lambda_vm_prover::test_utils::asm_elf_bytes; use lambda_vm_prover::{prove, verify}; -use stark::gpu_lde::{gpu_batch_invert_calls, gpu_fri_calls, reset_all_gpu_call_counters}; +use stark::gpu_lde::{ + gpu_batch_invert_calls, gpu_composition_parts_downloads, gpu_device_only_calls, + gpu_device_only_downgrades, gpu_fri_calls, reset_all_gpu_call_counters, +}; /// FRI commit-phase CPU fallback: when the GPU dispatch errors after the /// first transcript mutation, `try_fri_commit_gpu` must restore the @@ -119,3 +122,119 @@ fn gpu_batch_invert_fault_falls_back_to_cpu() { stark::gpu_lde::schedule_inverse_fault(-1); } + +/// Warm up with a clean prove and require the device-only residency path to +/// have fired: the cliff sites these recovery tests cover (empty host trace / +/// empty host part evals) only arm on device-only tables. +fn warm_up_requiring_device_only(elf: &[u8]) { + reset_all_gpu_call_counters(); + let _ = prove(elf).expect("warm-up"); + assert!( + gpu_device_only_calls() > 0, + "device-only residency never fired on the warm-up prove; the cliff \ + this test covers cannot arm (workload too small for the gate?)" + ); +} + +/// R2 comp-tree cliff recovery: with every `build_comp_poly_tree_from_*` +/// dispatch failing (sticky — both the from-dev and the host-upload arms must +/// decline in the same prove), the commit falls back to the CPU +/// `commit_bit_reversed`, whose input part evals are empty under device-only. +/// The recovery must download them from the resident R2 parts handle instead +/// of hard-aborting, and the proof must verify. +#[test] +#[ignore = "requires GPU + test-cuda-faults; run with --ignored --nocapture"] +fn gpu_comp_tree_fault_recovers_device_only_parts() { + let elf = asm_elf_bytes("fib_iterative_1M"); + warm_up_requiring_device_only(&elf); + + stark::gpu_lde::schedule_comp_tree_fault_sticky(1); + reset_all_gpu_call_counters(); + let recovered = prove(&elf).expect("prove with sticky comp-tree fault"); + assert!( + stark::gpu_lde::comp_tree_fault_fired(), + "injected comp-tree fault never fired" + ); + stark::gpu_lde::schedule_comp_tree_fault_sticky(-1); + assert!( + gpu_composition_parts_downloads() > 0, + "no composition parts were downloaded: the CPU commit either never \ + ran on a device-only table or read empty part evals" + ); + assert!( + verify(&recovered, &elf).expect("verify recovered"), + "post-recovery proof failed verification (comp-tree cliff)" + ); +} + +/// R3 barycentric cliff recovery: with every math-cuda barycentric dispatch +/// failing (sticky — the per-eval-point main and aux arms all retry it), the +/// trace OOD falls back to the host loop, which reads an empty host trace +/// under device-only, and the parts OOD falls back to the host part evals, +/// empty likewise. Both recoveries must download the resident data instead of +/// hard-aborting, and the proof must verify. +#[test] +#[ignore = "requires GPU + test-cuda-faults; run with --ignored --nocapture"] +fn gpu_barycentric_fault_recovers_device_only_trace() { + let elf = asm_elf_bytes("fib_iterative_1M"); + warm_up_requiring_device_only(&elf); + + stark::gpu_lde::schedule_barycentric_fault_sticky(1); + reset_all_gpu_call_counters(); + let recovered = prove(&elf).expect("prove with sticky barycentric fault"); + assert!( + stark::gpu_lde::barycentric_fault_fired(), + "injected barycentric fault never fired" + ); + stark::gpu_lde::schedule_barycentric_fault_sticky(-1); + assert!( + gpu_device_only_downgrades() > 0, + "no device-only table was downgraded: the R3 trace-OOD host loop \ + either never ran on one or read an empty host trace" + ); + assert!( + gpu_composition_parts_downloads() > 0, + "no composition parts were downloaded: the R3 parts-OOD host arm \ + either never ran on a device-only table or read empty part evals" + ); + assert!( + verify(&recovered, &elf).expect("verify recovered"), + "post-recovery proof failed verification (R3 barycentric cliff)" + ); +} + +/// R4 DEEP cliff recovery: with every math-cuda DEEP composition dispatch +/// failing (sticky — the fully-resident arm and both mixed arms must all +/// decline in the same prove), R4 falls back to the host DEEP loop, which +/// reads the host trace AND the host part evals — both empty under +/// device-only. The recovery must download both from the resident handles +/// instead of hard-aborting, and the proof must verify. +#[test] +#[ignore = "requires GPU + test-cuda-faults; run with --ignored --nocapture"] +fn gpu_deep_fault_recovers_device_only_trace_and_parts() { + let elf = asm_elf_bytes("fib_iterative_1M"); + warm_up_requiring_device_only(&elf); + + stark::gpu_lde::schedule_deep_fault_sticky(1); + reset_all_gpu_call_counters(); + let recovered = prove(&elf).expect("prove with sticky DEEP fault"); + assert!( + stark::gpu_lde::deep_fault_fired(), + "injected DEEP fault never fired" + ); + stark::gpu_lde::schedule_deep_fault_sticky(-1); + assert!( + gpu_device_only_downgrades() > 0, + "no device-only table was downgraded: the R4 DEEP host loop either \ + never ran on one or read an empty host trace" + ); + assert!( + gpu_composition_parts_downloads() > 0, + "no composition parts were downloaded: the R4 DEEP host loop either \ + never ran on a device-only table or read empty part evals" + ); + assert!( + verify(&recovered, &elf).expect("verify recovered"), + "post-recovery proof failed verification (R4 DEEP cliff)" + ); +} From 88aa4390d84d5d5477503a696913bf4d095d43df Mon Sep 17 00:00:00 2001 From: Joaquin Carletti Date: Tue, 18 Aug 2026 12:00:04 -0300 Subject: [PATCH 3/5] =?UTF-8?q?fix(gpu):=20address=20cliff-recovery=20revi?= =?UTF-8?q?ew=20=E2=80=94=20race-free=20sticky=20hook,=20parallel=20parts?= =?UTF-8?q?=20download?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-ups on the device-only cliff recovery: - check_sticky: collapse the load-then-decrement into one fetch_update that saturates at 0, so concurrent per-table dispatches can't underflow the counter — which would break both the sticky guarantee and the `== 0` fired check. - cuda_fallback_tests: disarm the sticky faults with a Drop guard, so a panic in prove or a failing assert can't leave one armed and cascade into the next test in the single-threaded binary. - download_composition_parts_host: de-interleave under rayon and reinterpret the u64 buffer in place, matching materialize_lde_trace_host instead of copying again through u64_to_ext3_vec — this path fires often under VRAM pressure. - Docs: the device-only downgrade counter now also covers transient device declines, not only gate misses; note the new &mut contract on get_trace_evaluations_from_lde. --- crypto/math-cuda/src/faults.rs | 27 +++++++++++----- crypto/stark/src/gpu_lde.rs | 49 +++++++++++++++++++++++------ crypto/stark/src/trace.rs | 5 +++ prover/tests/cuda_fallback_tests.rs | 23 ++++++++++++-- 4 files changed, 83 insertions(+), 21 deletions(-) diff --git a/crypto/math-cuda/src/faults.rs b/crypto/math-cuda/src/faults.rs index 5e9fe4dcf..d9cb93bc6 100644 --- a/crypto/math-cuda/src/faults.rs +++ b/crypto/math-cuda/src/faults.rs @@ -22,15 +22,26 @@ pub static FAULT_COMP_TREE_STICKY: AtomicI64 = AtomicI64::new(-1); /// production state); N > 0 counts down across calls and the Nth call — and /// every call after it — returns Err (the counter parks at 0); 0 therefore /// doubles as the "fired" marker. Disarm by storing -1. +/// +/// The transition is a single `fetch_update`, so concurrent table dispatches +/// (the prover runs a rayon task per table) cannot race the load against the +/// decrement: the counter saturates at 0 and never underflows past it, which +/// keeps both the sticky guarantee and the `== 0` fired check sound. A CAS +/// loser's returned prior value can lag by one, so the Err decision reads the +/// post-update state instead. pub fn check_sticky(counter: &AtomicI64) -> Result<()> { - let v = counter.load(Ordering::Relaxed); - if v < 0 { - return Ok(()); - } - if v > 0 { - counter.fetch_sub(1, Ordering::Relaxed); - } - if v <= 1 { + // One atomic transition, so concurrent dispatches saturate at 0 rather + // than underflowing: a decrement only happens from a positive value. + let fired = counter + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |v| match v { + n if n < 0 => None, // disarmed: never fires + 0 => None, // already parked: stay fired (sticky) + _ => Some(v - 1), // count down toward the parked 0 + }) + // Ok(prev): this call decremented — the 1 → 0 step fires. + // Err(cur): no-op — fires only if already parked at 0. + .map_or_else(|cur| cur == 0, |prev| prev <= 1); + if fired { return Err(cudarc::driver::DriverError( cudarc::driver::sys::CUresult::CUDA_ERROR_UNKNOWN, )); diff --git a/crypto/stark/src/gpu_lde.rs b/crypto/stark/src/gpu_lde.rs index 166a20f7a..a1ec18fa7 100644 --- a/crypto/stark/src/gpu_lde.rs +++ b/crypto/stark/src/gpu_lde.rs @@ -1471,11 +1471,13 @@ pub(crate) static GPU_BATCH_INVERT_CALLS: AtomicU64 = AtomicU64::new(0); /// entered from the R2 host evaluator, the R3 barycentric arms and the R4 /// DEEP host loop). Nonzero means the device-only gate cleared a table whose /// downstream dispatch then declined at runtime — the table continued -/// host-backed, correct but slower — so every count is a gate miss, and the -/// fix is to mirror the missing condition into the gate. The R1 resident-aux -/// downgrade is counted by [`GPU_RESIDENT_AUX_DOWNGRADES`] instead: it fires -/// on tables the gate never marked device-only, so summing the two would -/// blame the gate for declines it never made. +/// host-backed, correct but slower. A count is either a gate miss (a static +/// condition worth mirroring into the gate) or a transient device decline +/// (VRAM pressure), which by definition cannot be gated out — see +/// [`materialize_lde_trace_host`]'s own note. The R1 resident-aux downgrade +/// is counted by [`GPU_RESIDENT_AUX_DOWNGRADES`] instead: it fires on tables +/// the gate never marked device-only, so summing the two would blame the gate +/// for declines it never made. pub(crate) static GPU_DEVICE_ONLY_DOWNGRADES: AtomicU64 = AtomicU64::new(0); pub fn gpu_device_only_downgrades() -> u64 { GPU_DEVICE_ONLY_DOWNGRADES.load(Ordering::Relaxed) @@ -1763,16 +1765,43 @@ where if slabs.len() != m * lde * 3 { return None; } + // Per part: de-interleave the 3 slabs into row-major ext3 and reinterpret + // the u64 buffer in place — mirroring `materialize_lde_trace_host` rather + // than copying again through `u64_to_ext3_vec`. The row fill is parallel; + // this path fires often under VRAM pressure and otherwise dominates the + // D2H it follows. let parts = (0..m) .map(|p| { let mut interleaved = vec![0u64; lde * 3]; - for k in 0..3 { - let slab = &slabs[(p * 3 + k) * lde..(p * 3 + k + 1) * lde]; - for (r, v) in slab.iter().enumerate() { - interleaved[r * 3 + k] = *v; + #[cfg(feature = "parallel")] + interleaved + .par_chunks_exact_mut(3) + .enumerate() + .for_each(|(r, dst)| { + for (k, d) in dst.iter_mut().enumerate() { + *d = slabs[(p * 3 + k) * lde + r]; + } + }); + #[cfg(not(feature = "parallel"))] + for (r, dst) in interleaved.chunks_exact_mut(3).enumerate() { + for (k, d) in dst.iter_mut().enumerate() { + *d = slabs[(p * 3 + k) * lde + r]; } } - u64_to_ext3_vec::(&interleaved) + // SAFETY: E == Ext3 per the tower check above; FieldElement + // is [u64; 3]. `vec![0u64; lde*3]` has len == capacity == lde*3. + unsafe { + let mut v = std::mem::ManuallyDrop::new(interleaved); + debug_assert!( + v.len().is_multiple_of(3) && v.capacity().is_multiple_of(3), + "interleaved len/capacity must be a multiple of 3 for Fp3 reinterpret" + ); + Vec::from_raw_parts( + v.as_mut_ptr() as *mut FieldElement, + v.len() / 3, + v.capacity() / 3, + ) + } }) .collect(); GPU_COMPOSITION_PARTS_DOWNLOADS.fetch_add(1, Ordering::Relaxed); diff --git a/crypto/stark/src/trace.rs b/crypto/stark/src/trace.rs index 387fc0dfa..b1f8e9bf3 100644 --- a/crypto/stark/src/trace.rs +++ b/crypto/stark/src/trace.rs @@ -705,6 +705,11 @@ where /// Accepts a [`DomainConstants`] to avoid redundant computation when the caller /// has already derived these values (e.g., round_3 shares them with composition /// poly evaluation). +/// +/// Takes `lde_trace` by `&mut` so a device-only table whose GPU barycentric arm +/// declines can recover in place: the arm downloads the resident LDEs into the +/// host buffers ([`crate::gpu_lde::materialize_lde_trace_host`]) and continues +/// on the host path, rather than reading an empty host trace. pub fn get_trace_evaluations_from_lde( lde_trace: &mut LDETraceTable, domain: &Domain, diff --git a/prover/tests/cuda_fallback_tests.rs b/prover/tests/cuda_fallback_tests.rs index 1c952d3c3..cbeaaea50 100644 --- a/prover/tests/cuda_fallback_tests.rs +++ b/prover/tests/cuda_fallback_tests.rs @@ -123,6 +123,20 @@ fn gpu_batch_invert_fault_falls_back_to_cpu() { stark::gpu_lde::schedule_inverse_fault(-1); } +/// Disarms every sticky fault on drop. The hooks are process-global and these +/// tests run `--test-threads=1`, so a panic inside `prove` or a failing assert +/// must not leave a fault armed — a later test would otherwise prove with all +/// of that stage's dispatches failing and cascade into confusing failures. The +/// one-shot hooks self-heal; the sticky ones need this. +struct StickyFaultGuard; +impl Drop for StickyFaultGuard { + fn drop(&mut self) { + stark::gpu_lde::schedule_comp_tree_fault_sticky(-1); + stark::gpu_lde::schedule_barycentric_fault_sticky(-1); + stark::gpu_lde::schedule_deep_fault_sticky(-1); + } +} + /// Warm up with a clean prove and require the device-only residency path to /// have fired: the cliff sites these recovery tests cover (empty host trace / /// empty host part evals) only arm on device-only tables. @@ -148,6 +162,8 @@ fn gpu_comp_tree_fault_recovers_device_only_parts() { let elf = asm_elf_bytes("fib_iterative_1M"); warm_up_requiring_device_only(&elf); + // Disarms on scope exit — including a panic in `prove` or a failing assert. + let _guard = StickyFaultGuard; stark::gpu_lde::schedule_comp_tree_fault_sticky(1); reset_all_gpu_call_counters(); let recovered = prove(&elf).expect("prove with sticky comp-tree fault"); @@ -155,7 +171,6 @@ fn gpu_comp_tree_fault_recovers_device_only_parts() { stark::gpu_lde::comp_tree_fault_fired(), "injected comp-tree fault never fired" ); - stark::gpu_lde::schedule_comp_tree_fault_sticky(-1); assert!( gpu_composition_parts_downloads() > 0, "no composition parts were downloaded: the CPU commit either never \ @@ -179,6 +194,8 @@ fn gpu_barycentric_fault_recovers_device_only_trace() { let elf = asm_elf_bytes("fib_iterative_1M"); warm_up_requiring_device_only(&elf); + // Disarms on scope exit — including a panic in `prove` or a failing assert. + let _guard = StickyFaultGuard; stark::gpu_lde::schedule_barycentric_fault_sticky(1); reset_all_gpu_call_counters(); let recovered = prove(&elf).expect("prove with sticky barycentric fault"); @@ -186,7 +203,6 @@ fn gpu_barycentric_fault_recovers_device_only_trace() { stark::gpu_lde::barycentric_fault_fired(), "injected barycentric fault never fired" ); - stark::gpu_lde::schedule_barycentric_fault_sticky(-1); assert!( gpu_device_only_downgrades() > 0, "no device-only table was downgraded: the R3 trace-OOD host loop \ @@ -215,6 +231,8 @@ fn gpu_deep_fault_recovers_device_only_trace_and_parts() { let elf = asm_elf_bytes("fib_iterative_1M"); warm_up_requiring_device_only(&elf); + // Disarms on scope exit — including a panic in `prove` or a failing assert. + let _guard = StickyFaultGuard; stark::gpu_lde::schedule_deep_fault_sticky(1); reset_all_gpu_call_counters(); let recovered = prove(&elf).expect("prove with sticky DEEP fault"); @@ -222,7 +240,6 @@ fn gpu_deep_fault_recovers_device_only_trace_and_parts() { stark::gpu_lde::deep_fault_fired(), "injected DEEP fault never fired" ); - stark::gpu_lde::schedule_deep_fault_sticky(-1); assert!( gpu_device_only_downgrades() > 0, "no device-only table was downgraded: the R4 DEEP host loop either \ From b64e7a3695354aaf97a555841f2ae47462cbca09 Mon Sep 17 00:00:00 2001 From: Joaquin Carletti Date: Tue, 18 Aug 2026 12:14:44 -0300 Subject: [PATCH 4/5] style(gpu): rustfmt check_sticky and correct its doc cargo fmt collapses the aligned match-arm comments (the CI lint failure); also drop a stale doc sentence describing an earlier post-load variant that the fetch_update version does not use. --- crypto/math-cuda/src/faults.rs | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/crypto/math-cuda/src/faults.rs b/crypto/math-cuda/src/faults.rs index d9cb93bc6..34599b908 100644 --- a/crypto/math-cuda/src/faults.rs +++ b/crypto/math-cuda/src/faults.rs @@ -25,18 +25,19 @@ pub static FAULT_COMP_TREE_STICKY: AtomicI64 = AtomicI64::new(-1); /// /// The transition is a single `fetch_update`, so concurrent table dispatches /// (the prover runs a rayon task per table) cannot race the load against the -/// decrement: the counter saturates at 0 and never underflows past it, which -/// keeps both the sticky guarantee and the `== 0` fired check sound. A CAS -/// loser's returned prior value can lag by one, so the Err decision reads the -/// post-update state instead. +/// decrement: each caller walks the counter one step (the closure returns +/// `None` at `<= 0`, so it parks at 0 and never underflows), which keeps both +/// the sticky guarantee and the `== 0` fired check sound. The fire decision +/// reads `fetch_update`'s own result — `Ok(prev)` for the call that +/// decremented, `Err(cur)` for a no-op — so no second load is needed. pub fn check_sticky(counter: &AtomicI64) -> Result<()> { // One atomic transition, so concurrent dispatches saturate at 0 rather // than underflowing: a decrement only happens from a positive value. let fired = counter .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |v| match v { - n if n < 0 => None, // disarmed: never fires - 0 => None, // already parked: stay fired (sticky) - _ => Some(v - 1), // count down toward the parked 0 + n if n < 0 => None, // disarmed: never fires + 0 => None, // already parked: stay fired (sticky) + _ => Some(v - 1), // count down toward the parked 0 }) // Ok(prev): this call decremented — the 1 → 0 step fires. // Err(cur): no-op — fires only if already parked at 0. From 7863a6740d0d00c1cfc8410f9a0b61959b27ee6e Mon Sep 17 00:00:00 2001 From: Mauro Toscano <12560266+MauroToscano@users.noreply.github.com> Date: Tue, 18 Aug 2026 17:06:38 -0300 Subject: [PATCH 5/5] test(gpu): assert the parts-download counter is zero on the happy path (#938) The device-only cliff recoveries replace hard aborts with a silent download-and-continue, so the counters are now the only thing that surfaces a gate/dispatch lockstep break. GPU_DEVICE_ONLY_DOWNGRADES (trace side) already has its == 0 guard here; its parts-side counterpart did not, and its only readers were the > 0 assertions in cuda_fallback_tests, which run with a fault deliberately armed. Without this, a decline in the R2 comp-poly tree build on a device-only table recovers, verifies and passes green, while every such table pays a full parts D2H plus a CPU commit_bit_reversed and loses the resident composition tree. The R4 DEEP site is already covered transitively (it needs the trace to be device-only too, which moves the trace counter), so this closes the R2 commit and R3 parts-OOD sites. Zero is the right expectation: materialize_composition_parts_host early-returns without bumping when the part evals are already populated, so the counter only moves for a device-only table that had to pull its parts back. The message names both causes rather than blaming the gate, matching the counter's own doc, which now allows a transient VRAM decline as well as a gate miss. --- prover/tests/cuda_path_integration.rs | 28 ++++++++++++++++++--------- 1 file changed, 19 insertions(+), 9 deletions(-) diff --git a/prover/tests/cuda_path_integration.rs b/prover/tests/cuda_path_integration.rs index 7ae50afad..29f0070d8 100644 --- a/prover/tests/cuda_path_integration.rs +++ b/prover/tests/cuda_path_integration.rs @@ -181,15 +181,16 @@ fn gpu_opening_gather_fires_and_verifies() { /// The full-residency Stage-3 device-only path fires: at least one table keeps /// its round-1 LDE device-resident (the host D2H is skipped), and the proof -/// still verifies. This exercises every `host_trace_empty` hard-abort guard on -/// the happy path (none may fire) plus the GPU-only R2/R3/R4 paths reading the -/// device LDE with no host trace behind them. A regression that silently -/// reverts to the host D2H drops the counter to 0 (while the proof would still -/// verify). A mis-gate that forces a host fallback shows up one of two ways: -/// at R3/R4 it panics one of the guards, while at R2 and the R1 resident-aux -/// commit it recovers silently and is caught by the downgrade-counter -/// assertions below — one per site, since the R1 counter also covers tables the -/// device-only gate never cleared. +/// still verifies. This exercises the GPU-only R2/R3/R4 paths reading the +/// device LDE with no host trace behind them, plus the `host_trace_empty` +/// hard-abort guards that remain on the R4 opening path (none may fire). A +/// regression that silently reverts to the host D2H drops the counter to 0 +/// (while the proof would still verify). A mis-gate that forces a host +/// fallback does not panic at the R2 commit, R3 or the R4 DEEP loop: those +/// sites download the resident data and continue host-backed, so the counter +/// assertions below are the only thing that surfaces one — one per site, since +/// the R1 counter also covers tables the device-only gate never cleared, and +/// the parts counter covers the H part evaluations rather than the trace. #[test] #[ignore = "requires GPU; run with --ignored --nocapture"] fn gpu_device_only_residency_fires_and_verifies() { @@ -207,6 +208,15 @@ fn gpu_device_only_residency_fires_and_verifies() { path (its R2 dispatch declined at runtime: the gate should mirror the \ missing condition)" ); + assert_eq!( + stark::gpu_lde::gpu_composition_parts_downloads(), + 0, + "a device-only table's composition-poly parts were downloaded back to \ + the host on the happy path (the R2 commit, the R3 parts OOD or the R4 \ + DEEP H terms fell back to the host part evals: either the gate should \ + mirror a missing dispatch condition, or the dispatch declined \ + transiently under VRAM pressure)" + ); assert_eq!( stark::gpu_lde::gpu_resident_aux_downgrades(), 0,