diff --git a/sp1-gpu/crates/cuda/src/task.rs b/sp1-gpu/crates/cuda/src/task.rs index a503a86..813016b 100644 --- a/sp1-gpu/crates/cuda/src/task.rs +++ b/sp1-gpu/crates/cuda/src/task.rs @@ -149,7 +149,15 @@ pub enum GlobalTaskPoolBuildError { impl TaskPoolBuilder { pub fn new() -> Self { - Self { capacity: None, device: CudaDevice(0), mem_release_threshold: u64::MAX } + // Igneum prover-floor patch: upstream keeps every freed device allocation in the pool for the process's + // life (threshold u64::MAX), so the prover holds its high-water mark between shards on a card it shares + // with a miner. `SP1_GPU_MEM_RELEASE_THRESHOLD=` sets the pool's release threshold (0 returns + // freed memory to the driver at once); unset, upstream's behaviour. + let mem_release_threshold = std::env::var("SP1_GPU_MEM_RELEASE_THRESHOLD") + .ok() + .and_then(|s| s.parse::().ok()) + .unwrap_or(u64::MAX); + Self { capacity: None, device: CudaDevice(0), mem_release_threshold } } pub fn num_tasks(mut self, num_tasks: usize) -> Self { diff --git a/sp1-gpu/crates/jagged_tracegen/src/lib.rs b/sp1-gpu/crates/jagged_tracegen/src/lib.rs index 579f70a..2264044 100644 --- a/sp1-gpu/crates/jagged_tracegen/src/lib.rs +++ b/sp1-gpu/crates/jagged_tracegen/src/lib.rs @@ -481,6 +481,33 @@ async fn device_preprocessed_tracegen>( named_traces } +/// Igneum prover-floor patch: the dense elements a set of traces will occupy once `generate_jagged_traces` +/// has laid them out, that is the sum of their buffers padded to the next multiple of 2^log_stacking_height +/// (the "final padding" step below). Each phase (preprocessed, then main) is padded on its own. +pub fn padded_trace_elements( + traces: &BTreeMap>, + log_stacking_height: u32, +) -> usize { + let total: usize = traces + .values() + .map(|t| match t { + Trace::Real(trace) => trace.guts().as_buffer().len(), + Trace::Padding(_) => 0, + }) + .sum(); + total.next_multiple_of(1 << log_stacking_height) +} + +/// Igneum prover-floor patch: the capacity to allocate for a trace set: the exact padded size plus one +/// stacking height of slack, never more than the prover's `max_trace_size`. `SP1_GPU_FLOOR_EXACT=0` restores +/// upstream's full-capacity allocation. +fn floor_capacity(max_trace_size: usize, needed: usize, log_stacking_height: u32) -> usize { + if std::env::var("SP1_GPU_FLOOR_EXACT").map(|v| v == "0").unwrap_or(false) { + return max_trace_size; + } + max_trace_size.min(needed + (1 << log_stacking_height)) +} + async fn allocate_and_initialize_traces( preprocessed_traces: BTreeMap>, max_trace_size: usize, @@ -494,6 +521,11 @@ async fn allocate_and_initialize_traces( let total_gb = total_bytes as f64 / (1 << 30) as f64; tracing::debug!("Allocating {:?} GB of traces", total_gb); + if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() { + eprintln!( + "FLOOR tracegen alloc capacity_elements={max_trace_size} bytes={total_bytes} ({total_gb:.3} GB)" + ); + } let mut dense_data: Buffer = Buffer::with_capacity_in(max_trace_size, backend.clone()); let mut col_index: Buffer = @@ -677,9 +709,14 @@ pub async fn setup_tracegen>( let preprocessed_traces = device_preprocessed_tracegen(program, host_phase_tracegen, backend).await; + let capacity = floor_capacity( + max_trace_size, + padded_trace_elements(&preprocessed_traces, log_stacking_height), + log_stacking_height, + ); let jagged_traces = allocate_and_initialize_traces( preprocessed_traces, - max_trace_size, + capacity, log_stacking_height, max_log_row_count, backend, @@ -906,6 +943,11 @@ pub async fn main_tracegen, A: CudaTracegenAir>( log_chip_stats(machine, &chip_set, &traces); + // Igneum prover-floor patch: the key's buffer is sized to its preprocessed traces at setup (upstream sized it + // for a whole shard), so grow it here to what this shard needs before the main traces are appended: a bigger + // dense buffer and column index, the preprocessed region copied device to device, swapped into the key. + grow_for_main(&mut jagged_traces.preprocessed_traces, &traces, log_stacking_height, backend); + copy_main_jagged_traces( traces, &mut jagged_traces.preprocessed_traces, @@ -918,6 +960,61 @@ pub async fn main_tracegen, A: CudaTracegenAir>( (public_values, chip_set, permit) } +/// Igneum prover-floor patch: see `main_tracegen`. The need is the preprocessed phase as laid out (its padded +/// end, `preprocessed_offset`) plus the main traces padded to the stacking height plus one stacking height of +/// slack; a buffer at least that big is left alone. The process aborts, loudly, if the copy cannot be made, +/// because a panic inside a prover task is what left sweep 2 hanging on the client's socket. +fn grow_for_main( + jagged: &mut JaggedTraceMle, + main_traces: &BTreeMap>, + log_stacking_height: u32, + backend: &TaskScope, +) { + let pre_end = jagged.dense().preprocessed_offset; + let needed = pre_end + + padded_trace_elements(main_traces, log_stacking_height) + + (1 << log_stacking_height); + let have = jagged.dense().dense.capacity(); + if have >= needed { + return; + } + let mut new_dense: Buffer = Buffer::with_capacity_in(needed, backend.clone()); + let mut new_col_index: Buffer = + Buffer::with_capacity_in(needed >> 1, backend.clone()); + unsafe { + new_dense.assume_init(); + new_col_index.assume_init(); + } + { + let JaggedMle { dense_data, col_index, .. } = &mut **jagged; + let src_dense: &Slice<_, _> = &dense_data.dense[..pre_end]; + let dst_dense: &mut Slice<_, _> = &mut new_dense[..pre_end]; + let src_col: &Slice<_, _> = &col_index[..pre_end >> 1]; + let dst_col: &mut Slice<_, _> = &mut new_col_index[..pre_end >> 1]; + unsafe { + if dst_dense.copy_from_slice(src_dense, backend).is_err() + || dst_col.copy_from_slice(src_col, backend).is_err() + { + eprintln!("FLOOR grow FAILED: could not copy the preprocessed region ({pre_end} elements) into the grown buffer ({needed} elements); aborting instead of hanging"); + std::process::abort(); + } + } + } + if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() { + eprintln!( + "FLOOR grow key buffer {have} -> {needed} elements (preprocessed {pre_end}, {} bytes)", + needed * 6 + ); + } + let JaggedMle { dense_data, col_index, .. } = &mut **jagged; + dense_data.dense = new_dense; + *col_index = new_col_index; + unsafe { + dense_data.dense.set_len(pre_end); + col_index.set_len(pre_end >> 1); + } +} + #[allow(clippy::too_many_arguments)] pub async fn main_tracegen_permit, A: CudaTracegenAir>( machine: &Machine, @@ -984,9 +1081,15 @@ pub async fn full_tracegen>( log_chip_stats(machine, &chip_set, &main_traces); + let capacity = floor_capacity( + max_trace_size, + padded_trace_elements(&preprocessed_traces, log_stacking_height) + + padded_trace_elements(&main_traces, log_stacking_height), + log_stacking_height, + ); let mut jagged_mle = allocate_and_initialize_traces( preprocessed_traces, - max_trace_size, + capacity, log_stacking_height, max_log_row_count, backend, @@ -1002,6 +1105,18 @@ pub async fn full_tracegen>( ) .await; + if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() { + let dense = jagged_mle.dense(); + let (free, total) = sp1_gpu_cudart::cuda_memory_info().unwrap_or((0, 0)); + eprintln!( + "FLOOR tracegen used preprocessed_elements={} main_elements={} dense_len={} capacity_elements={capacity} max_trace_size={max_trace_size} device_used_mib={}", + dense.preprocessed_offset, + dense.main_size(), + dense.dense.len(), + (total - free) >> 20 + ); + } + (public_values, jagged_mle, chip_set, permit) } diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/prover_components/src/builder.rs index 5dccd9d..574d4fa 100644 --- a/sp1-gpu/crates/prover_components/src/builder.rs +++ b/sp1-gpu/crates/prover_components/src/builder.rs @@ -23,28 +23,75 @@ use crate::{ SP1CudaProverComponents, }; +/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger +/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a +/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. +/// The proof format, the verifier and the program ids do not change: the element threshold only decides where +/// the executor splits shards, as upstream's own 24 GB tier already does. +fn env_usize(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} + +fn env_f64(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} + +/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it +/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16). +pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 { + if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) { + ELEMENT_THRESHOLD + } else if gpu_memory_gb >= 24 { + ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24) + } else if gpu_memory_gb >= 18 { + (1 << 27) + (1 << 26) + } else { + 1 << 27 + } +} + +/// The recursion trace allocation (elements) for a memory budget. +pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize { + if gpu_memory_gb >= 24 { + RECURSION_TRACE_ALLOCATION + } else { + RECURSION_TRACE_ALLOCATION + } +} + +pub fn gpu_memory_gb() -> usize { + let gb = 1024.0 * 1024.0 * 1024.0; + match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { + Some(b) => (b.ceil() as usize) + 4, + None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4, + } +} + +pub fn recursion_trace_allocation() -> usize { + env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION") + .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb())) +} + pub fn local_gpu_opts() -> SP1CoreOpts { let mut opts = SP1CoreOpts::default(); let log2_shard_size = 24; opts.shard_size = 1 << log2_shard_size; - let gb = 1024.0 * 1024.0 * 1024.0; - - // Get the amount of memory on the GPU. - let gpu_memory_gb: usize = (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4; - - if gpu_memory_gb < 24 { - panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"); - } + // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given. + let gpu_memory_gb = gpu_memory_gb(); - let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 { - ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24) - } else { - ELEMENT_THRESHOLD + let shard_threshold = match env_usize("SP1_GPU_ELEMENT_THRESHOLD") { + Some(t) => t as u64, + None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards), }; + let height_threshold = opts.sharding_threshold.height_threshold; - tracing::debug!("Shard threshold: {shard_threshold}"); + eprintln!( + "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", + recursion_trace_allocation(), + opts.full_size_shards + ); opts.sharding_threshold.element_threshold = shard_threshold; opts.global_dependencies_opt = true; @@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier( ) { let recursion_verifier = SP1CudaProverComponents::compress_verifier(); ( - new_cuda_prover(&recursion_verifier, RECURSION_TRACE_ALLOCATION, 4, false, false, scope) + new_cuda_prover(&recursion_verifier, recursion_trace_allocation(), 4, false, false, scope) .await, recursion_verifier, ) diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs index 4035f1f..0d0d907 100644 --- a/sp1-gpu/crates/server/src/server.rs +++ b/sp1-gpu/crates/server/src/server.rs @@ -157,6 +157,7 @@ impl Server { }; let pk = CachedProgram { elf: Arc::new(Elf::Dynamic(elf.into())), vk: vk.clone() }; ctx.pk_cache.insert(elf_hash, pk); + floor_memory_line("after setup"); Response::Setup { id: elf_hash, vk } } Request::Destroy { key } => { @@ -177,15 +178,31 @@ impl Server { ); }; let context = SP1Context::builder().proof_nonce(proof_nonce).build(); - match prover.prove_with_mode(&cached.elf, stdin, context, mode).await { + let started = std::time::Instant::now(); + let response = match prover.prove_with_mode(&cached.elf, stdin, context, mode).await { Ok(proof) => Response::Proof { proof }, Err(e) => Response::ProverError(e.to_string()), - } + }; + floor_memory_line(&format!("after prove {:?} in {:.1} s", mode, started.elapsed().as_secs_f64())); + response } } } } +/// Igneum prover-floor patch: the device memory in use (total minus free, as the driver reports it) at the +/// points that bound a proof, so a run's log carries the terms of the peak without a sampler. +fn floor_memory_line(what: &str) { + if let Ok((free, total)) = sp1_gpu_cudart::cuda_memory_info() { + eprintln!( + "FLOOR memory {what}: device_used_mib={} free_mib={} total_mib={}", + (total - free) >> 20, + free >> 20, + total >> 20 + ); + } +} + fn sha256(data: &[u8]) -> [u8; 32] { use sha2::{Digest, Sha256}; let mut hasher = Sha256::new();