345 lines
15 KiB
Diff
345 lines
15 KiB
Diff
diff --git a/sp1-gpu/crates/cuda/src/task.rs b/sp1-gpu/crates/cuda/src/task.rs
|
|
index a503a86..813016b 100644
|
|
--- a/sp1-gpu/crates/cuda/src/task.rs
|
|
+++ b/sp1-gpu/crates/cuda/src/task.rs
|
|
@@ -149,7 +149,15 @@ pub enum GlobalTaskPoolBuildError {
|
|
|
|
impl TaskPoolBuilder {
|
|
pub fn new() -> Self {
|
|
- Self { capacity: None, device: CudaDevice(0), mem_release_threshold: u64::MAX }
|
|
+ // Igneum prover-floor patch: upstream keeps every freed device allocation in the pool for the process's
|
|
+ // life (threshold u64::MAX), so the prover holds its high-water mark between shards on a card it shares
|
|
+ // with a miner. `SP1_GPU_MEM_RELEASE_THRESHOLD=<bytes>` sets the pool's release threshold (0 returns
|
|
+ // freed memory to the driver at once); unset, upstream's behaviour.
|
|
+ let mem_release_threshold = std::env::var("SP1_GPU_MEM_RELEASE_THRESHOLD")
|
|
+ .ok()
|
|
+ .and_then(|s| s.parse::<u64>().ok())
|
|
+ .unwrap_or(u64::MAX);
|
|
+ Self { capacity: None, device: CudaDevice(0), mem_release_threshold }
|
|
}
|
|
|
|
pub fn num_tasks(mut self, num_tasks: usize) -> Self {
|
|
diff --git a/sp1-gpu/crates/jagged_tracegen/src/lib.rs b/sp1-gpu/crates/jagged_tracegen/src/lib.rs
|
|
index 579f70a..2264044 100644
|
|
--- a/sp1-gpu/crates/jagged_tracegen/src/lib.rs
|
|
+++ b/sp1-gpu/crates/jagged_tracegen/src/lib.rs
|
|
@@ -481,6 +481,33 @@ async fn device_preprocessed_tracegen<A: CudaTracegenAir<Felt>>(
|
|
named_traces
|
|
}
|
|
|
|
+/// Igneum prover-floor patch: the dense elements a set of traces will occupy once `generate_jagged_traces`
|
|
+/// has laid them out, that is the sum of their buffers padded to the next multiple of 2^log_stacking_height
|
|
+/// (the "final padding" step below). Each phase (preprocessed, then main) is padded on its own.
|
|
+pub fn padded_trace_elements(
|
|
+ traces: &BTreeMap<String, Trace<TaskScope>>,
|
|
+ log_stacking_height: u32,
|
|
+) -> usize {
|
|
+ let total: usize = traces
|
|
+ .values()
|
|
+ .map(|t| match t {
|
|
+ Trace::Real(trace) => trace.guts().as_buffer().len(),
|
|
+ Trace::Padding(_) => 0,
|
|
+ })
|
|
+ .sum();
|
|
+ total.next_multiple_of(1 << log_stacking_height)
|
|
+}
|
|
+
|
|
+/// Igneum prover-floor patch: the capacity to allocate for a trace set: the exact padded size plus one
|
|
+/// stacking height of slack, never more than the prover's `max_trace_size`. `SP1_GPU_FLOOR_EXACT=0` restores
|
|
+/// upstream's full-capacity allocation.
|
|
+fn floor_capacity(max_trace_size: usize, needed: usize, log_stacking_height: u32) -> usize {
|
|
+ if std::env::var("SP1_GPU_FLOOR_EXACT").map(|v| v == "0").unwrap_or(false) {
|
|
+ return max_trace_size;
|
|
+ }
|
|
+ max_trace_size.min(needed + (1 << log_stacking_height))
|
|
+}
|
|
+
|
|
async fn allocate_and_initialize_traces(
|
|
preprocessed_traces: BTreeMap<String, Trace<TaskScope>>,
|
|
max_trace_size: usize,
|
|
@@ -494,6 +521,11 @@ async fn allocate_and_initialize_traces(
|
|
|
|
let total_gb = total_bytes as f64 / (1 << 30) as f64;
|
|
tracing::debug!("Allocating {:?} GB of traces", total_gb);
|
|
+ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() {
|
|
+ eprintln!(
|
|
+ "FLOOR tracegen alloc capacity_elements={max_trace_size} bytes={total_bytes} ({total_gb:.3} GB)"
|
|
+ );
|
|
+ }
|
|
let mut dense_data: Buffer<Felt, TaskScope> =
|
|
Buffer::with_capacity_in(max_trace_size, backend.clone());
|
|
let mut col_index: Buffer<u32, TaskScope> =
|
|
@@ -677,9 +709,14 @@ pub async fn setup_tracegen<A: CudaTracegenAir<Felt>>(
|
|
let preprocessed_traces =
|
|
device_preprocessed_tracegen(program, host_phase_tracegen, backend).await;
|
|
|
|
+ let capacity = floor_capacity(
|
|
+ max_trace_size,
|
|
+ padded_trace_elements(&preprocessed_traces, log_stacking_height),
|
|
+ log_stacking_height,
|
|
+ );
|
|
let jagged_traces = allocate_and_initialize_traces(
|
|
preprocessed_traces,
|
|
- max_trace_size,
|
|
+ capacity,
|
|
log_stacking_height,
|
|
max_log_row_count,
|
|
backend,
|
|
@@ -906,6 +943,11 @@ pub async fn main_tracegen<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
|
|
|
|
log_chip_stats(machine, &chip_set, &traces);
|
|
|
|
+ // Igneum prover-floor patch: the key's buffer is sized to its preprocessed traces at setup (upstream sized it
|
|
+ // for a whole shard), so grow it here to what this shard needs before the main traces are appended: a bigger
|
|
+ // dense buffer and column index, the preprocessed region copied device to device, swapped into the key.
|
|
+ grow_for_main(&mut jagged_traces.preprocessed_traces, &traces, log_stacking_height, backend);
|
|
+
|
|
copy_main_jagged_traces(
|
|
traces,
|
|
&mut jagged_traces.preprocessed_traces,
|
|
@@ -918,6 +960,61 @@ pub async fn main_tracegen<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
|
|
(public_values, chip_set, permit)
|
|
}
|
|
|
|
+/// Igneum prover-floor patch: see `main_tracegen`. The need is the preprocessed phase as laid out (its padded
|
|
+/// end, `preprocessed_offset`) plus the main traces padded to the stacking height plus one stacking height of
|
|
+/// slack; a buffer at least that big is left alone. The process aborts, loudly, if the copy cannot be made,
|
|
+/// because a panic inside a prover task is what left sweep 2 hanging on the client's socket.
|
|
+fn grow_for_main(
|
|
+ jagged: &mut JaggedTraceMle<Felt, TaskScope>,
|
|
+ main_traces: &BTreeMap<String, Trace<TaskScope>>,
|
|
+ log_stacking_height: u32,
|
|
+ backend: &TaskScope,
|
|
+) {
|
|
+ let pre_end = jagged.dense().preprocessed_offset;
|
|
+ let needed = pre_end
|
|
+ + padded_trace_elements(main_traces, log_stacking_height)
|
|
+ + (1 << log_stacking_height);
|
|
+ let have = jagged.dense().dense.capacity();
|
|
+ if have >= needed {
|
|
+ return;
|
|
+ }
|
|
+ let mut new_dense: Buffer<Felt, TaskScope> = Buffer::with_capacity_in(needed, backend.clone());
|
|
+ let mut new_col_index: Buffer<u32, TaskScope> =
|
|
+ Buffer::with_capacity_in(needed >> 1, backend.clone());
|
|
+ unsafe {
|
|
+ new_dense.assume_init();
|
|
+ new_col_index.assume_init();
|
|
+ }
|
|
+ {
|
|
+ let JaggedMle { dense_data, col_index, .. } = &mut **jagged;
|
|
+ let src_dense: &Slice<_, _> = &dense_data.dense[..pre_end];
|
|
+ let dst_dense: &mut Slice<_, _> = &mut new_dense[..pre_end];
|
|
+ let src_col: &Slice<_, _> = &col_index[..pre_end >> 1];
|
|
+ let dst_col: &mut Slice<_, _> = &mut new_col_index[..pre_end >> 1];
|
|
+ unsafe {
|
|
+ if dst_dense.copy_from_slice(src_dense, backend).is_err()
|
|
+ || dst_col.copy_from_slice(src_col, backend).is_err()
|
|
+ {
|
|
+ eprintln!("FLOOR grow FAILED: could not copy the preprocessed region ({pre_end} elements) into the grown buffer ({needed} elements); aborting instead of hanging");
|
|
+ std::process::abort();
|
|
+ }
|
|
+ }
|
|
+ }
|
|
+ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() {
|
|
+ eprintln!(
|
|
+ "FLOOR grow key buffer {have} -> {needed} elements (preprocessed {pre_end}, {} bytes)",
|
|
+ needed * 6
|
|
+ );
|
|
+ }
|
|
+ let JaggedMle { dense_data, col_index, .. } = &mut **jagged;
|
|
+ dense_data.dense = new_dense;
|
|
+ *col_index = new_col_index;
|
|
+ unsafe {
|
|
+ dense_data.dense.set_len(pre_end);
|
|
+ col_index.set_len(pre_end >> 1);
|
|
+ }
|
|
+}
|
|
+
|
|
#[allow(clippy::too_many_arguments)]
|
|
pub async fn main_tracegen_permit<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
|
|
machine: &Machine<Felt, A>,
|
|
@@ -984,9 +1081,15 @@ pub async fn full_tracegen<A: CudaTracegenAir<Felt>>(
|
|
|
|
log_chip_stats(machine, &chip_set, &main_traces);
|
|
|
|
+ let capacity = floor_capacity(
|
|
+ max_trace_size,
|
|
+ padded_trace_elements(&preprocessed_traces, log_stacking_height)
|
|
+ + padded_trace_elements(&main_traces, log_stacking_height),
|
|
+ log_stacking_height,
|
|
+ );
|
|
let mut jagged_mle = allocate_and_initialize_traces(
|
|
preprocessed_traces,
|
|
- max_trace_size,
|
|
+ capacity,
|
|
log_stacking_height,
|
|
max_log_row_count,
|
|
backend,
|
|
@@ -1002,6 +1105,18 @@ pub async fn full_tracegen<A: CudaTracegenAir<Felt>>(
|
|
)
|
|
.await;
|
|
|
|
+ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() {
|
|
+ let dense = jagged_mle.dense();
|
|
+ let (free, total) = sp1_gpu_cudart::cuda_memory_info().unwrap_or((0, 0));
|
|
+ eprintln!(
|
|
+ "FLOOR tracegen used preprocessed_elements={} main_elements={} dense_len={} capacity_elements={capacity} max_trace_size={max_trace_size} device_used_mib={}",
|
|
+ dense.preprocessed_offset,
|
|
+ dense.main_size(),
|
|
+ dense.dense.len(),
|
|
+ (total - free) >> 20
|
|
+ );
|
|
+ }
|
|
+
|
|
(public_values, jagged_mle, chip_set, permit)
|
|
}
|
|
|
|
diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/prover_components/src/builder.rs
|
|
index 5dccd9d..574d4fa 100644
|
|
--- a/sp1-gpu/crates/prover_components/src/builder.rs
|
|
+++ b/sp1-gpu/crates/prover_components/src/builder.rs
|
|
@@ -23,28 +23,75 @@ use crate::{
|
|
SP1CudaProverComponents,
|
|
};
|
|
|
|
+/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger
|
|
+/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a
|
|
+/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly.
|
|
+/// The proof format, the verifier and the program ids do not change: the element threshold only decides where
|
|
+/// the executor splits shards, as upstream's own 24 GB tier already does.
|
|
+fn env_usize(name: &str) -> Option<usize> {
|
|
+ std::env::var(name).ok().and_then(|s| s.parse::<usize>().ok())
|
|
+}
|
|
+
|
|
+fn env_f64(name: &str) -> Option<f64> {
|
|
+ std::env::var(name).ok().and_then(|s| s.parse::<f64>().ok())
|
|
+}
|
|
+
|
|
+/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it
|
|
+/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16).
|
|
+pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 {
|
|
+ if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) {
|
|
+ ELEMENT_THRESHOLD
|
|
+ } else if gpu_memory_gb >= 24 {
|
|
+ ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)
|
|
+ } else if gpu_memory_gb >= 18 {
|
|
+ (1 << 27) + (1 << 26)
|
|
+ } else {
|
|
+ 1 << 27
|
|
+ }
|
|
+}
|
|
+
|
|
+/// The recursion trace allocation (elements) for a memory budget.
|
|
+pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize {
|
|
+ if gpu_memory_gb >= 24 {
|
|
+ RECURSION_TRACE_ALLOCATION
|
|
+ } else {
|
|
+ RECURSION_TRACE_ALLOCATION
|
|
+ }
|
|
+}
|
|
+
|
|
+pub fn gpu_memory_gb() -> usize {
|
|
+ let gb = 1024.0 * 1024.0 * 1024.0;
|
|
+ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") {
|
|
+ Some(b) => (b.ceil() as usize) + 4,
|
|
+ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4,
|
|
+ }
|
|
+}
|
|
+
|
|
+pub fn recursion_trace_allocation() -> usize {
|
|
+ env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION")
|
|
+ .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb()))
|
|
+}
|
|
+
|
|
pub fn local_gpu_opts() -> SP1CoreOpts {
|
|
let mut opts = SP1CoreOpts::default();
|
|
|
|
let log2_shard_size = 24;
|
|
opts.shard_size = 1 << log2_shard_size;
|
|
|
|
- let gb = 1024.0 * 1024.0 * 1024.0;
|
|
-
|
|
- // Get the amount of memory on the GPU.
|
|
- let gpu_memory_gb: usize = (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4;
|
|
-
|
|
- if gpu_memory_gb < 24 {
|
|
- panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB");
|
|
- }
|
|
+ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given.
|
|
+ let gpu_memory_gb = gpu_memory_gb();
|
|
|
|
- let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 {
|
|
- ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)
|
|
- } else {
|
|
- ELEMENT_THRESHOLD
|
|
+ let shard_threshold = match env_usize("SP1_GPU_ELEMENT_THRESHOLD") {
|
|
+ Some(t) => t as u64,
|
|
+ None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards),
|
|
};
|
|
+ let height_threshold = opts.sharding_threshold.height_threshold;
|
|
|
|
- tracing::debug!("Shard threshold: {shard_threshold}");
|
|
+ eprintln!(
|
|
+ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}",
|
|
+ recursion_trace_allocation(),
|
|
+ opts.full_size_shards
|
|
+ );
|
|
opts.sharding_threshold.element_threshold = shard_threshold;
|
|
|
|
opts.global_dependencies_opt = true;
|
|
@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier(
|
|
) {
|
|
let recursion_verifier = SP1CudaProverComponents::compress_verifier();
|
|
(
|
|
- new_cuda_prover(&recursion_verifier, RECURSION_TRACE_ALLOCATION, 4, false, false, scope)
|
|
+ new_cuda_prover(&recursion_verifier, recursion_trace_allocation(), 4, false, false, scope)
|
|
.await,
|
|
recursion_verifier,
|
|
)
|
|
diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs
|
|
index 4035f1f..0d0d907 100644
|
|
--- a/sp1-gpu/crates/server/src/server.rs
|
|
+++ b/sp1-gpu/crates/server/src/server.rs
|
|
@@ -157,6 +157,7 @@ impl Server {
|
|
};
|
|
let pk = CachedProgram { elf: Arc::new(Elf::Dynamic(elf.into())), vk: vk.clone() };
|
|
ctx.pk_cache.insert(elf_hash, pk);
|
|
+ floor_memory_line("after setup");
|
|
Response::Setup { id: elf_hash, vk }
|
|
}
|
|
Request::Destroy { key } => {
|
|
@@ -177,15 +178,31 @@ impl Server {
|
|
);
|
|
};
|
|
let context = SP1Context::builder().proof_nonce(proof_nonce).build();
|
|
- match prover.prove_with_mode(&cached.elf, stdin, context, mode).await {
|
|
+ let started = std::time::Instant::now();
|
|
+ let response = match prover.prove_with_mode(&cached.elf, stdin, context, mode).await {
|
|
Ok(proof) => Response::Proof { proof },
|
|
Err(e) => Response::ProverError(e.to_string()),
|
|
- }
|
|
+ };
|
|
+ floor_memory_line(&format!("after prove {:?} in {:.1} s", mode, started.elapsed().as_secs_f64()));
|
|
+ response
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
+/// Igneum prover-floor patch: the device memory in use (total minus free, as the driver reports it) at the
|
|
+/// points that bound a proof, so a run's log carries the terms of the peak without a sampler.
|
|
+fn floor_memory_line(what: &str) {
|
|
+ if let Ok((free, total)) = sp1_gpu_cudart::cuda_memory_info() {
|
|
+ eprintln!(
|
|
+ "FLOOR memory {what}: device_used_mib={} free_mib={} total_mib={}",
|
|
+ (total - free) >> 20,
|
|
+ free >> 20,
|
|
+ total >> 20
|
|
+ );
|
|
+ }
|
|
+}
|
|
+
|
|
fn sha256(data: &[u8]) -> [u8; 32] {
|
|
use sha2::{Digest, Sha256};
|
|
let mut hasher = Sha256::new();
|