igneum/tools/fleet/floor.patch

345 lines
15 KiB
Diff

diff --git a/sp1-gpu/crates/cuda/src/task.rs b/sp1-gpu/crates/cuda/src/task.rs
index a503a86..813016b 100644
--- a/sp1-gpu/crates/cuda/src/task.rs
+++ b/sp1-gpu/crates/cuda/src/task.rs
@@ -149,7 +149,15 @@ pub enum GlobalTaskPoolBuildError {
impl TaskPoolBuilder {
pub fn new() -> Self {
- Self { capacity: None, device: CudaDevice(0), mem_release_threshold: u64::MAX }
+ // Igneum prover-floor patch: upstream keeps every freed device allocation in the pool for the process's
+ // life (threshold u64::MAX), so the prover holds its high-water mark between shards on a card it shares
+ // with a miner. `SP1_GPU_MEM_RELEASE_THRESHOLD=<bytes>` sets the pool's release threshold (0 returns
+ // freed memory to the driver at once); unset, upstream's behaviour.
+ let mem_release_threshold = std::env::var("SP1_GPU_MEM_RELEASE_THRESHOLD")
+ .ok()
+ .and_then(|s| s.parse::<u64>().ok())
+ .unwrap_or(u64::MAX);
+ Self { capacity: None, device: CudaDevice(0), mem_release_threshold }
}
pub fn num_tasks(mut self, num_tasks: usize) -> Self {
diff --git a/sp1-gpu/crates/jagged_tracegen/src/lib.rs b/sp1-gpu/crates/jagged_tracegen/src/lib.rs
index 579f70a..2264044 100644
--- a/sp1-gpu/crates/jagged_tracegen/src/lib.rs
+++ b/sp1-gpu/crates/jagged_tracegen/src/lib.rs
@@ -481,6 +481,33 @@ async fn device_preprocessed_tracegen<A: CudaTracegenAir<Felt>>(
named_traces
}
+/// Igneum prover-floor patch: the dense elements a set of traces will occupy once `generate_jagged_traces`
+/// has laid them out, that is the sum of their buffers padded to the next multiple of 2^log_stacking_height
+/// (the "final padding" step below). Each phase (preprocessed, then main) is padded on its own.
+pub fn padded_trace_elements(
+ traces: &BTreeMap<String, Trace<TaskScope>>,
+ log_stacking_height: u32,
+) -> usize {
+ let total: usize = traces
+ .values()
+ .map(|t| match t {
+ Trace::Real(trace) => trace.guts().as_buffer().len(),
+ Trace::Padding(_) => 0,
+ })
+ .sum();
+ total.next_multiple_of(1 << log_stacking_height)
+}
+
+/// Igneum prover-floor patch: the capacity to allocate for a trace set: the exact padded size plus one
+/// stacking height of slack, never more than the prover's `max_trace_size`. `SP1_GPU_FLOOR_EXACT=0` restores
+/// upstream's full-capacity allocation.
+fn floor_capacity(max_trace_size: usize, needed: usize, log_stacking_height: u32) -> usize {
+ if std::env::var("SP1_GPU_FLOOR_EXACT").map(|v| v == "0").unwrap_or(false) {
+ return max_trace_size;
+ }
+ max_trace_size.min(needed + (1 << log_stacking_height))
+}
+
async fn allocate_and_initialize_traces(
preprocessed_traces: BTreeMap<String, Trace<TaskScope>>,
max_trace_size: usize,
@@ -494,6 +521,11 @@ async fn allocate_and_initialize_traces(
let total_gb = total_bytes as f64 / (1 << 30) as f64;
tracing::debug!("Allocating {:?} GB of traces", total_gb);
+ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() {
+ eprintln!(
+ "FLOOR tracegen alloc capacity_elements={max_trace_size} bytes={total_bytes} ({total_gb:.3} GB)"
+ );
+ }
let mut dense_data: Buffer<Felt, TaskScope> =
Buffer::with_capacity_in(max_trace_size, backend.clone());
let mut col_index: Buffer<u32, TaskScope> =
@@ -677,9 +709,14 @@ pub async fn setup_tracegen<A: CudaTracegenAir<Felt>>(
let preprocessed_traces =
device_preprocessed_tracegen(program, host_phase_tracegen, backend).await;
+ let capacity = floor_capacity(
+ max_trace_size,
+ padded_trace_elements(&preprocessed_traces, log_stacking_height),
+ log_stacking_height,
+ );
let jagged_traces = allocate_and_initialize_traces(
preprocessed_traces,
- max_trace_size,
+ capacity,
log_stacking_height,
max_log_row_count,
backend,
@@ -906,6 +943,11 @@ pub async fn main_tracegen<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
log_chip_stats(machine, &chip_set, &traces);
+ // Igneum prover-floor patch: the key's buffer is sized to its preprocessed traces at setup (upstream sized it
+ // for a whole shard), so grow it here to what this shard needs before the main traces are appended: a bigger
+ // dense buffer and column index, the preprocessed region copied device to device, swapped into the key.
+ grow_for_main(&mut jagged_traces.preprocessed_traces, &traces, log_stacking_height, backend);
+
copy_main_jagged_traces(
traces,
&mut jagged_traces.preprocessed_traces,
@@ -918,6 +960,61 @@ pub async fn main_tracegen<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
(public_values, chip_set, permit)
}
+/// Igneum prover-floor patch: see `main_tracegen`. The need is the preprocessed phase as laid out (its padded
+/// end, `preprocessed_offset`) plus the main traces padded to the stacking height plus one stacking height of
+/// slack; a buffer at least that big is left alone. The process aborts, loudly, if the copy cannot be made,
+/// because a panic inside a prover task is what left sweep 2 hanging on the client's socket.
+fn grow_for_main(
+ jagged: &mut JaggedTraceMle<Felt, TaskScope>,
+ main_traces: &BTreeMap<String, Trace<TaskScope>>,
+ log_stacking_height: u32,
+ backend: &TaskScope,
+) {
+ let pre_end = jagged.dense().preprocessed_offset;
+ let needed = pre_end
+ + padded_trace_elements(main_traces, log_stacking_height)
+ + (1 << log_stacking_height);
+ let have = jagged.dense().dense.capacity();
+ if have >= needed {
+ return;
+ }
+ let mut new_dense: Buffer<Felt, TaskScope> = Buffer::with_capacity_in(needed, backend.clone());
+ let mut new_col_index: Buffer<u32, TaskScope> =
+ Buffer::with_capacity_in(needed >> 1, backend.clone());
+ unsafe {
+ new_dense.assume_init();
+ new_col_index.assume_init();
+ }
+ {
+ let JaggedMle { dense_data, col_index, .. } = &mut **jagged;
+ let src_dense: &Slice<_, _> = &dense_data.dense[..pre_end];
+ let dst_dense: &mut Slice<_, _> = &mut new_dense[..pre_end];
+ let src_col: &Slice<_, _> = &col_index[..pre_end >> 1];
+ let dst_col: &mut Slice<_, _> = &mut new_col_index[..pre_end >> 1];
+ unsafe {
+ if dst_dense.copy_from_slice(src_dense, backend).is_err()
+ || dst_col.copy_from_slice(src_col, backend).is_err()
+ {
+ eprintln!("FLOOR grow FAILED: could not copy the preprocessed region ({pre_end} elements) into the grown buffer ({needed} elements); aborting instead of hanging");
+ std::process::abort();
+ }
+ }
+ }
+ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() {
+ eprintln!(
+ "FLOOR grow key buffer {have} -> {needed} elements (preprocessed {pre_end}, {} bytes)",
+ needed * 6
+ );
+ }
+ let JaggedMle { dense_data, col_index, .. } = &mut **jagged;
+ dense_data.dense = new_dense;
+ *col_index = new_col_index;
+ unsafe {
+ dense_data.dense.set_len(pre_end);
+ col_index.set_len(pre_end >> 1);
+ }
+}
+
#[allow(clippy::too_many_arguments)]
pub async fn main_tracegen_permit<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
machine: &Machine<Felt, A>,
@@ -984,9 +1081,15 @@ pub async fn full_tracegen<A: CudaTracegenAir<Felt>>(
log_chip_stats(machine, &chip_set, &main_traces);
+ let capacity = floor_capacity(
+ max_trace_size,
+ padded_trace_elements(&preprocessed_traces, log_stacking_height)
+ + padded_trace_elements(&main_traces, log_stacking_height),
+ log_stacking_height,
+ );
let mut jagged_mle = allocate_and_initialize_traces(
preprocessed_traces,
- max_trace_size,
+ capacity,
log_stacking_height,
max_log_row_count,
backend,
@@ -1002,6 +1105,18 @@ pub async fn full_tracegen<A: CudaTracegenAir<Felt>>(
)
.await;
+ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() {
+ let dense = jagged_mle.dense();
+ let (free, total) = sp1_gpu_cudart::cuda_memory_info().unwrap_or((0, 0));
+ eprintln!(
+ "FLOOR tracegen used preprocessed_elements={} main_elements={} dense_len={} capacity_elements={capacity} max_trace_size={max_trace_size} device_used_mib={}",
+ dense.preprocessed_offset,
+ dense.main_size(),
+ dense.dense.len(),
+ (total - free) >> 20
+ );
+ }
+
(public_values, jagged_mle, chip_set, permit)
}
diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/prover_components/src/builder.rs
index 5dccd9d..574d4fa 100644
--- a/sp1-gpu/crates/prover_components/src/builder.rs
+++ b/sp1-gpu/crates/prover_components/src/builder.rs
@@ -23,28 +23,75 @@ use crate::{
SP1CudaProverComponents,
};
+/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger
+/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a
+/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly.
+/// The proof format, the verifier and the program ids do not change: the element threshold only decides where
+/// the executor splits shards, as upstream's own 24 GB tier already does.
+fn env_usize(name: &str) -> Option<usize> {
+ std::env::var(name).ok().and_then(|s| s.parse::<usize>().ok())
+}
+
+fn env_f64(name: &str) -> Option<f64> {
+ std::env::var(name).ok().and_then(|s| s.parse::<f64>().ok())
+}
+
+/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it
+/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16).
+pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 {
+ if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) {
+ ELEMENT_THRESHOLD
+ } else if gpu_memory_gb >= 24 {
+ ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)
+ } else if gpu_memory_gb >= 18 {
+ (1 << 27) + (1 << 26)
+ } else {
+ 1 << 27
+ }
+}
+
+/// The recursion trace allocation (elements) for a memory budget.
+pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize {
+ if gpu_memory_gb >= 24 {
+ RECURSION_TRACE_ALLOCATION
+ } else {
+ RECURSION_TRACE_ALLOCATION
+ }
+}
+
+pub fn gpu_memory_gb() -> usize {
+ let gb = 1024.0 * 1024.0 * 1024.0;
+ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") {
+ Some(b) => (b.ceil() as usize) + 4,
+ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4,
+ }
+}
+
+pub fn recursion_trace_allocation() -> usize {
+ env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION")
+ .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb()))
+}
+
pub fn local_gpu_opts() -> SP1CoreOpts {
let mut opts = SP1CoreOpts::default();
let log2_shard_size = 24;
opts.shard_size = 1 << log2_shard_size;
- let gb = 1024.0 * 1024.0 * 1024.0;
-
- // Get the amount of memory on the GPU.
- let gpu_memory_gb: usize = (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4;
-
- if gpu_memory_gb < 24 {
- panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB");
- }
+ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given.
+ let gpu_memory_gb = gpu_memory_gb();
- let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 {
- ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)
- } else {
- ELEMENT_THRESHOLD
+ let shard_threshold = match env_usize("SP1_GPU_ELEMENT_THRESHOLD") {
+ Some(t) => t as u64,
+ None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards),
};
+ let height_threshold = opts.sharding_threshold.height_threshold;
- tracing::debug!("Shard threshold: {shard_threshold}");
+ eprintln!(
+ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}",
+ recursion_trace_allocation(),
+ opts.full_size_shards
+ );
opts.sharding_threshold.element_threshold = shard_threshold;
opts.global_dependencies_opt = true;
@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier(
) {
let recursion_verifier = SP1CudaProverComponents::compress_verifier();
(
- new_cuda_prover(&recursion_verifier, RECURSION_TRACE_ALLOCATION, 4, false, false, scope)
+ new_cuda_prover(&recursion_verifier, recursion_trace_allocation(), 4, false, false, scope)
.await,
recursion_verifier,
)
diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs
index 4035f1f..0d0d907 100644
--- a/sp1-gpu/crates/server/src/server.rs
+++ b/sp1-gpu/crates/server/src/server.rs
@@ -157,6 +157,7 @@ impl Server {
};
let pk = CachedProgram { elf: Arc::new(Elf::Dynamic(elf.into())), vk: vk.clone() };
ctx.pk_cache.insert(elf_hash, pk);
+ floor_memory_line("after setup");
Response::Setup { id: elf_hash, vk }
}
Request::Destroy { key } => {
@@ -177,15 +178,31 @@ impl Server {
);
};
let context = SP1Context::builder().proof_nonce(proof_nonce).build();
- match prover.prove_with_mode(&cached.elf, stdin, context, mode).await {
+ let started = std::time::Instant::now();
+ let response = match prover.prove_with_mode(&cached.elf, stdin, context, mode).await {
Ok(proof) => Response::Proof { proof },
Err(e) => Response::ProverError(e.to_string()),
- }
+ };
+ floor_memory_line(&format!("after prove {:?} in {:.1} s", mode, started.elapsed().as_secs_f64()));
+ response
}
}
}
}
+/// Igneum prover-floor patch: the device memory in use (total minus free, as the driver reports it) at the
+/// points that bound a proof, so a run's log carries the terms of the peak without a sampler.
+fn floor_memory_line(what: &str) {
+ if let Ok((free, total)) = sp1_gpu_cudart::cuda_memory_info() {
+ eprintln!(
+ "FLOOR memory {what}: device_used_mib={} free_mib={} total_mib={}",
+ (total - free) >> 20,
+ free >> 20,
+ total >> 20
+ );
+ }
+}
+
fn sha256(data: &[u8]) -> [u8; 32] {
use sha2::{Digest, Sha256};
let mut hasher = Sha256::new();