From 0a108a3449fed053494683edaf0b9d8340a14afc Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 19:47:49 +0000 Subject: [PATCH 01/15] V6-07: the floor's limits from FREE memory, the small-card threshold 2^26, two recursion budgets, a pinned memory profile per workload in the host Master review R1 residual V6-07 (the floor patch picked limits from total VRAM, defaulted small cards to 2^27 where the passing rows used 2^26, and returned one recursion constant in both branches). - proving/prover-floor/sp1-gpu-6.8.1-floor.patch (and the fleet's two copies, box-setup.sh re-pinned): gpu_memory_gb() reads the device's FREE memory once per process (OnceLock; SP1_GPU_MEMORY_BUDGET_GB when the host leases it), never the total; the small tier's element threshold is 2^26 (the alone-comp-26-v1 rows of 6 October on the 3060, 3080, 4060, 4060 Ti, 4070, 5070 and the 8 October proof_alone rows on the 3060 at 7,525 MiB and the 4060 at 7,532 MiB); recursion_trace_allocation_for_budget returns upstream's 2^27 on the 24 GB tier and RECURSION_TRACE_ALLOCATION_SMALL = 2^26 + 2^25 under it (a recursion key or shard uses 90,177,536 elements); floor_tests pin the small tier and that the two branches differ; the FLOOR opts line carries free_mib and total_mib. - host/src/memory_profile.rs: the pinned table (full, 24gb, 16gb, small by free MiB; floors per workload shard, aggregate, chain) with tests pinning every value; the row is chosen from the engine's lease (IGNEUM_PROVE_MEM_BUDGET_MB, else IGNEUM_PROVE_MEM_FREE_MB, with IGNEUM_PROVE_DEVICE, IGNEUM_PROVE_WORKLOAD, IGNEUM_PROVE_DEADLINE_S: the app lane's device coordinator interface) or, with no engine, from nvidia-smi memory.free on the device; applied to the floor server by environment before the SP1 client spawns it; a hand override is kept and named; under the floor the host refuses with one line and exit 78 before any setup. - tools/fleet/box-prover.py: the default path names its workload (IGNEUM_PROVE_WORKLOAD=chain, IGNEUM_PROVE_DEVICE) and closes a refused segment as cancelled/memory. Co-Authored-By: Claude Fable 5.1 --- proving/igneum-prove/host/src/main.rs | 10 + .../igneum-prove/host/src/memory_profile.rs | 368 ++++++++++++++++++ .../prover-floor/sp1-gpu-6.8.1-floor.patch | 87 ++++- tools/fleet/box-prover.py | 12 +- tools/fleet/box-setup.sh | 2 +- tools/fleet/floor-v5.patch | 87 ++++- tools/fleet/floor.patch | 87 ++++- 7 files changed, 593 insertions(+), 60 deletions(-) create mode 100644 proving/igneum-prove/host/src/memory_profile.rs diff --git a/proving/igneum-prove/host/src/main.rs b/proving/igneum-prove/host/src/main.rs index 98e45f3fe..ae2cec3ce 100644 --- a/proving/igneum-prove/host/src/main.rs +++ b/proving/igneum-prove/host/src/main.rs @@ -20,6 +20,7 @@ //! ids with no setup. `--mode verify` uses SP1's light verifier and the pinned verifying key: no prover client, //! no key generation (the 114 s to 138 s the Mac's node spent per proof on 5 October). +mod memory_profile; mod pinned; mod proof_system; @@ -69,6 +70,15 @@ fn run() -> Result<()> { let args: Vec = std::env::args().collect(); let arg = |name: &str| args.iter().position(|a| a == name).and_then(|i| args.get(i + 1)).cloned(); let mode = arg("--mode").unwrap_or_else(|| "all".into()); + // V6-07 (8 October 2026): on the GPU prover the memory profile for this process's workload is chosen here, once, + // from the card's FREE memory (the engine's lease, else nvidia-smi) and handed to the floor server by environment + // before the SP1 client spawns it; a budget under the workload's floor is refused with exit 78 before any setup. + if std::env::var("SP1_PROVER").map(|p| p == "cuda").unwrap_or(false) { + if let Err(refusal) = memory_profile::apply(&mode) { + println!("RESULT memory_profile refused: {refusal}"); + std::process::exit(memory_profile::EXIT_REFUSED); + } + } let pinned = pinned::Pinned::load()?; if mode == "id" { println!("RESULT id: {}", pinned.describe()); diff --git a/proving/igneum-prove/host/src/memory_profile.rs b/proving/igneum-prove/host/src/memory_profile.rs new file mode 100644 index 000000000..f6f1c82eb --- /dev/null +++ b/proving/igneum-prove/host/src/memory_profile.rs @@ -0,0 +1,368 @@ +//! The pinned memory profile per workload (master review R1 residual V6-07, 8 October 2026): the GPU server's two +//! device buffers (the core element threshold and the recursion trace allocation) are chosen HERE, once, from the +//! memory that is FREE on the card when the host starts, never from the card's total. The table below is the +//! profile; `choose` picks a row for the workload this process is admitted for; `apply` hands the row to the server +//! through the floor patch's environment (`SP1_GPU_ELEMENT_THRESHOLD`, `SP1_GPU_RECURSION_TRACE_ALLOCATION`, +//! `SP1_GPU_MEMORY_BUDGET_GB`) before the SP1 client spawns it. An explicit `SP1_GPU_ELEMENT_THRESHOLD` or +//! `SP1_GPU_RECURSION_TRACE_ALLOCATION` already in the environment is a hand override and wins, named in the line. +//! +//! Where the free memory comes from, in order (the app lane's device coordinator interface, the shipper's answer of +//! 20:4x UK 8 October 2026; the engine owns admission and the host never re-reads a card the engine leased): +//! 1. `IGNEUM_PROVE_MEM_BUDGET_MB`: the lease's grant, the hard ceiling for this process (the engine's coordinator, +//! app/igneum-app `src/device.rs`, passes it with `IGNEUM_PROVE_DEVICE`, `IGNEUM_PROVE_WORKLOAD`, +//! `IGNEUM_PROVE_MEM_FREE_MB` and `IGNEUM_PROVE_DEADLINE_S` on every spawn); +//! 2. `IGNEUM_PROVE_MEM_FREE_MB`: the engine's own read at admission, when no grant is given; +//! 3. `nvidia-smi --query-gpu=memory.free` on the device (`IGNEUM_PROVE_DEVICE`, else `IGNEUM_CUDA_DEVICE`, else the +//! first of `CUDA_VISIBLE_DEVICES`, else 0): the fleet's `tools/fleet/box-prover.py` path and a hand run, where no +//! engine admits the job. +//! When none answers (no nvidia-smi on the path), no row is applied and the server's own free-memory rule decides. +//! +//! A budget under the workload's floor is refused before the server starts: one line, exit code 78 (the engine +//! records the refusal on the lease and the window reads "proving needs N GB free"). +//! +//! The rows cited for the small tier's 2^26: `docs/analysis/prover-tiers-real-cards.md` (6 October 2026, +//! `alone-comp-26-v1`: RTX 3060 7.4 GB 14.4 s, RTX 3080 8.0 GB 7.1 s, RTX 4060 7.4 GB 18.4 s, RTX 4060 Ti 8 GB 7.6 GB +//! 9.6 s, RTX 4060 Ti 16 GB 7.8 GB 11.6 s, RTX 4070 7.6 GB 12.1 s, RTX 5070 7.6 GB 4.8 s, all verified) and +//! `docs/analysis/class-v6/coexist-rows.md` (8 October 2026, `proof_alone` at `SP1_GPU_ELEMENT_THRESHOLD=67108864`: +//! RTX 3060 peak 7,525 MiB 13.2 s verified, RTX 4060 peak 7,532 MiB 8.2 s verified). No small-card row ever passed +//! at the patch's earlier default of 2^27 on this host; the 10 GB tier's 2^27 rows (8,642 MiB, 7 October 2026) ran +//! alone on the segment host and are not this path. + +use std::fmt; + +/// Upstream's core element threshold (sp1-core-executor 6.8.1 `opts.rs` 12): 2^28 + 2^27. +pub const ELEMENT_THRESHOLD_FULL: u64 = (1 << 28) + (1 << 27); +/// Upstream's 24 GB tier (sp1-gpu `builder.rs` 44): the full threshold less 2^26 + 2^25 + 2^24. +pub const ELEMENT_THRESHOLD_24GB: u64 = ELEMENT_THRESHOLD_FULL - (1 << 26) - (1 << 25) - (1 << 24); +/// The floor patch's 16 GB tier: 2^27 + 2^26 (unmeasured on this fixture; the patch's own figure). +pub const ELEMENT_THRESHOLD_16GB: u64 = (1 << 27) + (1 << 26); +/// The small-card threshold: 2^26, the value every passing small-card row used (module note). +pub const ELEMENT_THRESHOLD_SMALL: u64 = 1 << 26; +/// Upstream's recursion trace allocation (sp1-gpu `builder.rs` 15): 2^27 elements, 0.75 GiB each. +pub const RECURSION_TRACE_ALLOCATION: usize = 1 << 27; +/// The small-card recursion trace allocation: the recursion keys and shards use 90,177,536 elements each (35.6 M +/// preprocessed, 54.5 M main; `docs/analysis/prover-floor.md`, sweep 1), so 2^26 + 2^25 = 100,663,296 holds them +/// with the stacking slack the patch's `floor_capacity` adds. The two constants differ (`recursion_branches_differ`). +pub const RECURSION_TRACE_ALLOCATION_SMALL: usize = (1 << 26) + (1 << 25); +/// What a recursion key or shard was measured to use (elements); the small allocation must clear it with slack. +pub const RECURSION_TRACE_USED: usize = 90_177_536; + +/// The one job this process is admitted for. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Workload { + /// One shard's compressed (or core) proof. + Shard, + /// The aggregator guest over shard proofs (and the previous segment's proof). + Aggregate, + /// A whole segment in one process: every shard, then the aggregation. + Chain, +} + +impl Workload { + pub fn name(self) -> &'static str { + match self { + Workload::Shard => "shard", + Workload::Aggregate => "aggregate", + Workload::Chain => "chain", + } + } + /// `IGNEUM_PROVE_WORKLOAD` when the engine names it, else the host's own mode. + pub fn from_env_or_mode(mode: &str) -> Option { + if let Ok(w) = std::env::var("IGNEUM_PROVE_WORKLOAD") { + return match w.as_str() { + "shard" => Some(Workload::Shard), + "aggregate" => Some(Workload::Aggregate), + "chain" => Some(Workload::Chain), + _ => None, + }; + } + Workload::from_mode(mode) + } + /// The workload a host mode runs on the GPU (modes that never prove return None). + pub fn from_mode(mode: &str) -> Option { + match mode { + "shard" | "compressed" | "core" => Some(Workload::Shard), + "aggregate" => Some(Workload::Aggregate), + "chain" | "block" | "all" => Some(Workload::Chain), + _ => None, + } + } + /// The least free memory (MiB) a workload runs in. Shard: the small tier's measured peak (7,525 to 7,532 MiB + /// on the 3060 and 4060 at 2^26) plus headroom for the driver's own context. Aggregate and chain: the same + /// floor until the 8 October 2026 rows on the default path land their aggregation peak (the analysis document + /// `docs/analysis/floor-memory-profile-2026-10-08.md` carries the measured figure beside this constant). + pub fn floor_mib(self) -> u64 { + match self { + Workload::Shard => 7_700, + Workload::Aggregate => 7_700, + Workload::Chain => 7_700, + } + } +} + +/// One tier of the profile table: the least free memory it needs and the two buffers it sets. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Tier { + pub name: &'static str, + pub min_free_mib: u64, + pub element_threshold: u64, + pub recursion_trace_allocation: usize, +} + +/// The pinned table, largest first. The free-memory lines are the card tiers as the floor patch reads them (free +/// GiB, ceiling, +4 as upstream computed it from the total): 26 GiB free reads over 30 (a 32 GB card alone), 20 GiB +/// reads 24 (a 24 GB card alone), 14 GiB reads 18 (a 16 GB card alone), under that the small tier (12 GB and 8 GB +/// cards alone; any card beside a miner's resident set). +pub const TIERS: [Tier; 4] = [ + Tier { name: "full", min_free_mib: 26 * 1024, element_threshold: ELEMENT_THRESHOLD_FULL, recursion_trace_allocation: RECURSION_TRACE_ALLOCATION }, + Tier { name: "24gb", min_free_mib: 20 * 1024, element_threshold: ELEMENT_THRESHOLD_24GB, recursion_trace_allocation: RECURSION_TRACE_ALLOCATION }, + Tier { name: "16gb", min_free_mib: 14 * 1024, element_threshold: ELEMENT_THRESHOLD_16GB, recursion_trace_allocation: RECURSION_TRACE_ALLOCATION_SMALL }, + Tier { name: "small", min_free_mib: 0, element_threshold: ELEMENT_THRESHOLD_SMALL, recursion_trace_allocation: RECURSION_TRACE_ALLOCATION_SMALL }, +]; + +/// The row chosen for a process. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Profile { + pub workload: Workload, + pub tier: Tier, + pub free_mib: u64, +} + +/// Why a process is refused before the server starts. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Refusal { + pub workload: Workload, + pub free_mib: u64, + pub floor_mib: u64, +} + +impl fmt::Display for Refusal { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + f, + "proving needs {} GB free on the card for the {} workload ({} MiB floor); {} MiB free", + (self.floor_mib + 1023) / 1024, + self.workload.name(), + self.floor_mib, + self.free_mib + ) + } +} + +/// The exit code of a refusal (the engine reads it on the lease). +pub const EXIT_REFUSED: i32 = 78; + +/// The tier for a free-memory reading (pure; the table's first row whose line the reading clears). +pub fn tier_for(free_mib: u64) -> Tier { + *TIERS.iter().find(|t| free_mib >= t.min_free_mib).expect("the small tier has no floor") +} + +/// The row for a workload at a free-memory reading, or the refusal. +pub fn choose(workload: Workload, free_mib: u64) -> Result { + let floor_mib = workload.floor_mib(); + if free_mib < floor_mib { + return Err(Refusal { workload, free_mib, floor_mib }); + } + Ok(Profile { workload, tier: tier_for(free_mib), free_mib }) +} + +/// Where a free-memory reading came from, for the line. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Source { + /// `IGNEUM_PROVE_MEM_BUDGET_MB`, the engine's lease. + Lease, + /// `IGNEUM_PROVE_MEM_FREE_MB`, the engine's read without a grant. + EngineRead, + /// `nvidia-smi --query-gpu=memory.free` on the named device. + NvidiaSmi(String), +} + +impl fmt::Display for Source { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Source::Lease => write!(f, "lease IGNEUM_PROVE_MEM_BUDGET_MB"), + Source::EngineRead => write!(f, "engine IGNEUM_PROVE_MEM_FREE_MB"), + Source::NvidiaSmi(d) => write!(f, "nvidia-smi device {d}"), + } + } +} + +fn env_u64(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.trim().parse::().ok()) +} + +/// The device ordinal the host runs on, as the module note orders it. +pub fn device_ordinal() -> String { + if let Ok(d) = std::env::var("IGNEUM_PROVE_DEVICE") { + return d; + } + if let Ok(d) = std::env::var("IGNEUM_CUDA_DEVICE") { + return d; + } + if let Ok(v) = std::env::var("CUDA_VISIBLE_DEVICES") { + if let Some(first) = v.split(',').next() { + if !first.trim().is_empty() { + return first.trim().to_string(); + } + } + } + "0".into() +} + +/// The free memory on the card at this call, from the sources in the module note's order. +pub fn read_free_mib() -> Option<(u64, Source)> { + if let Some(b) = env_u64("IGNEUM_PROVE_MEM_BUDGET_MB") { + return Some((b, Source::Lease)); + } + if let Some(b) = env_u64("IGNEUM_PROVE_MEM_FREE_MB") { + return Some((b, Source::EngineRead)); + } + let dev = device_ordinal(); + let out = std::process::Command::new("nvidia-smi") + .args(["--query-gpu=memory.free", "--format=csv,noheader,nounits", "-i", &dev]) + .output() + .ok()?; + if !out.status.success() { + return None; + } + let text = String::from_utf8_lossy(&out.stdout); + let first = text.lines().next()?.trim(); + first.parse::().ok().map(|v| (v, Source::NvidiaSmi(dev))) +} + +/// The environment the row sets for the floor server (what `apply` writes), as (name, value) pairs; an explicit +/// hand override already present keeps its value and is reported. +pub fn env_for(p: &Profile) -> Vec<(&'static str, String)> { + vec![ + ("SP1_GPU_ELEMENT_THRESHOLD", p.tier.element_threshold.to_string()), + ("SP1_GPU_RECURSION_TRACE_ALLOCATION", p.tier.recursion_trace_allocation.to_string()), + ("SP1_GPU_MEMORY_BUDGET_GB", format!("{:.1}", p.free_mib as f64 / 1024.0)), + ] +} + +/// Chooses and applies the row for this process: prints one `RESULT memory_profile` line and returns Ok(Some) with +/// the profile, Ok(None) when no reading was possible (the server's own rule decides) or the workload never proves, +/// and Err with the refusal line (the caller exits `EXIT_REFUSED`). +pub fn apply(mode: &str) -> Result, Refusal> { + let Some(workload) = Workload::from_env_or_mode(mode) else { return Ok(None) }; + let Some((free_mib, source)) = read_free_mib() else { + println!("RESULT memory_profile: no free-memory reading (no lease, no nvidia-smi on device {}); the server's own free-memory rule decides for workload {}", device_ordinal(), workload.name()); + return Ok(None); + }; + let profile = choose(workload, free_mib)?; + let mut set = Vec::new(); + let mut kept = Vec::new(); + for (k, v) in env_for(&profile) { + match std::env::var(k) { + Ok(have) if k != "SP1_GPU_MEMORY_BUDGET_GB" => kept.push(format!("{k}={have} (hand override kept, the row said {v})")), + _ => { + std::env::set_var(k, &v); + set.push(format!("{k}={v}")); + } + } + } + let deadline = std::env::var("IGNEUM_PROVE_DEADLINE_S").ok().map(|d| format!(", deadline {d} s")).unwrap_or_default(); + println!( + "RESULT memory_profile: workload {} tier {} from {} MiB free ({}){}: element_threshold {} recursion_trace_allocation {}; set {}{}", + workload.name(), + profile.tier.name, + free_mib, + source, + deadline, + profile.tier.element_threshold, + profile.tier.recursion_trace_allocation, + set.join(" "), + if kept.is_empty() { String::new() } else { format!("; {}", kept.join("; ")) } + ); + Ok(Some(profile)) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The table's values are pinned: a change here is a change of the profile and needs its rows. + #[test] + fn table_is_pinned() { + assert_eq!(ELEMENT_THRESHOLD_FULL, 402_653_184); + assert_eq!(ELEMENT_THRESHOLD_24GB, 285_212_672); + assert_eq!(ELEMENT_THRESHOLD_16GB, 201_326_592); + assert_eq!(ELEMENT_THRESHOLD_SMALL, 67_108_864); + assert_eq!(RECURSION_TRACE_ALLOCATION, 134_217_728); + assert_eq!(RECURSION_TRACE_ALLOCATION_SMALL, 100_663_296); + let names: Vec<&str> = TIERS.iter().map(|t| t.name).collect(); + assert_eq!(names, ["full", "24gb", "16gb", "small"]); + assert_eq!(TIERS[0].min_free_mib, 26_624); + assert_eq!(TIERS[1].min_free_mib, 20_480); + assert_eq!(TIERS[2].min_free_mib, 14_336); + assert_eq!(TIERS[3].min_free_mib, 0); + assert_eq!(TIERS[3].element_threshold, ELEMENT_THRESHOLD_SMALL); + assert_eq!(TIERS[3].recursion_trace_allocation, RECURSION_TRACE_ALLOCATION_SMALL); + assert_eq!(TIERS[0].recursion_trace_allocation, RECURSION_TRACE_ALLOCATION); + assert_eq!(Workload::Shard.floor_mib(), 7_700); + assert_eq!(Workload::Aggregate.floor_mib(), 7_700); + assert_eq!(Workload::Chain.floor_mib(), 7_700); + assert_eq!(EXIT_REFUSED, 78); + } + + /// The two recursion branches are two values (the review found one constant in both), and the small one + /// clears what a recursion key or shard was measured to use, with a stacking height of slack. + #[test] + fn recursion_branches_differ() { + assert_ne!(RECURSION_TRACE_ALLOCATION, RECURSION_TRACE_ALLOCATION_SMALL); + assert!(RECURSION_TRACE_ALLOCATION_SMALL < RECURSION_TRACE_ALLOCATION); + assert!(RECURSION_TRACE_ALLOCATION_SMALL >= RECURSION_TRACE_USED + (1 << 22)); + assert_ne!(tier_for(24 * 1024).recursion_trace_allocation, tier_for(12 * 1024).recursion_trace_allocation); + } + + /// The small tier's threshold is 2^26, the passing rows' value; the card tiers alone read their own rows. + #[test] + fn tiers_from_free_memory() { + assert_eq!(tier_for(8_186).name, "small"); + assert_eq!(tier_for(8_186).element_threshold, 1 << 26); + assert_eq!(tier_for(12_100).name, "small"); + assert_eq!(tier_for(12_100).element_threshold, 1 << 26); + assert_eq!(tier_for(16_100).name, "16gb"); + assert_eq!(tier_for(24_200).name, "24gb"); + assert_eq!(tier_for(32_300).name, "full"); + // a 12 GB card beside the 5.5 GiB miner (6,129 MiB resident) reads the small tier by what is FREE + assert_eq!(tier_for(12_288 - 6_129).name, "small"); + } + + /// The refusal: under the floor, before the server starts; at the floor, the small row. + #[test] + fn refuses_under_the_floor() { + let r = choose(Workload::Shard, 12_288 - 6_129).unwrap_err(); + assert_eq!(r.floor_mib, 7_700); + assert_eq!(r.free_mib, 6_159); + assert_eq!(r.to_string(), "proving needs 8 GB free on the card for the shard workload (7,700 MiB floor); 6,159 MiB free".replace(",", "")); + let p = choose(Workload::Shard, 7_700).unwrap(); + assert_eq!(p.tier.name, "small"); + let p = choose(Workload::Chain, 8_186).unwrap(); + assert_eq!(p.tier.element_threshold, ELEMENT_THRESHOLD_SMALL); + assert_eq!(p.tier.recursion_trace_allocation, RECURSION_TRACE_ALLOCATION_SMALL); + } + + /// The host's modes map to the three workloads; modes that never prove map to none. + #[test] + fn workload_from_mode() { + assert_eq!(Workload::from_mode("compressed"), Some(Workload::Shard)); + assert_eq!(Workload::from_mode("core"), Some(Workload::Shard)); + assert_eq!(Workload::from_mode("aggregate"), Some(Workload::Aggregate)); + assert_eq!(Workload::from_mode("chain"), Some(Workload::Chain)); + assert_eq!(Workload::from_mode("all"), Some(Workload::Chain)); + assert_eq!(Workload::from_mode("verify"), None); + assert_eq!(Workload::from_mode("id"), None); + assert_eq!(Workload::from_mode("native"), None); + } + + /// The environment a row hands the floor server. + #[test] + fn env_for_the_server() { + let p = choose(Workload::Shard, 8_186).unwrap(); + let env = env_for(&p); + assert_eq!(env[0], ("SP1_GPU_ELEMENT_THRESHOLD", "67108864".to_string())); + assert_eq!(env[1], ("SP1_GPU_RECURSION_TRACE_ALLOCATION", "100663296".to_string())); + assert_eq!(env[2], ("SP1_GPU_MEMORY_BUDGET_GB", "8.0".to_string())); + } +} diff --git a/proving/prover-floor/sp1-gpu-6.8.1-floor.patch b/proving/prover-floor/sp1-gpu-6.8.1-floor.patch index c66939337..dcc6fbad4 100644 --- a/proving/prover-floor/sp1-gpu-6.8.1-floor.patch +++ b/proving/prover-floor/sp1-gpu-6.8.1-floor.patch @@ -104,15 +104,16 @@ diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/pr index 5dccd9d..574d4fa 100644 --- a/sp1-gpu/crates/prover_components/src/builder.rs +++ b/sp1-gpu/crates/prover_components/src/builder.rs -@@ -23,28 +23,75 @@ use crate::{ +@@ -23,28 +23,124 @@ use crate::{ SP1CudaProverComponents, }; -+/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger -+/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a -+/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. -+/// The proof format, the verifier and the program ids do not change: the element threshold only decides where -+/// the executor splits shards, as upstream's own 24 GB tier already does. ++/// Igneum prover-floor patch (5 October 2026; the memory rules of V6-07, 8 October 2026). Upstream sizes every ++/// device buffer for a 24 GB card or larger and panics below that, whatever the shard. Here the card's FREE memory ++/// at start (or `SP1_GPU_MEMORY_BUDGET_GB`, the host's lease) picks a tier, read once for the process, and ++/// `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. The proof ++/// format, the verifier and the program ids do not change: the element threshold only decides where the executor ++/// splits shards, as upstream's own 24 GB tier already does. +fn env_usize(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} @@ -121,8 +122,17 @@ index 5dccd9d..574d4fa 100644 + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} + -+/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it -+/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16). ++/// The recursion trace allocation under the 24 GB tier (V6-07): a recursion key or shard uses 90,177,536 elements ++/// (35.6 M preprocessed and 54.5 M main; the floor sweep of 5 October 2026), so 2^26 + 2^25 = 100,663,296 holds it ++/// with the stacking slack `floor_capacity` adds; the pinned host copies (four per prover) shrink with it. Upstream's ++/// 2^27 stays on the 24 GB tier and above. Two distinct values: `floor_tests::recursion_branches_differ`. ++pub const RECURSION_TRACE_ALLOCATION_SMALL: usize = (1 << 26) + (1 << 25); ++ ++/// The core element threshold for a memory budget in GB (the budget is the FREE memory, +4, as upstream computed ++/// its tiers from the total: a 32 GB card alone reads 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16, an ++/// 8 GB card 12). Under the 18 tier the threshold is 2^26, the value every passing small-card row used (the ++/// alone-comp-26-v1 rows of 6 October 2026 on the 3060, 3080, 4060, 4060 Ti, 4070 and 5070; the 8 October 2026 ++/// proof_alone rows on the 3060 at 7,525 MiB and the 4060 at 7,532 MiB); no small-card row passed at 2^27 here. +pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 { + if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) { + ELEMENT_THRESHOLD @@ -131,31 +141,69 @@ index 5dccd9d..574d4fa 100644 + } else if gpu_memory_gb >= 18 { + (1 << 27) + (1 << 26) + } else { -+ 1 << 27 ++ 1 << 26 + } +} + -+/// The recursion trace allocation (elements) for a memory budget. ++/// The recursion trace allocation (elements) for a memory budget: upstream's on the 24 GB tier and above, the ++/// small constant under it. +pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize { + if gpu_memory_gb >= 24 { + RECURSION_TRACE_ALLOCATION + } else { -+ RECURSION_TRACE_ALLOCATION ++ RECURSION_TRACE_ALLOCATION_SMALL + } +} + ++/// The memory budget in GB, read ONCE for the process (the core opts and the recursion prover are built at ++/// different moments, after allocations that lower the free figure; one reading keeps both on one tier): ++/// `SP1_GPU_MEMORY_BUDGET_GB` when set (the host's lease), else the device's FREE memory as the driver reports it ++/// at the first call, never the total (a card beside a miner has the miner's resident set gone), +4 as upstream ++/// computed its tiers. +pub fn gpu_memory_gb() -> usize { -+ let gb = 1024.0 * 1024.0 * 1024.0; -+ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { -+ Some(b) => (b.ceil() as usize) + 4, -+ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4, -+ } ++ static BUDGET: std::sync::OnceLock = std::sync::OnceLock::new(); ++ *BUDGET.get_or_init(|| { ++ let gb = 1024.0 * 1024.0 * 1024.0; ++ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { ++ Some(b) => (b.ceil() as usize) + 4, ++ None => { ++ let (free, _total) = cuda_memory_info().unwrap(); ++ (((free as f64) / gb).ceil() as usize) + 4 ++ } ++ } ++ }) +} + +pub fn recursion_trace_allocation() -> usize { + env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION") + .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb())) +} ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::*; ++ ++ #[test] ++ fn small_tier_is_two_to_the_26() { ++ assert_eq!(element_threshold_for_budget(12, false), 1 << 26, "an 8 GB card reads 12"); ++ assert_eq!(element_threshold_for_budget(16, false), 1 << 26, "a 12 GB card reads 16"); ++ assert_eq!(element_threshold_for_budget(17, false), 1 << 26); ++ assert_eq!(element_threshold_for_budget(20, false), (1 << 27) + (1 << 26), "a 16 GB card reads 20"); ++ assert_eq!(element_threshold_for_budget(28, false), ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)); ++ assert_eq!(element_threshold_for_budget(28, true), ELEMENT_THRESHOLD); ++ assert_eq!(element_threshold_for_budget(36, false), ELEMENT_THRESHOLD); ++ } ++ ++ #[test] ++ fn recursion_branches_differ() { ++ assert_ne!(recursion_trace_allocation_for_budget(28), recursion_trace_allocation_for_budget(16)); ++ assert_eq!(recursion_trace_allocation_for_budget(28), RECURSION_TRACE_ALLOCATION); ++ assert_eq!(recursion_trace_allocation_for_budget(16), RECURSION_TRACE_ALLOCATION_SMALL); ++ assert_eq!(RECURSION_TRACE_ALLOCATION_SMALL, 100_663_296); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL < RECURSION_TRACE_ALLOCATION); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL >= 90_177_536 + (1 << 22)); ++ } ++} + pub fn local_gpu_opts() -> SP1CoreOpts { let mut opts = SP1CoreOpts::default(); @@ -171,7 +219,7 @@ index 5dccd9d..574d4fa 100644 - if gpu_memory_gb < 24 { - panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"); - } -+ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given. ++ // The card's FREE memory plus 4, as upstream computed its tiers from the total, or the budget given. + let gpu_memory_gb = gpu_memory_gb(); - let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 { @@ -183,17 +231,18 @@ index 5dccd9d..574d4fa 100644 + None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards), }; + let height_threshold = opts.sharding_threshold.height_threshold; ++ let (free_mib, total_mib) = cuda_memory_info().map(|(f, t)| (f >> 20, t >> 20)).unwrap_or((0, 0)); - tracing::debug!("Shard threshold: {shard_threshold}"); + eprintln!( -+ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", ++ "FLOOR opts gpu_memory_gb={gpu_memory_gb} free_mib={free_mib} total_mib={total_mib} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", + recursion_trace_allocation(), + opts.full_size_shards + ); opts.sharding_threshold.element_threshold = shard_threshold; opts.global_dependencies_opt = true; -@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier( +@@ -92,7 +188,7 @@ pub async fn recursion_prover_and_verifier( ) { let recursion_verifier = SP1CudaProverComponents::compress_verifier(); ( diff --git a/tools/fleet/box-prover.py b/tools/fleet/box-prover.py index 521a1afc9..48c4ea968 100644 --- a/tools/fleet/box-prover.py +++ b/tools/fleet/box-prover.py @@ -11,7 +11,8 @@ proving-v1, 272b025) and tools/proving-v1/pc2-segments.ps1 around the four binar offered again every pass until the segment's deadline (the 272b025 behaviour). 3. one export (igneum_exportSegments 0..last), one fixture per block (igneum-prove-export), one host run (igneum-prove-host --mode chain --chain ... --save-shards [--prev]) on the patched server (HOME=/opt/igneum-floor/home, - SP1_GPU_ELEMENT_THRESHOLD from THRESHOLD), the miner paused for the run when MINER=pause (prove-alone cards). + IGNEUM_PROVE_WORKLOAD=chain and no threshold by default, the host's memory profile from the card's free memory; + SP1_GPU_ELEMENT_THRESHOLD from THRESHOLD as a hand override), the miner paused for the run when MINER=pause (prove-alone cards). 4. every shard record signed (igneum-miner sign-record) and submitted (igneum_submitProofRecord); the segment record (sign-segment-record, igneum_submitSegmentRecord) once every shard is accepted and the statement equals the node's. 5. the paid state of every submitted segment polled each pass (igneum_getSegmentRecords); a state file for the @@ -255,7 +256,10 @@ while (time.time() - t_run0) / 3600 < RUN_HOURS: # chain if MINER == "pause": miner_stop() kill_server() - env = dict(os.environ, HOME=f"{FLOOR}/home", SP1_PROVER="cuda", RUST_LOG="off") + # the default job path (V6-07, 8 October 2026): no threshold override; the host picks its memory profile for the chain + # workload from the card's FREE memory (host/src/memory_profile.rs) and hands it to the floor server; THRESHOLD, when + # set, is a hand override the host keeps and names in its RESULT memory_profile line + env = dict(os.environ, HOME=f"{FLOOR}/home", SP1_PROVER="cuda", RUST_LOG="off", IGNEUM_PROVE_WORKLOAD="chain", IGNEUM_PROVE_DEVICE=str(DEV)) if THRESHOLD: env["SP1_GPU_ELEMENT_THRESHOLD"] = THRESHOLD args = [HOST, "--mode", "chain", "--chain", ",".join(fixtures), "--prover", WALLET, "--save-shards", "--out", f"{d}/chain-results.json"] if prev_file: args += ["--prev", prev_file] @@ -269,6 +273,10 @@ while (time.time() - t_run0) / 3600 < RUN_HOURS: except Exception: peak = 0 seg["peak_mib"] = peak open(f"{d}/chain.log", "w").write((rr.stdout if rr else "") + "\n" + (rr.stderr if rr else "TIMEOUT")) + if rr and rr.returncode == 78: + # the host refused the card's free memory for the chain workload before any setup (exit 78, one line) + line = [l for l in rr.stdout.split("\n") if "memory_profile refused" in l] + say(f"RESULT seg {first} chain REFUSED {stamp()} {line[-1].strip() if line else 'memory profile refused'}"); seg["failed"] = "memory"; close(first, "cancelled", "memory"); time.sleep(60); continue if not rr or rr.returncode != 0 or not os.path.exists(f"{d}/chain-results.json"): say(f"RESULT seg {first} chain FAILED {stamp()} rc={rr.returncode if rr else 'timeout'} wall={seg['chain_s']} s: {((rr.stderr if rr else '') or '')[-200:].strip()}"); seg["failed"] = "chain"; close(first, "cancelled", "chain" if rr else "timeout"); continue res = json.load(open(f"{d}/chain-results.json")) diff --git a/tools/fleet/box-setup.sh b/tools/fleet/box-setup.sh index 634c34238..30e8e308e 100755 --- a/tools/fleet/box-setup.sh +++ b/tools/fleet/box-setup.sh @@ -22,7 +22,7 @@ fail() { echo "RESULT setup_failed $1 $(stamp)"; exit 2; } ARCHS="${ARCHS:-86,89,120}"; LABEL="${LABEL:-box}"; WALLET="${WALLET:-}" PKG_URL=https://dl.igneum.network/dl/public/igneum-hive-0.3.12.tar.gz PKG_SHA=7972af92e7cd9a032303eca4d95b533f53e0e68d1b9cae5bfe406a5b7c30a454 -PATCH_SHA=e81cb0d03b291f9fd4bf0a109d6da2d7c897795c9ffd7f797c0ddce723eee2b1 +PATCH_SHA=f53871d92e27b743b82545c03bfa6ef21cf799f43fae6138ad4136637f3ce2c2 SEED=188.245.5.161:26611 echo "RESULT start $(stamp) label=$LABEL archs=$ARCHS host=$(hostname) nproc=$(nproc) ram_gb=$(( $(awk '/MemTotal/{print $2}' /proc/meminfo) / 1048576 )) disk_avail=$(df -BG /root | awk 'NR==2{print $4}')" echo "RESULT gpu $(nvidia-smi --query-gpu=name,memory.total,driver_version,pci.bus_id,power.limit,power.min_limit,power.max_limit,clocks.max.sm --format=csv,noheader 2>&1 | tr '\n' ';')" diff --git a/tools/fleet/floor-v5.patch b/tools/fleet/floor-v5.patch index 8b28f2aad..c3a8ec532 100644 --- a/tools/fleet/floor-v5.patch +++ b/tools/fleet/floor-v5.patch @@ -199,15 +199,16 @@ diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/pr index 5dccd9d..574d4fa 100644 --- a/sp1-gpu/crates/prover_components/src/builder.rs +++ b/sp1-gpu/crates/prover_components/src/builder.rs -@@ -23,28 +23,75 @@ use crate::{ +@@ -23,28 +23,124 @@ use crate::{ SP1CudaProverComponents, }; -+/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger -+/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a -+/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. -+/// The proof format, the verifier and the program ids do not change: the element threshold only decides where -+/// the executor splits shards, as upstream's own 24 GB tier already does. ++/// Igneum prover-floor patch (5 October 2026; the memory rules of V6-07, 8 October 2026). Upstream sizes every ++/// device buffer for a 24 GB card or larger and panics below that, whatever the shard. Here the card's FREE memory ++/// at start (or `SP1_GPU_MEMORY_BUDGET_GB`, the host's lease) picks a tier, read once for the process, and ++/// `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. The proof ++/// format, the verifier and the program ids do not change: the element threshold only decides where the executor ++/// splits shards, as upstream's own 24 GB tier already does. +fn env_usize(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} @@ -216,8 +217,17 @@ index 5dccd9d..574d4fa 100644 + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} + -+/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it -+/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16). ++/// The recursion trace allocation under the 24 GB tier (V6-07): a recursion key or shard uses 90,177,536 elements ++/// (35.6 M preprocessed and 54.5 M main; the floor sweep of 5 October 2026), so 2^26 + 2^25 = 100,663,296 holds it ++/// with the stacking slack `floor_capacity` adds; the pinned host copies (four per prover) shrink with it. Upstream's ++/// 2^27 stays on the 24 GB tier and above. Two distinct values: `floor_tests::recursion_branches_differ`. ++pub const RECURSION_TRACE_ALLOCATION_SMALL: usize = (1 << 26) + (1 << 25); ++ ++/// The core element threshold for a memory budget in GB (the budget is the FREE memory, +4, as upstream computed ++/// its tiers from the total: a 32 GB card alone reads 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16, an ++/// 8 GB card 12). Under the 18 tier the threshold is 2^26, the value every passing small-card row used (the ++/// alone-comp-26-v1 rows of 6 October 2026 on the 3060, 3080, 4060, 4060 Ti, 4070 and 5070; the 8 October 2026 ++/// proof_alone rows on the 3060 at 7,525 MiB and the 4060 at 7,532 MiB); no small-card row passed at 2^27 here. +pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 { + if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) { + ELEMENT_THRESHOLD @@ -226,31 +236,69 @@ index 5dccd9d..574d4fa 100644 + } else if gpu_memory_gb >= 18 { + (1 << 27) + (1 << 26) + } else { -+ 1 << 27 ++ 1 << 26 + } +} + -+/// The recursion trace allocation (elements) for a memory budget. ++/// The recursion trace allocation (elements) for a memory budget: upstream's on the 24 GB tier and above, the ++/// small constant under it. +pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize { + if gpu_memory_gb >= 24 { + RECURSION_TRACE_ALLOCATION + } else { -+ RECURSION_TRACE_ALLOCATION ++ RECURSION_TRACE_ALLOCATION_SMALL + } +} + ++/// The memory budget in GB, read ONCE for the process (the core opts and the recursion prover are built at ++/// different moments, after allocations that lower the free figure; one reading keeps both on one tier): ++/// `SP1_GPU_MEMORY_BUDGET_GB` when set (the host's lease), else the device's FREE memory as the driver reports it ++/// at the first call, never the total (a card beside a miner has the miner's resident set gone), +4 as upstream ++/// computed its tiers. +pub fn gpu_memory_gb() -> usize { -+ let gb = 1024.0 * 1024.0 * 1024.0; -+ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { -+ Some(b) => (b.ceil() as usize) + 4, -+ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4, -+ } ++ static BUDGET: std::sync::OnceLock = std::sync::OnceLock::new(); ++ *BUDGET.get_or_init(|| { ++ let gb = 1024.0 * 1024.0 * 1024.0; ++ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { ++ Some(b) => (b.ceil() as usize) + 4, ++ None => { ++ let (free, _total) = cuda_memory_info().unwrap(); ++ (((free as f64) / gb).ceil() as usize) + 4 ++ } ++ } ++ }) +} + +pub fn recursion_trace_allocation() -> usize { + env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION") + .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb())) +} ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::*; ++ ++ #[test] ++ fn small_tier_is_two_to_the_26() { ++ assert_eq!(element_threshold_for_budget(12, false), 1 << 26, "an 8 GB card reads 12"); ++ assert_eq!(element_threshold_for_budget(16, false), 1 << 26, "a 12 GB card reads 16"); ++ assert_eq!(element_threshold_for_budget(17, false), 1 << 26); ++ assert_eq!(element_threshold_for_budget(20, false), (1 << 27) + (1 << 26), "a 16 GB card reads 20"); ++ assert_eq!(element_threshold_for_budget(28, false), ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)); ++ assert_eq!(element_threshold_for_budget(28, true), ELEMENT_THRESHOLD); ++ assert_eq!(element_threshold_for_budget(36, false), ELEMENT_THRESHOLD); ++ } ++ ++ #[test] ++ fn recursion_branches_differ() { ++ assert_ne!(recursion_trace_allocation_for_budget(28), recursion_trace_allocation_for_budget(16)); ++ assert_eq!(recursion_trace_allocation_for_budget(28), RECURSION_TRACE_ALLOCATION); ++ assert_eq!(recursion_trace_allocation_for_budget(16), RECURSION_TRACE_ALLOCATION_SMALL); ++ assert_eq!(RECURSION_TRACE_ALLOCATION_SMALL, 100_663_296); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL < RECURSION_TRACE_ALLOCATION); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL >= 90_177_536 + (1 << 22)); ++ } ++} + pub fn local_gpu_opts() -> SP1CoreOpts { let mut opts = SP1CoreOpts::default(); @@ -266,7 +314,7 @@ index 5dccd9d..574d4fa 100644 - if gpu_memory_gb < 24 { - panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"); - } -+ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given. ++ // The card's FREE memory plus 4, as upstream computed its tiers from the total, or the budget given. + let gpu_memory_gb = gpu_memory_gb(); - let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 { @@ -278,17 +326,18 @@ index 5dccd9d..574d4fa 100644 + None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards), }; + let height_threshold = opts.sharding_threshold.height_threshold; ++ let (free_mib, total_mib) = cuda_memory_info().map(|(f, t)| (f >> 20, t >> 20)).unwrap_or((0, 0)); - tracing::debug!("Shard threshold: {shard_threshold}"); + eprintln!( -+ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", ++ "FLOOR opts gpu_memory_gb={gpu_memory_gb} free_mib={free_mib} total_mib={total_mib} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", + recursion_trace_allocation(), + opts.full_size_shards + ); opts.sharding_threshold.element_threshold = shard_threshold; opts.global_dependencies_opt = true; -@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier( +@@ -92,7 +188,7 @@ pub async fn recursion_prover_and_verifier( ) { let recursion_verifier = SP1CudaProverComponents::compress_verifier(); ( diff --git a/tools/fleet/floor.patch b/tools/fleet/floor.patch index 71a8d88e4..3d01b9ee5 100644 --- a/tools/fleet/floor.patch +++ b/tools/fleet/floor.patch @@ -199,15 +199,16 @@ diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/pr index 5dccd9d..574d4fa 100644 --- a/sp1-gpu/crates/prover_components/src/builder.rs +++ b/sp1-gpu/crates/prover_components/src/builder.rs -@@ -23,28 +23,75 @@ use crate::{ +@@ -23,28 +23,124 @@ use crate::{ SP1CudaProverComponents, }; -+/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger -+/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a -+/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. -+/// The proof format, the verifier and the program ids do not change: the element threshold only decides where -+/// the executor splits shards, as upstream's own 24 GB tier already does. ++/// Igneum prover-floor patch (5 October 2026; the memory rules of V6-07, 8 October 2026). Upstream sizes every ++/// device buffer for a 24 GB card or larger and panics below that, whatever the shard. Here the card's FREE memory ++/// at start (or `SP1_GPU_MEMORY_BUDGET_GB`, the host's lease) picks a tier, read once for the process, and ++/// `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. The proof ++/// format, the verifier and the program ids do not change: the element threshold only decides where the executor ++/// splits shards, as upstream's own 24 GB tier already does. +fn env_usize(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} @@ -216,8 +217,17 @@ index 5dccd9d..574d4fa 100644 + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} + -+/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it -+/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16). ++/// The recursion trace allocation under the 24 GB tier (V6-07): a recursion key or shard uses 90,177,536 elements ++/// (35.6 M preprocessed and 54.5 M main; the floor sweep of 5 October 2026), so 2^26 + 2^25 = 100,663,296 holds it ++/// with the stacking slack `floor_capacity` adds; the pinned host copies (four per prover) shrink with it. Upstream's ++/// 2^27 stays on the 24 GB tier and above. Two distinct values: `floor_tests::recursion_branches_differ`. ++pub const RECURSION_TRACE_ALLOCATION_SMALL: usize = (1 << 26) + (1 << 25); ++ ++/// The core element threshold for a memory budget in GB (the budget is the FREE memory, +4, as upstream computed ++/// its tiers from the total: a 32 GB card alone reads 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16, an ++/// 8 GB card 12). Under the 18 tier the threshold is 2^26, the value every passing small-card row used (the ++/// alone-comp-26-v1 rows of 6 October 2026 on the 3060, 3080, 4060, 4060 Ti, 4070 and 5070; the 8 October 2026 ++/// proof_alone rows on the 3060 at 7,525 MiB and the 4060 at 7,532 MiB); no small-card row passed at 2^27 here. +pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 { + if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) { + ELEMENT_THRESHOLD @@ -226,31 +236,69 @@ index 5dccd9d..574d4fa 100644 + } else if gpu_memory_gb >= 18 { + (1 << 27) + (1 << 26) + } else { -+ 1 << 27 ++ 1 << 26 + } +} + -+/// The recursion trace allocation (elements) for a memory budget. ++/// The recursion trace allocation (elements) for a memory budget: upstream's on the 24 GB tier and above, the ++/// small constant under it. +pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize { + if gpu_memory_gb >= 24 { + RECURSION_TRACE_ALLOCATION + } else { -+ RECURSION_TRACE_ALLOCATION ++ RECURSION_TRACE_ALLOCATION_SMALL + } +} + ++/// The memory budget in GB, read ONCE for the process (the core opts and the recursion prover are built at ++/// different moments, after allocations that lower the free figure; one reading keeps both on one tier): ++/// `SP1_GPU_MEMORY_BUDGET_GB` when set (the host's lease), else the device's FREE memory as the driver reports it ++/// at the first call, never the total (a card beside a miner has the miner's resident set gone), +4 as upstream ++/// computed its tiers. +pub fn gpu_memory_gb() -> usize { -+ let gb = 1024.0 * 1024.0 * 1024.0; -+ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { -+ Some(b) => (b.ceil() as usize) + 4, -+ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4, -+ } ++ static BUDGET: std::sync::OnceLock = std::sync::OnceLock::new(); ++ *BUDGET.get_or_init(|| { ++ let gb = 1024.0 * 1024.0 * 1024.0; ++ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { ++ Some(b) => (b.ceil() as usize) + 4, ++ None => { ++ let (free, _total) = cuda_memory_info().unwrap(); ++ (((free as f64) / gb).ceil() as usize) + 4 ++ } ++ } ++ }) +} + +pub fn recursion_trace_allocation() -> usize { + env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION") + .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb())) +} ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::*; ++ ++ #[test] ++ fn small_tier_is_two_to_the_26() { ++ assert_eq!(element_threshold_for_budget(12, false), 1 << 26, "an 8 GB card reads 12"); ++ assert_eq!(element_threshold_for_budget(16, false), 1 << 26, "a 12 GB card reads 16"); ++ assert_eq!(element_threshold_for_budget(17, false), 1 << 26); ++ assert_eq!(element_threshold_for_budget(20, false), (1 << 27) + (1 << 26), "a 16 GB card reads 20"); ++ assert_eq!(element_threshold_for_budget(28, false), ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)); ++ assert_eq!(element_threshold_for_budget(28, true), ELEMENT_THRESHOLD); ++ assert_eq!(element_threshold_for_budget(36, false), ELEMENT_THRESHOLD); ++ } ++ ++ #[test] ++ fn recursion_branches_differ() { ++ assert_ne!(recursion_trace_allocation_for_budget(28), recursion_trace_allocation_for_budget(16)); ++ assert_eq!(recursion_trace_allocation_for_budget(28), RECURSION_TRACE_ALLOCATION); ++ assert_eq!(recursion_trace_allocation_for_budget(16), RECURSION_TRACE_ALLOCATION_SMALL); ++ assert_eq!(RECURSION_TRACE_ALLOCATION_SMALL, 100_663_296); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL < RECURSION_TRACE_ALLOCATION); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL >= 90_177_536 + (1 << 22)); ++ } ++} + pub fn local_gpu_opts() -> SP1CoreOpts { let mut opts = SP1CoreOpts::default(); @@ -266,7 +314,7 @@ index 5dccd9d..574d4fa 100644 - if gpu_memory_gb < 24 { - panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"); - } -+ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given. ++ // The card's FREE memory plus 4, as upstream computed its tiers from the total, or the budget given. + let gpu_memory_gb = gpu_memory_gb(); - let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 { @@ -278,17 +326,18 @@ index 5dccd9d..574d4fa 100644 + None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards), }; + let height_threshold = opts.sharding_threshold.height_threshold; ++ let (free_mib, total_mib) = cuda_memory_info().map(|(f, t)| (f >> 20, t >> 20)).unwrap_or((0, 0)); - tracing::debug!("Shard threshold: {shard_threshold}"); + eprintln!( -+ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", ++ "FLOOR opts gpu_memory_gb={gpu_memory_gb} free_mib={free_mib} total_mib={total_mib} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", + recursion_trace_allocation(), + opts.full_size_shards + ); opts.sharding_threshold.element_threshold = shard_threshold; opts.global_dependencies_opt = true; -@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier( +@@ -92,7 +188,7 @@ pub async fn recursion_prover_and_verifier( ) { let recursion_verifier = SP1CudaProverComponents::compress_verifier(); ( From 0d208582a51b8099c9ba54ce34a604f424df6688 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 19:54:26 +0000 Subject: [PATCH 02/15] V6-07: the host's row from the lesser of the engine's grant and its free reading; prover-floor.md names the free-memory rule The window lane's engine half (reviewb-202 af5ade43) grants a fixed proof budget beside the miner (7,532 MiB plus 10 percent) while the card's free figure can be under it (a 12 GB card beside the 6.1 GiB miner: 6,159 MiB free), so memory_profile::lease_reading takes the lesser of IGNEUM_PROVE_MEM_BUDGET_MB and IGNEUM_PROVE_MEM_FREE_MB when both are set; test lease_takes_the_lesser_of_grant_and_free. Suite on box 3: 7 passed. Co-Authored-By: Claude Fable 5.1 --- docs/analysis/prover-floor.md | 12 ++++-- .../igneum-prove/host/src/memory_profile.rs | 40 +++++++++++++++---- 2 files changed, 41 insertions(+), 11 deletions(-) diff --git a/docs/analysis/prover-floor.md b/docs/analysis/prover-floor.md index 8c15af8c2..fe227bff1 100644 --- a/docs/analysis/prover-floor.md +++ b/docs/analysis/prover-floor.md @@ -48,10 +48,14 @@ from 20 M cycles is the threshold's padded area reached. The witness (5 to 22 KB ## What the patch does (`proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, three files) -1. `builder.rs`: the panic is gone; the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks the element threshold - from a tier table (`element_threshold_for_budget`: over 30 as read, the full 402.6 M; 24 to 30, upstream's 24 GB - figure; 18 to 24 (a 16 GB card), 2^27 + 2^26 = 201.3 M; under 18 (a 12 GB card), 2^27 = 134.2 M); - `SP1_GPU_ELEMENT_THRESHOLD` sets it directly and `SP1_GPU_RECURSION_TRACE_ALLOCATION` the recursion buffer. +1. `builder.rs`: the panic is gone; the card's FREE memory at start (read once per process; or + `SP1_GPU_MEMORY_BUDGET_GB`, the host's lease), never its total, picks the element threshold from a tier table + (`element_threshold_for_budget`: over 30 as read, the full 402.6 M; 24 to 30, upstream's 24 GB figure; 18 to 24 + (a 16 GB card alone), 2^27 + 2^26 = 201.3 M; under 18 (a 12 GB or 8 GB card, any card beside a miner), 2^26 = + 67.1 M, the value the passing small-card rows used) and the recursion trace allocation (upstream's 2^27 on the + 24 GB tier and above, 2^26 + 2^25 under it: V6-07, 8 October 2026, `docs/analysis/floor-memory-profile-2026-10-08.md`); + `SP1_GPU_ELEMENT_THRESHOLD` sets it directly and `SP1_GPU_RECURSION_TRACE_ALLOCATION` the recursion buffer. The + host's own profile table (`host/src/memory_profile.rs`) sets both by environment before the server starts. The chosen numbers are printed as a `FLOOR opts` line. Every other option is as upstream. 2. `jagged_tracegen/src/lib.rs`: with `SP1_GPU_FLOOR_LOG` set, every trace allocation prints its capacity and, after the shard's traces are in, the elements actually used and the device memory in use. diff --git a/proving/igneum-prove/host/src/memory_profile.rs b/proving/igneum-prove/host/src/memory_profile.rs index f6f1c82eb..61a6fb52a 100644 --- a/proving/igneum-prove/host/src/memory_profile.rs +++ b/proving/igneum-prove/host/src/memory_profile.rs @@ -10,8 +10,9 @@ //! 20:4x UK 8 October 2026; the engine owns admission and the host never re-reads a card the engine leased): //! 1. `IGNEUM_PROVE_MEM_BUDGET_MB`: the lease's grant, the hard ceiling for this process (the engine's coordinator, //! app/igneum-app `src/device.rs`, passes it with `IGNEUM_PROVE_DEVICE`, `IGNEUM_PROVE_WORKLOAD`, -//! `IGNEUM_PROVE_MEM_FREE_MB` and `IGNEUM_PROVE_DEADLINE_S` on every spawn); -//! 2. `IGNEUM_PROVE_MEM_FREE_MB`: the engine's own read at admission, when no grant is given; +//! `IGNEUM_PROVE_MEM_FREE_MB` and `IGNEUM_PROVE_DEADLINE_S` on every spawn); with `IGNEUM_PROVE_MEM_FREE_MB` +//! beside it the LESSER of the two decides (`lease_reading`); +//! 2. `IGNEUM_PROVE_MEM_FREE_MB` alone: the engine's own read at admission, when no grant is given; //! 3. `nvidia-smi --query-gpu=memory.free` on the device (`IGNEUM_PROVE_DEVICE`, else `IGNEUM_CUDA_DEVICE`, else the //! first of `CUDA_VISIBLE_DEVICES`, else 0): the fleet's `tools/fleet/box-prover.py` path and a hand run, where no //! engine admits the job. @@ -187,6 +188,20 @@ impl fmt::Display for Source { } } +/// The reading from the engine's two numbers: the grant is the ceiling and the engine's free reading is a fact, so +/// the row is chosen from the LESSER of the two when both are given (the window lane's engine half, 20:5x UK +/// 8 October 2026, grants a fixed proof budget beside the miner, 7,532 MiB plus 10 percent, while a 12 GB card +/// beside the 6.1 GiB miner has 6,159 MiB free: the grant alone would pick a row the card cannot hold, and the +/// refusal must come from the free figure). One number alone is used as it is. +pub fn lease_reading(budget: Option, free: Option) -> Option<(u64, Source)> { + match (budget, free) { + (Some(b), Some(f)) => Some((b.min(f), Source::Lease)), + (Some(b), None) => Some((b, Source::Lease)), + (None, Some(f)) => Some((f, Source::EngineRead)), + (None, None) => None, + } +} + fn env_u64(name: &str) -> Option { std::env::var(name).ok().and_then(|s| s.trim().parse::().ok()) } @@ -211,11 +226,10 @@ pub fn device_ordinal() -> String { /// The free memory on the card at this call, from the sources in the module note's order. pub fn read_free_mib() -> Option<(u64, Source)> { - if let Some(b) = env_u64("IGNEUM_PROVE_MEM_BUDGET_MB") { - return Some((b, Source::Lease)); - } - if let Some(b) = env_u64("IGNEUM_PROVE_MEM_FREE_MB") { - return Some((b, Source::EngineRead)); + let budget = env_u64("IGNEUM_PROVE_MEM_BUDGET_MB"); + let free = env_u64("IGNEUM_PROVE_MEM_FREE_MB"); + if let Some(choice) = lease_reading(budget, free) { + return Some(choice); } let dev = device_ordinal(); let out = std::process::Command::new("nvidia-smi") @@ -343,6 +357,18 @@ mod tests { assert_eq!(p.tier.recursion_trace_allocation, RECURSION_TRACE_ALLOCATION_SMALL); } + /// The engine's grant beside its free reading: the lesser decides; one alone is used as it is. + #[test] + fn lease_takes_the_lesser_of_grant_and_free() { + assert_eq!(lease_reading(Some(8_285), Some(6_159)), Some((6_159, Source::Lease))); + assert_eq!(lease_reading(Some(8_285), Some(12_287)), Some((8_285, Source::Lease))); + assert_eq!(lease_reading(Some(8_285), None), Some((8_285, Source::Lease))); + assert_eq!(lease_reading(None, Some(12_287)), Some((12_287, Source::EngineRead))); + assert_eq!(lease_reading(None, None), None); + // a 12 GB card beside the miner under a fixed grant is refused by its free figure + assert!(choose(Workload::Shard, lease_reading(Some(8_285), Some(6_159)).unwrap().0).is_err()); + } + /// The host's modes map to the three workloads; modes that never prove map to none. #[test] fn workload_from_mode() { From 6ea01336a36f68512a188972a225f38b0e77cd1d Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 20:49:20 +0000 Subject: [PATCH 03/15] V6-07: one floor patch; the canonical copy carried no grow_for_main and built a server that overflows its key buffer at prove proving/prover-floor/sp1-gpu-6.8.1-floor.patch and tools/fleet/floor.patch are now byte-equal to tools/fleet/floor-v5.patch (the complete patch: the key buffer grown for the main traces, the pool release threshold, the stage hook on a panic). Measured 8 October 2026 20:5x UK: the server built from the old canonical copy (floor-build.py's input; sha 75d0b4be) panicked at sp1-gpu/crates/jagged_tracegen/src/lib.rs:240 "range end index 37428736 out of range for slice of length 36700160" on the first recursion prove (the RTX 4060, chain mode); the server built from floor-v5 (db37c38b) proves. box-setup.sh re-pinned to the one sha. Co-Authored-By: Claude Fable 5.1 --- .../prover-floor/sp1-gpu-6.8.1-floor.patch | 181 +++++++++++++++++- tools/fleet/box-setup.sh | 2 +- tools/fleet/floor.patch | 80 ++++++++ 3 files changed, 259 insertions(+), 4 deletions(-) diff --git a/proving/prover-floor/sp1-gpu-6.8.1-floor.patch b/proving/prover-floor/sp1-gpu-6.8.1-floor.patch index dcc6fbad4..c3a8ec532 100644 --- a/proving/prover-floor/sp1-gpu-6.8.1-floor.patch +++ b/proving/prover-floor/sp1-gpu-6.8.1-floor.patch @@ -1,5 +1,26 @@ +diff --git a/sp1-gpu/crates/cuda/src/task.rs b/sp1-gpu/crates/cuda/src/task.rs +index a503a86..813016b 100644 +--- a/sp1-gpu/crates/cuda/src/task.rs ++++ b/sp1-gpu/crates/cuda/src/task.rs +@@ -149,7 +149,15 @@ pub enum GlobalTaskPoolBuildError { + + impl TaskPoolBuilder { + pub fn new() -> Self { +- Self { capacity: None, device: CudaDevice(0), mem_release_threshold: u64::MAX } ++ // Igneum prover-floor patch: upstream keeps every freed device allocation in the pool for the process's ++ // life (threshold u64::MAX), so the prover holds its high-water mark between shards on a card it shares ++ // with a miner. `SP1_GPU_MEM_RELEASE_THRESHOLD=` sets the pool's release threshold (0 returns ++ // freed memory to the driver at once); unset, upstream's behaviour. ++ let mem_release_threshold = std::env::var("SP1_GPU_MEM_RELEASE_THRESHOLD") ++ .ok() ++ .and_then(|s| s.parse::().ok()) ++ .unwrap_or(u64::MAX); ++ Self { capacity: None, device: CudaDevice(0), mem_release_threshold } + } + + pub fn num_tasks(mut self, num_tasks: usize) -> Self { diff --git a/sp1-gpu/crates/jagged_tracegen/src/lib.rs b/sp1-gpu/crates/jagged_tracegen/src/lib.rs -index 579f70a..09e73e8 100644 +index 579f70a..2264044 100644 --- a/sp1-gpu/crates/jagged_tracegen/src/lib.rs +++ b/sp1-gpu/crates/jagged_tracegen/src/lib.rs @@ -481,6 +481,33 @@ async fn device_preprocessed_tracegen>( @@ -64,7 +85,81 @@ index 579f70a..09e73e8 100644 log_stacking_height, max_log_row_count, backend, -@@ -984,9 +1021,15 @@ pub async fn full_tracegen>( +@@ -906,6 +943,11 @@ pub async fn main_tracegen, A: CudaTracegenAir>( + + log_chip_stats(machine, &chip_set, &traces); + ++ // Igneum prover-floor patch: the key's buffer is sized to its preprocessed traces at setup (upstream sized it ++ // for a whole shard), so grow it here to what this shard needs before the main traces are appended: a bigger ++ // dense buffer and column index, the preprocessed region copied device to device, swapped into the key. ++ grow_for_main(&mut jagged_traces.preprocessed_traces, &traces, log_stacking_height, backend); ++ + copy_main_jagged_traces( + traces, + &mut jagged_traces.preprocessed_traces, +@@ -918,6 +960,61 @@ pub async fn main_tracegen, A: CudaTracegenAir>( + (public_values, chip_set, permit) + } + ++/// Igneum prover-floor patch: see `main_tracegen`. The need is the preprocessed phase as laid out (its padded ++/// end, `preprocessed_offset`) plus the main traces padded to the stacking height plus one stacking height of ++/// slack; a buffer at least that big is left alone. The process aborts, loudly, if the copy cannot be made, ++/// because a panic inside a prover task is what left sweep 2 hanging on the client's socket. ++fn grow_for_main( ++ jagged: &mut JaggedTraceMle, ++ main_traces: &BTreeMap>, ++ log_stacking_height: u32, ++ backend: &TaskScope, ++) { ++ let pre_end = jagged.dense().preprocessed_offset; ++ let needed = pre_end ++ + padded_trace_elements(main_traces, log_stacking_height) ++ + (1 << log_stacking_height); ++ let have = jagged.dense().dense.capacity(); ++ if have >= needed { ++ return; ++ } ++ let mut new_dense: Buffer = Buffer::with_capacity_in(needed, backend.clone()); ++ let mut new_col_index: Buffer = ++ Buffer::with_capacity_in(needed >> 1, backend.clone()); ++ unsafe { ++ new_dense.assume_init(); ++ new_col_index.assume_init(); ++ } ++ { ++ let JaggedMle { dense_data, col_index, .. } = &mut **jagged; ++ let src_dense: &Slice<_, _> = &dense_data.dense[..pre_end]; ++ let dst_dense: &mut Slice<_, _> = &mut new_dense[..pre_end]; ++ let src_col: &Slice<_, _> = &col_index[..pre_end >> 1]; ++ let dst_col: &mut Slice<_, _> = &mut new_col_index[..pre_end >> 1]; ++ unsafe { ++ if dst_dense.copy_from_slice(src_dense, backend).is_err() ++ || dst_col.copy_from_slice(src_col, backend).is_err() ++ { ++ eprintln!("FLOOR grow FAILED: could not copy the preprocessed region ({pre_end} elements) into the grown buffer ({needed} elements); aborting instead of hanging"); ++ std::process::abort(); ++ } ++ } ++ } ++ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() { ++ eprintln!( ++ "FLOOR grow key buffer {have} -> {needed} elements (preprocessed {pre_end}, {} bytes)", ++ needed * 6 ++ ); ++ } ++ let JaggedMle { dense_data, col_index, .. } = &mut **jagged; ++ dense_data.dense = new_dense; ++ *col_index = new_col_index; ++ unsafe { ++ dense_data.dense.set_len(pre_end); ++ col_index.set_len(pre_end >> 1); ++ } ++} ++ + #[allow(clippy::too_many_arguments)] + pub async fn main_tracegen_permit, A: CudaTracegenAir>( + machine: &Machine, +@@ -984,9 +1081,15 @@ pub async fn full_tracegen>( log_chip_stats(machine, &chip_set, &main_traces); @@ -81,7 +176,7 @@ index 579f70a..09e73e8 100644 log_stacking_height, max_log_row_count, backend, -@@ -1002,6 +1045,18 @@ pub async fn full_tracegen>( +@@ -1002,6 +1105,18 @@ pub async fn full_tracegen>( ) .await; @@ -251,6 +346,86 @@ index 5dccd9d..574d4fa 100644 .await, recursion_verifier, ) +diff --git a/sp1-gpu/crates/server/src/main.rs b/sp1-gpu/crates/server/src/main.rs +index 65e94f5..498357c 100644 +--- a/sp1-gpu/crates/server/src/main.rs ++++ b/sp1-gpu/crates/server/src/main.rs +@@ -17,9 +17,55 @@ struct Args { + version: bool, + } + ++/// Igneum prover-floor patch (6 October 2026, the GPU fleet's finding): a panic inside a prover task (an allocation ++/// the card cannot meet, `cudaMallocAsync` failing and `Buffer::with_capacity_in` panicking in a tokio worker) left ++/// the request's future waiting for ever, the client on its socket, the card at 0% for the 15 minutes until someone ++/// killed it (the 8 GB and 10 GB cards at threshold 2^27). The server must fail the shard instead: this hook names ++/// the stage from the panic's location and exits, so the client's proof fails at once with the server's last line. ++fn stage_of(location: &str) -> &'static str { ++ let l = location.to_ascii_lowercase(); ++ if l.contains("jagged_tracegen") || l.contains("/tracegen") { ++ "trace generation" ++ } else if l.contains("commit") || l.contains("basefold") || l.contains("merkle") { ++ "the commit (codewords and Merkle trees)" ++ } else if l.contains("logup_gkr") { ++ "LogUp GKR" ++ } else if l.contains("zerocheck") { ++ "the zerocheck" ++ } else if l.contains("jagged") { ++ "the jagged sumcheck" ++ } else if l.contains("prover_components") || l.contains("recursion") || l.contains("sp1-prover") || l.contains("sp1_prover") { ++ "the recursion (compression)" ++ } else if l.contains("cuda") || l.contains("slop") { ++ "a device allocation" ++ } else { ++ "the prover" ++ } ++} ++ ++fn install_abort_on_panic() { ++ std::panic::set_hook(Box::new(|info| { ++ let location = info.location().map(|l| format!("{}:{}", l.file(), l.line())).unwrap_or_else(|| "unknown".into()); ++ let message = info ++ .payload() ++ .downcast_ref::<&str>() ++ .map(|s| s.to_string()) ++ .or_else(|| info.payload().downcast_ref::().cloned()) ++ .unwrap_or_default(); ++ let oom = message.to_ascii_lowercase().contains("alloc") || message.contains("MemoryAllocation") || message.contains("OUT_OF_MEMORY"); ++ eprintln!( ++ "FLOOR abort: {} failed at {location}: {message}{}; the server exits so the client's proof fails instead of waiting", ++ stage_of(&location), ++ if oom { " (the card's memory could not meet an allocation: lower SP1_GPU_ELEMENT_THRESHOLD one notch)" } else { "" } ++ ); ++ std::process::exit(70); ++ })); ++} ++ + #[tokio::main] + #[allow(clippy::print_stdout)] + async fn main() { ++ install_abort_on_panic(); + tracing_subscriber::fmt::init(); + + let args = Args::parse(); +@@ -40,3 +86,19 @@ async fn main() { + eprintln!("Error running server: {e}"); + } + } ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::stage_of; ++ ++ #[test] ++ fn the_stage_is_named_from_the_panic_location() { ++ assert_eq!(stage_of("sp1-gpu/crates/jagged_tracegen/src/lib.rs:240"), "trace generation"); ++ assert_eq!(stage_of("sp1-gpu/crates/basefold/src/fri.rs:97"), "the commit (codewords and Merkle trees)"); ++ assert_eq!(stage_of("sp1-gpu/crates/logup_gkr/src/tracegen.rs:72"), "LogUp GKR"); ++ assert_eq!(stage_of("sp1-gpu/crates/zerocheck/src/prover.rs:1163"), "the zerocheck"); ++ assert_eq!(stage_of("sp1-gpu/crates/prover_components/src/builder.rs:70"), "the recursion (compression)"); ++ assert_eq!(stage_of("sp1-gpu/crates/cuda/src/stream.rs:330"), "a device allocation"); ++ assert_eq!(stage_of("somewhere/else.rs:1"), "the prover"); ++ } ++} diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs index 4035f1f..0d0d907 100644 --- a/sp1-gpu/crates/server/src/server.rs diff --git a/tools/fleet/box-setup.sh b/tools/fleet/box-setup.sh index 30e8e308e..bec5c23b1 100755 --- a/tools/fleet/box-setup.sh +++ b/tools/fleet/box-setup.sh @@ -22,7 +22,7 @@ fail() { echo "RESULT setup_failed $1 $(stamp)"; exit 2; } ARCHS="${ARCHS:-86,89,120}"; LABEL="${LABEL:-box}"; WALLET="${WALLET:-}" PKG_URL=https://dl.igneum.network/dl/public/igneum-hive-0.3.12.tar.gz PKG_SHA=7972af92e7cd9a032303eca4d95b533f53e0e68d1b9cae5bfe406a5b7c30a454 -PATCH_SHA=f53871d92e27b743b82545c03bfa6ef21cf799f43fae6138ad4136637f3ce2c2 +PATCH_SHA=9098c3e5979d057031f43588668201d1f7a53ab17e980c1eaf896cd5e6f38575 SEED=188.245.5.161:26611 echo "RESULT start $(stamp) label=$LABEL archs=$ARCHS host=$(hostname) nproc=$(nproc) ram_gb=$(( $(awk '/MemTotal/{print $2}' /proc/meminfo) / 1048576 )) disk_avail=$(df -BG /root | awk 'NR==2{print $4}')" echo "RESULT gpu $(nvidia-smi --query-gpu=name,memory.total,driver_version,pci.bus_id,power.limit,power.min_limit,power.max_limit,clocks.max.sm --format=csv,noheader 2>&1 | tr '\n' ';')" diff --git a/tools/fleet/floor.patch b/tools/fleet/floor.patch index 3d01b9ee5..c3a8ec532 100644 --- a/tools/fleet/floor.patch +++ b/tools/fleet/floor.patch @@ -346,6 +346,86 @@ index 5dccd9d..574d4fa 100644 .await, recursion_verifier, ) +diff --git a/sp1-gpu/crates/server/src/main.rs b/sp1-gpu/crates/server/src/main.rs +index 65e94f5..498357c 100644 +--- a/sp1-gpu/crates/server/src/main.rs ++++ b/sp1-gpu/crates/server/src/main.rs +@@ -17,9 +17,55 @@ struct Args { + version: bool, + } + ++/// Igneum prover-floor patch (6 October 2026, the GPU fleet's finding): a panic inside a prover task (an allocation ++/// the card cannot meet, `cudaMallocAsync` failing and `Buffer::with_capacity_in` panicking in a tokio worker) left ++/// the request's future waiting for ever, the client on its socket, the card at 0% for the 15 minutes until someone ++/// killed it (the 8 GB and 10 GB cards at threshold 2^27). The server must fail the shard instead: this hook names ++/// the stage from the panic's location and exits, so the client's proof fails at once with the server's last line. ++fn stage_of(location: &str) -> &'static str { ++ let l = location.to_ascii_lowercase(); ++ if l.contains("jagged_tracegen") || l.contains("/tracegen") { ++ "trace generation" ++ } else if l.contains("commit") || l.contains("basefold") || l.contains("merkle") { ++ "the commit (codewords and Merkle trees)" ++ } else if l.contains("logup_gkr") { ++ "LogUp GKR" ++ } else if l.contains("zerocheck") { ++ "the zerocheck" ++ } else if l.contains("jagged") { ++ "the jagged sumcheck" ++ } else if l.contains("prover_components") || l.contains("recursion") || l.contains("sp1-prover") || l.contains("sp1_prover") { ++ "the recursion (compression)" ++ } else if l.contains("cuda") || l.contains("slop") { ++ "a device allocation" ++ } else { ++ "the prover" ++ } ++} ++ ++fn install_abort_on_panic() { ++ std::panic::set_hook(Box::new(|info| { ++ let location = info.location().map(|l| format!("{}:{}", l.file(), l.line())).unwrap_or_else(|| "unknown".into()); ++ let message = info ++ .payload() ++ .downcast_ref::<&str>() ++ .map(|s| s.to_string()) ++ .or_else(|| info.payload().downcast_ref::().cloned()) ++ .unwrap_or_default(); ++ let oom = message.to_ascii_lowercase().contains("alloc") || message.contains("MemoryAllocation") || message.contains("OUT_OF_MEMORY"); ++ eprintln!( ++ "FLOOR abort: {} failed at {location}: {message}{}; the server exits so the client's proof fails instead of waiting", ++ stage_of(&location), ++ if oom { " (the card's memory could not meet an allocation: lower SP1_GPU_ELEMENT_THRESHOLD one notch)" } else { "" } ++ ); ++ std::process::exit(70); ++ })); ++} ++ + #[tokio::main] + #[allow(clippy::print_stdout)] + async fn main() { ++ install_abort_on_panic(); + tracing_subscriber::fmt::init(); + + let args = Args::parse(); +@@ -40,3 +86,19 @@ async fn main() { + eprintln!("Error running server: {e}"); + } + } ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::stage_of; ++ ++ #[test] ++ fn the_stage_is_named_from_the_panic_location() { ++ assert_eq!(stage_of("sp1-gpu/crates/jagged_tracegen/src/lib.rs:240"), "trace generation"); ++ assert_eq!(stage_of("sp1-gpu/crates/basefold/src/fri.rs:97"), "the commit (codewords and Merkle trees)"); ++ assert_eq!(stage_of("sp1-gpu/crates/logup_gkr/src/tracegen.rs:72"), "LogUp GKR"); ++ assert_eq!(stage_of("sp1-gpu/crates/zerocheck/src/prover.rs:1163"), "the zerocheck"); ++ assert_eq!(stage_of("sp1-gpu/crates/prover_components/src/builder.rs:70"), "the recursion (compression)"); ++ assert_eq!(stage_of("sp1-gpu/crates/cuda/src/stream.rs:330"), "a device allocation"); ++ assert_eq!(stage_of("somewhere/else.rs:1"), "the prover"); ++ } ++} diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs index 4035f1f..0d0d907 100644 --- a/sp1-gpu/crates/server/src/server.rs From a556c3ad53d0eaf2e70ed769884776e4c8988c19 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 21:20:21 +0000 Subject: [PATCH 04/15] V6-07: the aggregate and chain floors at 8,500 MiB from the measured aggregation peak; the rows and their consequences docs/analysis/floor-memory-profile-2026-10-08.md: the 3060 and 4060 rows on the default, unoverridden job path (the served 0317 host on the V6-07 server: 3060 shard 11.4 s at 7,601 MiB, the whole chain 27.4 s at an 8,306 MiB aggregation peak, both VERIFIED; 4060 shard 16.3 s at 7,504 MiB VERIFIED, the chain aborts at the aggregation's 486 MiB allocation on 176 MiB free), the host's profile and refusal rows (exit 78 under the floor and beside the miner on both cards, the miner unharmed), the recursion constant isolated (no change of the device peak), the known-failed first server, the master-host 0-bytes fault with its NOT RUN rows, and the consequences per card tier. memory_profile.rs: aggregate and chain floors 8,500 MiB (an 8 GB card proves shards and is refused the chain before setup), tests re-pinned; suite 7 passed on box 3. Co-Authored-By: Claude Fable 5.1 --- .../floor-memory-profile-2026-10-08.md | 121 ++++++++++++++++++ .../igneum-prove/host/src/memory_profile.rs | 26 ++-- 2 files changed, 138 insertions(+), 9 deletions(-) create mode 100644 docs/analysis/floor-memory-profile-2026-10-08.md diff --git a/docs/analysis/floor-memory-profile-2026-10-08.md b/docs/analysis/floor-memory-profile-2026-10-08.md new file mode 100644 index 000000000..baab87dd9 --- /dev/null +++ b/docs/analysis/floor-memory-profile-2026-10-08.md @@ -0,0 +1,121 @@ +# The floor's memory profile (V6-07), 8 October 2026 + +Master review R1, residual V6-07 (`docs/plans/igneum-2.0-master/evidence/04_full_system/IGNEUM_V6_Full_System_Review.md`, pages 210 to 212): the floor patch picked its limits from the card's total VRAM, not from what was free; its small-card element threshold defaulted to 2^27 where every passing row had set 2^26 by hand; its recursion-allocation budget returned one constant in both branches. The order: a pinned memory profile per workload read from free memory, with the app lane's device coordinator, and the 3060 and 4060 rows rerun on the default, unoverridden job path. This document is the record: the code (section 1), the table (section 2), the rows as the pods wrote them (section 3), the consequences per card tier beside each number (section 4), the coordinator hook and what is next (section 5). + +Branch `v607-floor-memory` off the box mirror master cef5234b5. Code commits e6abe8c2c (the patch and the host profile), 573dad0ad (the lesser of grant and free), 8c1fb754c (one floor patch), then the floors re-pinned from the rows (the commit carrying this document). Clocks below are UTC as the pods wrote them; UK time is one hour later. + +## 1. What changed + +| Item | Where | Before | After | +|---|---|---|---| +| The limits' source | `proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, `builder.rs` `gpu_memory_gb()` (the fleet's copies `tools/fleet/floor.patch`, `floor-v5.patch` carry the same hunk; `box-setup.sh` re-pinned) | `cuda_memory_info().1` (the total), +4 as upstream | `cuda_memory_info().0` (the FREE memory as the driver reports it), +4, read ONCE per process in a `OnceLock` so the core opts and the recursion prover, built at different moments, sit on one tier; `SP1_GPU_MEMORY_BUDGET_GB` (the host's lease) still overrides | +| The small-card element threshold | same, `element_threshold_for_budget` | `1 << 27` under the 18 tier | `1 << 26`, the value the passing rows used (section 2 cites them) | +| The recursion-allocation budget | same, `recursion_trace_allocation_for_budget` | `RECURSION_TRACE_ALLOCATION` in both branches | upstream's 2^27 on the 24 GB tier and above; `RECURSION_TRACE_ALLOCATION_SMALL` = 2^26 + 2^25 = 100,663,296 under it (a recursion key or shard uses 90,177,536 elements, `docs/analysis/prover-floor.md` sweep 1; the patch's `floor_capacity` sizes the device buffer to the need, so the constant caps the buffer and shrinks the four pinned host copies per prover); `floor_tests::recursion_branches_differ` pins that the two branches differ and that the small one clears the measured use with a stacking height of slack; `floor_tests::small_tier_is_two_to_the_26` pins the tiers | +| The FLOOR opts line | same, `local_gpu_opts` | `gpu_memory_gb`, thresholds | adds `free_mib` and `total_mib` so a log names what the server read | +| The pinned profile per workload | `proving/igneum-prove/host/src/memory_profile.rs` (new), wired in `main.rs` before the SP1 client is built | nothing: the server guessed from the total; the fleet set `SP1_GPU_ELEMENT_THRESHOLD` by hand | one table (section 2) with tests pinning every value; the row is chosen from the engine's lease or, with no engine, from `nvidia-smi memory.free` on the device, and handed to the server by environment (`SP1_GPU_ELEMENT_THRESHOLD`, `SP1_GPU_RECURSION_TRACE_ALLOCATION`, `SP1_GPU_MEMORY_BUDGET_GB`); a hand override already in the environment is kept and named; under the workload's floor the host prints one line and exits 78 before any setup | +| The fleet's default path | `tools/fleet/box-prover.py` | `SP1_GPU_ELEMENT_THRESHOLD` from `THRESHOLD`, else the server's total-VRAM guess | `IGNEUM_PROVE_WORKLOAD=chain` and `IGNEUM_PROVE_DEVICE` on the host run, no threshold unless `THRESHOLD` is set by hand; a refusal (exit 78) closes the segment as cancelled/memory | + +Tests: `cargo test -p igneum-prove-host memory_profile` (box 3, section 6) and the patch's `floor_tests` (run on the build pod after the server build, section 6). + +## 2. The table + +The host's `TIERS` (largest first; the free-memory lines are the card tiers as the floor patch reads them, free GiB rounded up plus 4 as upstream computed its tiers from the total): + +| Tier | Free memory at start | Element threshold | Recursion trace allocation | Who lands here | +|---|---|---|---|---| +| full | 26,624 MiB and up | 2^28 + 2^27 = 402,653,184 | 2^27 = 134,217,728 | a 32 GB card alone | +| 24gb | 20,480 MiB and up | 285,212,672 (upstream's 24 GB figure) | 2^27 | a 24 GB card alone | +| 16gb | 14,336 MiB and up | 2^27 + 2^26 = 201,326,592 (the patch's own figure, unmeasured on this fixture) | 2^26 + 2^25 = 100,663,296 | a 16 GB card alone | +| small | under 14,336 MiB | 2^26 = 67,108,864 | 100,663,296 | a 12 GB or 8 GB card alone; any card beside a miner's resident set | + +Floors per workload (the least free memory the host runs in; under it the refusal, exit 78): shard 7,700 MiB, aggregate 8,500 MiB, chain 8,500 MiB. The shard floor is the small tier's measured peak (7,525 MiB on the 3060, 7,532 on the 4060, section 3) plus headroom for the driver's own context. The aggregate and chain floors are 8,500 MiB: the aggregation is the peak of a chain run, 8,306 MiB on the RTX 3060 (section 3, D2), and the RTX 4060 (7,807 MiB free) aborted at it, so an 8 GB card is refused the chain and the aggregation before any setup and proves shards only. + +The rows cited for 2^26 on small cards: `docs/analysis/prover-tiers-real-cards.md` (6 October 2026, the matrix's `alone-comp-26-v1` point, compressed, verified: RTX 3060 7.4 GB 14.4 s; RTX 3080 8.0 GB 7.1 s; RTX 4060 7.4 GB 18.4 s; RTX 4060 Ti 8 GB 7.6 GB 9.6 s; RTX 4060 Ti 16 GB 7.8 GB 11.6 s; RTX 4070 7.6 GB 12.1 s; RTX 5070 7.6 GB 4.8 s) and `docs/analysis/class-v6/coexist-rows.md` (8 October 2026, `proof_alone` at `SP1_GPU_ELEMENT_THRESHOLD=67108864`: RTX 3060 peak 7,525 MiB 13.2 s verified; RTX 4060 peak 7,532 MiB 8.2 s verified). No small-card row ever passed at 2^27 on this host path; the 10 GB tier's 2^27 reading of 7 October (8,642 MiB alone on a 3080) ran on the segment host and is not this path. + +Every value above is pinned by `memory_profile::tests::table_is_pinned`, `tiers_from_free_memory`, `refuses_under_the_floor`, `recursion_branches_differ`, `workload_from_mode` and `env_for_the_server`; a change of the table is a change of the profile and needs its rows. + +## 3. The rows, rerun on the default, unoverridden job path + +Every row is a RESULT line as the pod wrote it (the raw run logs, host logs and 1 Hz `nvidia-smi` samples are kept under the lane's scratch `v607/rows/