diff --git a/docs/analysis/prover-tiers-real-cards.md b/docs/analysis/prover-tiers-real-cards.md index 9dc58312..cf4f2aca 100644 --- a/docs/analysis/prover-tiers-real-cards.md +++ b/docs/analysis/prover-tiers-real-cards.md @@ -33,22 +33,27 @@ and the 4060 Ti 8 GB at 2^27, 15 minutes each until killed), so every prover run | Card | VRAM GB | Idle MiB | Miner | Stock SP1 6.8.1 | Patched, proves alone (own) | Beside the miner (peak) | Core-only beside the miner (own) | Verdict | |---|---|---|---|---|---|---|---|---| +| RTX 3060 | 12 | 1 | 23.78 MH/s at 103.7 W, 1.4 GB | refused: thread 'tokio-rt-worker' (48952) panicked at sp1-gpu/crates/ | 7.4 GB, 14.4 s (alone-comp-26-v1) | 8.9 GB peak, 37.5 s | 5.6 GB, 27.2 s | mines and proves | +| RTX 3080 | 10 | 11 | 40.82 MH/s at 204.9 W, 1.5 GB | refused: thread 'tokio-rt-worker' (49593) panicked at sp1-gpu/crates/ | 8.0 GB, 7.1 s (alone-comp-26-v1) | 9.2 GB peak, 25.6 s | 5.9 GB, 19.2 s | mines and proves | +| RTX 4060 Ti | 16 | 0 | 17.58 MH/s at 72.3 W, 1.4 GB | refused: thread 'tokio-rt-worker' (49293) panicked at sp1-gpu/crates/ | 7.8 GB, 11.6 s (alone-comp-26-v1) | 9.0 GB peak, 34.6 s | 5.8 GB, 25.9 s | mines and proves | | RTX 4070 | 12 | 9 | 24.99 MH/s at 91.1 W, 1.4 GB | refused: thread 'tokio-rt-worker' (47475) panicked at sp1-gpu/crates/ | 7.6 GB, 12.1 s (alone-comp-26-v1) | 10.1 GB peak, 27.3 s | 5.6 GB, 14.3 s | mines and proves | | RTX 4090 | 24 | 1 | 52.25 MH/s at 183.1 W, 1.7 GB | proved 5.6 s at 17.4 GB | 7.9 GB, 6.3 s (alone-comp-26-v1) | 10.7 GB peak, 26.1 s | 6.1 GB, 10.6 s | mines and proves | | RTX 5070 | 12 | 2 | 41.89 MH/s at 137.0 W, 2.7 GB | refused: thread 'tokio-rt-worker' (53680) panicked at sp1-gpu/crates/ | 7.6 GB, 4.8 s (alone-comp-26-v1) | 10.2 GB peak, 37.2 s | 5.8 GB, 19.8 s | mines and proves | +| RTX 5090 | 32 | 2 | 98.48 MH/s at 258.2 W, 1.8 GB | proved 8.4 s at 18.3 GB | 8.0 GB, 6.3 s (alone-comp-26-v1) | 9.9 GB peak, 10.7 s | 6.3 GB, 7.4 s | mines and proves | | RTX A5000 | 24 | 1 | 47.7 MH/s at 222.7 W, 1.5 GB | proved 6.4 s at 17.2 GB | 7.7 GB, 8.3 s (alone-comp-26-v1) | 10.5 GB peak, 34.6 s | 6.0 GB, 18.2 s | mines and proves | ## What the rows say, per tier -Measured so far (the 4090, A5000, 4070 and 5070 complete; the 3080, 4060 Ti 8 and 16 GB, 3060, 4060, 3090 and 5090 running): +Measured so far (eight cards complete: 4090, A5000, 5090, 4070, 5070, 3060, 3080, 4060 Ti 16 GB; the 4060, 4060 Ti 8 GB and 3090 running): | Tier | What the cards say | Consequence | What is being done | |---|---|---|---| | 24 GB (4090, A5000) | the STOCK server proves the v1 shard (5.6 s at 17.4 GB on the 4090, 6.4 s at 17.2 GB on the A5000), so no patch is needed to prove alone; the patched 2^26 profile does it in 7.7 to 7.9 GB (6.3 and 8.3 s) and beside the miner the peak is 10.5 to 10.7 GB at 26 to 35 s (the miner costs 4.1x on the 4090) | mines and proves with 13 GB to spare; the patched profile frees 9.5 GB for nothing but a 1.1 to 1.3x slower proof, so a 24 GB card keeps upstream's threshold and the public line "24 GB: mines and proves" stands on real cards | the 24 GB rows go into the fleet night as compressed provers at the default tier | -| 12 GB (4070, 5070) | the stock server refuses (the 24 GB gate); patched 2^26 proves alone at 7.6 GB (12.1 s on the 4070, 4.8 s on the 5070); BESIDE THE MINER the peak is 10.1 to 10.2 GB of 12 GB (27.3 s and 37.2 s), verified; core-only beside the miner 5.6 to 5.8 GB (14.3 and 19.8 s) | a 12 GB card mines and proves compressed shards on the patched server with about 2 GB to spare before the display (Windows and a monitor take 0.5 to 1.5 GB, so a desktop 12 GB card is at the edge; a headless Linux one is fine); the 9.0 GB line of prover-floor.md is not needed for Linux headless, and core-only (5.6 to 5.8 GB) keeps 6 GB spare for a desktop | the public line becomes "12 GB: mines and proves on Linux (the patched server), proves alone on a desktop; core-only mine-and-prove on a desktop once the hand-off ships"; the 4070 and 5070 join the fleet night with the miner PAUSED per segment (prove-alone profile), the 10.2 GB beside-row is the mine-and-prove candidate for a second night | -| 16 GB (4060 Ti 16 GB) | running | | | -| 10 GB (3080) and 8 GB (4060, 4060 Ti 8 GB) | the 3080 proves alone at 2^26 (7.0 to 7.2 s, 7.9 to 8.2 GB) and the 4060 Ti 8 GB too (9.6 s, 7.74 GB of 8.19); 2^27 does not fit either and the server hangs instead of failing | an 8 GB card proves alone (not beside its miner: 7.7 + 1.4 GB is over 8 GB); the public floor moves from "12 GB proves" to "8 GB proves alone, slowly"; a profile under 7.5 GB (2^25 compressed, core-only 2^25 at 5.9 GB) is the 8 GB mine-and-prove candidate | the beside-the-miner and 2^25 rows on the 3080 and 4060 Ti are running | -| 32 GB (5090) | running | | | +| 12 GB (3060, 4070, 5070) | the stock server refuses (the 24 GB gate); patched 2^26 proves alone at 7.4 to 7.6 GB (14.4 s on the 3060, 12.1 s on the 4070, 4.8 s on the 5070); BESIDE THE MINER the peak is 8.9 GB (3060, 37.5 s) to 10.1 to 10.2 GB (4070, 5070; 27.3 s and 37.2 s) of 12 GB, verified; core-only beside the miner 5.6 to 5.8 GB (14.3 to 27.2 s) | a 12 GB card mines and proves compressed shards on the patched server with about 2 GB to spare before the display (Windows and a monitor take 0.5 to 1.5 GB, so a desktop 12 GB card is at the edge; a headless Linux one is fine); the 9.0 GB line of prover-floor.md is not needed for Linux headless, and core-only (5.6 to 5.8 GB) keeps 6 GB spare for a desktop | the public line becomes "12 GB: mines and proves on Linux (the patched server), proves alone on a desktop; core-only mine-and-prove on a desktop once the hand-off ships"; the 4070 and 5070 join the fleet night with the miner PAUSED per segment (prove-alone profile), the 10.2 GB beside-row is the mine-and-prove candidate for a second night | +| 16 GB (4060 Ti 16 GB) | patched 2^26 alone 7.8 GB in 11.6 s; beside the miner 9.0 GB peak of 16 GB in 34.6 s; core-only beside 5.8 GB | mines and proves with 7 GB spare, display or not; the 2^27 profile (10 GB alone) fits too | joins the fleet night mining and proving compressed | +| 10 GB (3080) | proves alone at 2^26 (7.1 s, 8,158 MiB own); BESIDE THE MINER compressed 2^26 verified at a 9,412 MiB peak of 10,240 in 25.6 s; core-only beside 2^25 at a 7,618 MiB peak in 19.2 s; 2^27 panics after 568 s | mines and proves on headless Linux with 0.8 GB spare, too close for a desktop with a display, where core-only (2.6 GB spare) is the profile | the app keeps 8 and 10 GB cards "prove alone, off by default while mining" (the prover-floor agent's provedefault rows from these numbers) | +| 8 GB (4060, 4060 Ti 8 GB) | the 4060 Ti proves alone at 2^26 (9.6 s, 7,740 MiB of 8,188), at 2^25 (13.5 s, 7,676) and core-only at 2^25 (5.8 s, 5,916) and 2^24 (11.2 s, 5,404); beside its 1.4 GB miner the 2^26 compressed point hangs (7.7 + 1.4 over 8.2 GB); the core-only beside rows are running | an 8 GB card proves alone, slowly, or mines; mine-and-prove on 8 GB is core-only at 2^25 or 2^24 (5.4 to 5.9 GB plus 1.4) once the hand-off ships | the public floor moves from "12 GB proves" to "8 GB proves alone" | +| 32 GB (5090) | 98.5 MH/s at 258 W; the stock server proves in 8.4 s at 18.3 GB; patched 2^26 alone 8.0 GB in 6.3 s; beside the miner 9.9 GB peak in 10.7 s (the miner costs 1.7x here against 4x on Ada) | the strongest prover per card: beside its miner it proves a v1 shard every 11 s | the fleet night's compressed prover at upstream's tier | | rig | one server per card at the card's profile: a 4090 rig needs 8 x 10.7 GB device memory beside its miners and about 6 GB of host RAM per server | fits any 8x 4090 rig with 64 GB of host RAM | phase 3 measures it | | pool user | nothing changes: the pool's provers carry the proofs | | | diff --git a/tools/fleet/box-datadir.sh b/tools/fleet/box-datadir.sh new file mode 100755 index 00000000..ee6dab95 --- /dev/null +++ b/tools/fleet/box-datadir.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# The joiner stop-gap of 6 October 2026 (docs/analysis/prover-tiers-real-cards.md, the fleet night): a node that synced +# through the headers proof holds no genesis header, so the in-memory exec state never rebuilds and eth_blockNumber +# stays 0. This swaps the box's consensus datadir for a copy of the observer's full-history datadir (pulled from the +# hub over the throwaway fleet-internal key), restarts the node with the hub and the seed as peers and the proof +# verifier, and reports the follower's replay: eth_blockNumber every 15 s until it reaches the node's DAA tip. +set -uo pipefail +F=/root/fleet; OUT=$F/out; B=/opt/igneum/pkg/bin; FLOOR=/opt/igneum-floor; HOST=$FLOOR/bin/igneum-prove-host +HUB_SSH="${HUB_SSH:-213.173.107.74}"; HUB_PORT="${HUB_PORT:-16515}"; HUB_PEER="${HUB_PEER:-213.173.107.74:16516}" +mkdir -p $OUT; exec >> $OUT/datadir.log 2>&1 +stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +echo "RESULT datadir_start $(stamp)" +t0=$(date +%s) +if [ ! -s $F/observer-datadir.tgz ]; then + scp -i $F/in/fleet-internal -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -P "$HUB_PORT" "root@$HUB_SSH:/root/fleet/share/observer-datadir.tgz" $F/observer-datadir.tgz || { echo "RESULT datadir_failed pull"; exit 2; } +fi +echo "RESULT datadir_pulled $(stamp) bytes=$(stat -c %s $F/observer-datadir.tgz) s=$(( $(date +%s) - t0 ))" +bash $F/in/box-kill.sh >/dev/null 2>&1 # every stage process; never the node (next line) +pkill -x igneumd; sleep 4; pkill -9 -x igneumd 2>/dev/null; sleep 1 +rm -rf $F/node.proof && mv $F/node $F/node.proof && mkdir -p $F/node && tar -C $F/node -xzf $F/observer-datadir.tgz || { echo "RESULT datadir_failed untar"; exit 2; } +echo "RESULT datadir_swapped $(stamp) $(du -sh $F/node | cut -f1)" +IGNEUM_PROOF_VERIFIER=$HOST nohup $B/igneumd --devnet --appdir=$F/node --rpclisten=127.0.0.1:26610 --evm-rpclisten=127.0.0.1:26790 --listen=0.0.0.0:26611 \ + --addpeer=$HUB_PEER --addpeer=188.245.5.161:26611 --override-params-file=$F/override.json --nodnsseed --disable-upnp --nologfiles --yes >> $F/node.log 2>&1 & +t1=$(date +%s) +bn() { curl -s -m 8 -X POST -H 'Content-Type: application/json' --data '{"jsonrpc":"2.0","id":1,"method":"eth_blockNumber","params":[]}' http://127.0.0.1:26790/ | grep -o '"result":"[^"]*"' | cut -d'"' -f4; } +for i in $(seq 1 240); do + sleep 15 + h="$(bn)"; w="$($B/igneum-miner watch 1 grpc://127.0.0.1:26610 2>/dev/null | grep -o 'blocks=[0-9]*.*synced=[a-z]*' | tail -1)" + n=$(( ${h:-0x0} )) + echo "RESULT replay $(stamp) s=$(( $(date +%s) - t1 )) evm_block=$n $w" + if [ "$n" -gt 0 ] && [[ "$w" == *synced=true* ]]; then + st="$(curl -s -m 8 -X POST -H 'Content-Type: application/json' --data '{"jsonrpc":"2.0","id":1,"method":"igneum_getProvingStatus","params":[]}' http://127.0.0.1:26790/ | python3 -c 'import sys,json; d=json.load(sys.stdin).get("result",{}); print("tipDaa", int(d.get("tipDaa","0x0"),16), "active", d.get("v1",{}).get("active"), "fresh", d.get("v1",{}).get("freshRuleActive"))' 2>/dev/null)" + daa=$(printf '%s' "$w" | grep -o 'daa=[0-9]*' | cut -d= -f2) + tip=$(printf '%s' "$st" | awk '{print $2}') + if [ -n "$tip" ] && [ "$tip" -ge $(( ${daa:-0} - 20 )) ]; then echo "RESULT datadir_done $(stamp) replay_s=$(( $(date +%s) - t1 )) total_s=$(( $(date +%s) - t0 )) $st"; exit 0; fi + fi +done +echo "RESULT datadir_failed replay did not reach the tip in 60 min" diff --git a/tools/fleet/box-floor-v5.sh b/tools/fleet/box-floor-v5.sh new file mode 100755 index 00000000..c451a5db --- /dev/null +++ b/tools/fleet/box-floor-v5.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# The prover-floor agent's known-failed case (6 October 2026, 13:05Z): rebuild sp1-gpu-server from patch v5 (the panic +# hook that turns a failed device allocation into "FLOOR abort ..." and exit 70) and re-run the 2^27 compressed point +# alone on a card it cannot fit; report the host's failing line, the server's FLOOR abort line and the seconds. +set -uo pipefail +F=/root/fleet; OUT=$F/out; FLOOR=/opt/igneum-floor; SRC=$FLOOR/sp1; HOST=$FLOOR/bin/igneum-prove-host +export PATH="$HOME/.cargo/bin:$FLOOR/go/bin:$PATH" RUSTUP_TOOLCHAIN=stable GOPATH=$FLOOR/gopath GOCACHE=$FLOOR/gocache GOFLAGS=-mod=mod CUDA_ARCHS="${ARCHS:-86}" CARGO_TARGET_DIR=$FLOOR/target +exec >> $OUT/floor-v5.log 2>&1 +stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +echo "RESULT v5_start $(stamp) patch_sha256=$(sha256sum $F/in/floor-v5.patch | cut -c1-16)" +bash $F/in/box-kill.sh >/dev/null 2>&1 +cd "$SRC" && git checkout -q -- . && git clean -qfd sp1-gpu/crates >/dev/null 2>&1 +git apply $F/in/floor-v5.patch || { echo "RESULT v5_failed patch does not apply"; exit 2; } +touch sp1-gpu/crates/prover_components/src/builder.rs sp1-gpu/crates/jagged_tracegen/src/lib.rs sp1-gpu/crates/server/src/server.rs sp1-gpu/crates/cuda/src/task.rs sp1-gpu/crates/server/src/main.rs +echo "RESULT v5_patched $(git diff --stat | tail -1)" +t0=$(date +%s) +cargo build --release --bin sp1-gpu-server -j $(( $(nproc) > 16 ? 16 : $(nproc) )) > $FLOOR/logs/build-server-v5.log 2>&1; rc=$? +echo "RESULT v5_build_exit $rc time_s=$(( $(date +%s) - t0 ))" +[ $rc -eq 0 ] || { grep -n -A6 '^error' $FLOOR/logs/build-server-v5.log | head -30; echo "RESULT v5_failed build"; exit 2; } +mkdir -p $FLOOR/home-v5/.sp1/bin && cp $FLOOR/target/release/sp1-gpu-server $FLOOR/home-v5/.sp1/bin/ && chmod +x $FLOOR/home-v5/.sp1/bin/sp1-gpu-server +echo "RESULT v5_server sha256=$(sha256sum $FLOOR/home-v5/.sp1/bin/sp1-gpu-server | cut -c1-16) bytes=$(stat -c %s $FLOOR/home-v5/.sp1/bin/sp1-gpu-server)" +pkill -9 -x sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock; sleep 2 +t1=$(date +%s) +env HOME=$FLOOR/home-v5 SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 SP1_GPU_ELEMENT_THRESHOLD=134217728 timeout 900 $HOST $FLOOR/prove/proving/fixtures/fees-v1-shards2.json --mode compressed --shard 0 --out $OUT/v5-point.json > $OUT/v5-point.log 2>&1; rc=$? +echo "RESULT v5_point rc=$rc wall_s=$(( $(date +%s) - t1 ))" +grep -n -E 'FLOOR abort|panicked|Error|error|RESULT' $OUT/v5-point.log | head -12 | cut -c1-300 +pkill -9 -x sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock +echo "RESULT v5_done $(stamp)" diff --git a/tools/fleet/floor-v5.patch b/tools/fleet/floor-v5.patch new file mode 100644 index 00000000..8b28f2aa --- /dev/null +++ b/tools/fleet/floor-v5.patch @@ -0,0 +1,425 @@ +diff --git a/sp1-gpu/crates/cuda/src/task.rs b/sp1-gpu/crates/cuda/src/task.rs +index a503a86..813016b 100644 +--- a/sp1-gpu/crates/cuda/src/task.rs ++++ b/sp1-gpu/crates/cuda/src/task.rs +@@ -149,7 +149,15 @@ pub enum GlobalTaskPoolBuildError { + + impl TaskPoolBuilder { + pub fn new() -> Self { +- Self { capacity: None, device: CudaDevice(0), mem_release_threshold: u64::MAX } ++ // Igneum prover-floor patch: upstream keeps every freed device allocation in the pool for the process's ++ // life (threshold u64::MAX), so the prover holds its high-water mark between shards on a card it shares ++ // with a miner. `SP1_GPU_MEM_RELEASE_THRESHOLD=` sets the pool's release threshold (0 returns ++ // freed memory to the driver at once); unset, upstream's behaviour. ++ let mem_release_threshold = std::env::var("SP1_GPU_MEM_RELEASE_THRESHOLD") ++ .ok() ++ .and_then(|s| s.parse::().ok()) ++ .unwrap_or(u64::MAX); ++ Self { capacity: None, device: CudaDevice(0), mem_release_threshold } + } + + pub fn num_tasks(mut self, num_tasks: usize) -> Self { +diff --git a/sp1-gpu/crates/jagged_tracegen/src/lib.rs b/sp1-gpu/crates/jagged_tracegen/src/lib.rs +index 579f70a..2264044 100644 +--- a/sp1-gpu/crates/jagged_tracegen/src/lib.rs ++++ b/sp1-gpu/crates/jagged_tracegen/src/lib.rs +@@ -481,6 +481,33 @@ async fn device_preprocessed_tracegen>( + named_traces + } + ++/// Igneum prover-floor patch: the dense elements a set of traces will occupy once `generate_jagged_traces` ++/// has laid them out, that is the sum of their buffers padded to the next multiple of 2^log_stacking_height ++/// (the "final padding" step below). Each phase (preprocessed, then main) is padded on its own. ++pub fn padded_trace_elements( ++ traces: &BTreeMap>, ++ log_stacking_height: u32, ++) -> usize { ++ let total: usize = traces ++ .values() ++ .map(|t| match t { ++ Trace::Real(trace) => trace.guts().as_buffer().len(), ++ Trace::Padding(_) => 0, ++ }) ++ .sum(); ++ total.next_multiple_of(1 << log_stacking_height) ++} ++ ++/// Igneum prover-floor patch: the capacity to allocate for a trace set: the exact padded size plus one ++/// stacking height of slack, never more than the prover's `max_trace_size`. `SP1_GPU_FLOOR_EXACT=0` restores ++/// upstream's full-capacity allocation. ++fn floor_capacity(max_trace_size: usize, needed: usize, log_stacking_height: u32) -> usize { ++ if std::env::var("SP1_GPU_FLOOR_EXACT").map(|v| v == "0").unwrap_or(false) { ++ return max_trace_size; ++ } ++ max_trace_size.min(needed + (1 << log_stacking_height)) ++} ++ + async fn allocate_and_initialize_traces( + preprocessed_traces: BTreeMap>, + max_trace_size: usize, +@@ -494,6 +521,11 @@ async fn allocate_and_initialize_traces( + + let total_gb = total_bytes as f64 / (1 << 30) as f64; + tracing::debug!("Allocating {:?} GB of traces", total_gb); ++ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() { ++ eprintln!( ++ "FLOOR tracegen alloc capacity_elements={max_trace_size} bytes={total_bytes} ({total_gb:.3} GB)" ++ ); ++ } + let mut dense_data: Buffer = + Buffer::with_capacity_in(max_trace_size, backend.clone()); + let mut col_index: Buffer = +@@ -677,9 +709,14 @@ pub async fn setup_tracegen>( + let preprocessed_traces = + device_preprocessed_tracegen(program, host_phase_tracegen, backend).await; + ++ let capacity = floor_capacity( ++ max_trace_size, ++ padded_trace_elements(&preprocessed_traces, log_stacking_height), ++ log_stacking_height, ++ ); + let jagged_traces = allocate_and_initialize_traces( + preprocessed_traces, +- max_trace_size, ++ capacity, + log_stacking_height, + max_log_row_count, + backend, +@@ -906,6 +943,11 @@ pub async fn main_tracegen, A: CudaTracegenAir>( + + log_chip_stats(machine, &chip_set, &traces); + ++ // Igneum prover-floor patch: the key's buffer is sized to its preprocessed traces at setup (upstream sized it ++ // for a whole shard), so grow it here to what this shard needs before the main traces are appended: a bigger ++ // dense buffer and column index, the preprocessed region copied device to device, swapped into the key. ++ grow_for_main(&mut jagged_traces.preprocessed_traces, &traces, log_stacking_height, backend); ++ + copy_main_jagged_traces( + traces, + &mut jagged_traces.preprocessed_traces, +@@ -918,6 +960,61 @@ pub async fn main_tracegen, A: CudaTracegenAir>( + (public_values, chip_set, permit) + } + ++/// Igneum prover-floor patch: see `main_tracegen`. The need is the preprocessed phase as laid out (its padded ++/// end, `preprocessed_offset`) plus the main traces padded to the stacking height plus one stacking height of ++/// slack; a buffer at least that big is left alone. The process aborts, loudly, if the copy cannot be made, ++/// because a panic inside a prover task is what left sweep 2 hanging on the client's socket. ++fn grow_for_main( ++ jagged: &mut JaggedTraceMle, ++ main_traces: &BTreeMap>, ++ log_stacking_height: u32, ++ backend: &TaskScope, ++) { ++ let pre_end = jagged.dense().preprocessed_offset; ++ let needed = pre_end ++ + padded_trace_elements(main_traces, log_stacking_height) ++ + (1 << log_stacking_height); ++ let have = jagged.dense().dense.capacity(); ++ if have >= needed { ++ return; ++ } ++ let mut new_dense: Buffer = Buffer::with_capacity_in(needed, backend.clone()); ++ let mut new_col_index: Buffer = ++ Buffer::with_capacity_in(needed >> 1, backend.clone()); ++ unsafe { ++ new_dense.assume_init(); ++ new_col_index.assume_init(); ++ } ++ { ++ let JaggedMle { dense_data, col_index, .. } = &mut **jagged; ++ let src_dense: &Slice<_, _> = &dense_data.dense[..pre_end]; ++ let dst_dense: &mut Slice<_, _> = &mut new_dense[..pre_end]; ++ let src_col: &Slice<_, _> = &col_index[..pre_end >> 1]; ++ let dst_col: &mut Slice<_, _> = &mut new_col_index[..pre_end >> 1]; ++ unsafe { ++ if dst_dense.copy_from_slice(src_dense, backend).is_err() ++ || dst_col.copy_from_slice(src_col, backend).is_err() ++ { ++ eprintln!("FLOOR grow FAILED: could not copy the preprocessed region ({pre_end} elements) into the grown buffer ({needed} elements); aborting instead of hanging"); ++ std::process::abort(); ++ } ++ } ++ } ++ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() { ++ eprintln!( ++ "FLOOR grow key buffer {have} -> {needed} elements (preprocessed {pre_end}, {} bytes)", ++ needed * 6 ++ ); ++ } ++ let JaggedMle { dense_data, col_index, .. } = &mut **jagged; ++ dense_data.dense = new_dense; ++ *col_index = new_col_index; ++ unsafe { ++ dense_data.dense.set_len(pre_end); ++ col_index.set_len(pre_end >> 1); ++ } ++} ++ + #[allow(clippy::too_many_arguments)] + pub async fn main_tracegen_permit, A: CudaTracegenAir>( + machine: &Machine, +@@ -984,9 +1081,15 @@ pub async fn full_tracegen>( + + log_chip_stats(machine, &chip_set, &main_traces); + ++ let capacity = floor_capacity( ++ max_trace_size, ++ padded_trace_elements(&preprocessed_traces, log_stacking_height) ++ + padded_trace_elements(&main_traces, log_stacking_height), ++ log_stacking_height, ++ ); + let mut jagged_mle = allocate_and_initialize_traces( + preprocessed_traces, +- max_trace_size, ++ capacity, + log_stacking_height, + max_log_row_count, + backend, +@@ -1002,6 +1105,18 @@ pub async fn full_tracegen>( + ) + .await; + ++ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() { ++ let dense = jagged_mle.dense(); ++ let (free, total) = sp1_gpu_cudart::cuda_memory_info().unwrap_or((0, 0)); ++ eprintln!( ++ "FLOOR tracegen used preprocessed_elements={} main_elements={} dense_len={} capacity_elements={capacity} max_trace_size={max_trace_size} device_used_mib={}", ++ dense.preprocessed_offset, ++ dense.main_size(), ++ dense.dense.len(), ++ (total - free) >> 20 ++ ); ++ } ++ + (public_values, jagged_mle, chip_set, permit) + } + +diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/prover_components/src/builder.rs +index 5dccd9d..574d4fa 100644 +--- a/sp1-gpu/crates/prover_components/src/builder.rs ++++ b/sp1-gpu/crates/prover_components/src/builder.rs +@@ -23,28 +23,75 @@ use crate::{ + SP1CudaProverComponents, + }; + ++/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger ++/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a ++/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. ++/// The proof format, the verifier and the program ids do not change: the element threshold only decides where ++/// the executor splits shards, as upstream's own 24 GB tier already does. ++fn env_usize(name: &str) -> Option { ++ std::env::var(name).ok().and_then(|s| s.parse::().ok()) ++} ++ ++fn env_f64(name: &str) -> Option { ++ std::env::var(name).ok().and_then(|s| s.parse::().ok()) ++} ++ ++/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it ++/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16). ++pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 { ++ if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) { ++ ELEMENT_THRESHOLD ++ } else if gpu_memory_gb >= 24 { ++ ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24) ++ } else if gpu_memory_gb >= 18 { ++ (1 << 27) + (1 << 26) ++ } else { ++ 1 << 27 ++ } ++} ++ ++/// The recursion trace allocation (elements) for a memory budget. ++pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize { ++ if gpu_memory_gb >= 24 { ++ RECURSION_TRACE_ALLOCATION ++ } else { ++ RECURSION_TRACE_ALLOCATION ++ } ++} ++ ++pub fn gpu_memory_gb() -> usize { ++ let gb = 1024.0 * 1024.0 * 1024.0; ++ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { ++ Some(b) => (b.ceil() as usize) + 4, ++ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4, ++ } ++} ++ ++pub fn recursion_trace_allocation() -> usize { ++ env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION") ++ .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb())) ++} ++ + pub fn local_gpu_opts() -> SP1CoreOpts { + let mut opts = SP1CoreOpts::default(); + + let log2_shard_size = 24; + opts.shard_size = 1 << log2_shard_size; + +- let gb = 1024.0 * 1024.0 * 1024.0; +- +- // Get the amount of memory on the GPU. +- let gpu_memory_gb: usize = (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4; +- +- if gpu_memory_gb < 24 { +- panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"); +- } ++ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given. ++ let gpu_memory_gb = gpu_memory_gb(); + +- let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 { +- ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24) +- } else { +- ELEMENT_THRESHOLD ++ let shard_threshold = match env_usize("SP1_GPU_ELEMENT_THRESHOLD") { ++ Some(t) => t as u64, ++ None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards), + }; ++ let height_threshold = opts.sharding_threshold.height_threshold; + +- tracing::debug!("Shard threshold: {shard_threshold}"); ++ eprintln!( ++ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", ++ recursion_trace_allocation(), ++ opts.full_size_shards ++ ); + opts.sharding_threshold.element_threshold = shard_threshold; + + opts.global_dependencies_opt = true; +@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier( + ) { + let recursion_verifier = SP1CudaProverComponents::compress_verifier(); + ( +- new_cuda_prover(&recursion_verifier, RECURSION_TRACE_ALLOCATION, 4, false, false, scope) ++ new_cuda_prover(&recursion_verifier, recursion_trace_allocation(), 4, false, false, scope) + .await, + recursion_verifier, + ) +diff --git a/sp1-gpu/crates/server/src/main.rs b/sp1-gpu/crates/server/src/main.rs +index 65e94f5..498357c 100644 +--- a/sp1-gpu/crates/server/src/main.rs ++++ b/sp1-gpu/crates/server/src/main.rs +@@ -17,9 +17,55 @@ struct Args { + version: bool, + } + ++/// Igneum prover-floor patch (6 October 2026, the GPU fleet's finding): a panic inside a prover task (an allocation ++/// the card cannot meet, `cudaMallocAsync` failing and `Buffer::with_capacity_in` panicking in a tokio worker) left ++/// the request's future waiting for ever, the client on its socket, the card at 0% for the 15 minutes until someone ++/// killed it (the 8 GB and 10 GB cards at threshold 2^27). The server must fail the shard instead: this hook names ++/// the stage from the panic's location and exits, so the client's proof fails at once with the server's last line. ++fn stage_of(location: &str) -> &'static str { ++ let l = location.to_ascii_lowercase(); ++ if l.contains("jagged_tracegen") || l.contains("/tracegen") { ++ "trace generation" ++ } else if l.contains("commit") || l.contains("basefold") || l.contains("merkle") { ++ "the commit (codewords and Merkle trees)" ++ } else if l.contains("logup_gkr") { ++ "LogUp GKR" ++ } else if l.contains("zerocheck") { ++ "the zerocheck" ++ } else if l.contains("jagged") { ++ "the jagged sumcheck" ++ } else if l.contains("prover_components") || l.contains("recursion") || l.contains("sp1-prover") || l.contains("sp1_prover") { ++ "the recursion (compression)" ++ } else if l.contains("cuda") || l.contains("slop") { ++ "a device allocation" ++ } else { ++ "the prover" ++ } ++} ++ ++fn install_abort_on_panic() { ++ std::panic::set_hook(Box::new(|info| { ++ let location = info.location().map(|l| format!("{}:{}", l.file(), l.line())).unwrap_or_else(|| "unknown".into()); ++ let message = info ++ .payload() ++ .downcast_ref::<&str>() ++ .map(|s| s.to_string()) ++ .or_else(|| info.payload().downcast_ref::().cloned()) ++ .unwrap_or_default(); ++ let oom = message.to_ascii_lowercase().contains("alloc") || message.contains("MemoryAllocation") || message.contains("OUT_OF_MEMORY"); ++ eprintln!( ++ "FLOOR abort: {} failed at {location}: {message}{}; the server exits so the client's proof fails instead of waiting", ++ stage_of(&location), ++ if oom { " (the card's memory could not meet an allocation: lower SP1_GPU_ELEMENT_THRESHOLD one notch)" } else { "" } ++ ); ++ std::process::exit(70); ++ })); ++} ++ + #[tokio::main] + #[allow(clippy::print_stdout)] + async fn main() { ++ install_abort_on_panic(); + tracing_subscriber::fmt::init(); + + let args = Args::parse(); +@@ -40,3 +86,19 @@ async fn main() { + eprintln!("Error running server: {e}"); + } + } ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::stage_of; ++ ++ #[test] ++ fn the_stage_is_named_from_the_panic_location() { ++ assert_eq!(stage_of("sp1-gpu/crates/jagged_tracegen/src/lib.rs:240"), "trace generation"); ++ assert_eq!(stage_of("sp1-gpu/crates/basefold/src/fri.rs:97"), "the commit (codewords and Merkle trees)"); ++ assert_eq!(stage_of("sp1-gpu/crates/logup_gkr/src/tracegen.rs:72"), "LogUp GKR"); ++ assert_eq!(stage_of("sp1-gpu/crates/zerocheck/src/prover.rs:1163"), "the zerocheck"); ++ assert_eq!(stage_of("sp1-gpu/crates/prover_components/src/builder.rs:70"), "the recursion (compression)"); ++ assert_eq!(stage_of("sp1-gpu/crates/cuda/src/stream.rs:330"), "a device allocation"); ++ assert_eq!(stage_of("somewhere/else.rs:1"), "the prover"); ++ } ++} +diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs +index 4035f1f..0d0d907 100644 +--- a/sp1-gpu/crates/server/src/server.rs ++++ b/sp1-gpu/crates/server/src/server.rs +@@ -157,6 +157,7 @@ impl Server { + }; + let pk = CachedProgram { elf: Arc::new(Elf::Dynamic(elf.into())), vk: vk.clone() }; + ctx.pk_cache.insert(elf_hash, pk); ++ floor_memory_line("after setup"); + Response::Setup { id: elf_hash, vk } + } + Request::Destroy { key } => { +@@ -177,15 +178,31 @@ impl Server { + ); + }; + let context = SP1Context::builder().proof_nonce(proof_nonce).build(); +- match prover.prove_with_mode(&cached.elf, stdin, context, mode).await { ++ let started = std::time::Instant::now(); ++ let response = match prover.prove_with_mode(&cached.elf, stdin, context, mode).await { + Ok(proof) => Response::Proof { proof }, + Err(e) => Response::ProverError(e.to_string()), +- } ++ }; ++ floor_memory_line(&format!("after prove {:?} in {:.1} s", mode, started.elapsed().as_secs_f64())); ++ response + } + } + } + } + ++/// Igneum prover-floor patch: the device memory in use (total minus free, as the driver reports it) at the ++/// points that bound a proof, so a run's log carries the terms of the peak without a sampler. ++fn floor_memory_line(what: &str) { ++ if let Ok((free, total)) = sp1_gpu_cudart::cuda_memory_info() { ++ eprintln!( ++ "FLOOR memory {what}: device_used_mib={} free_mib={} total_mib={}", ++ (total - free) >> 20, ++ free >> 20, ++ total >> 20 ++ ); ++ } ++} ++ + fn sha256(data: &[u8]) -> [u8; 32] { + use sha2::{Digest, Sha256}; + let mut hasher = Sha256::new();