diff --git a/docs/bench-log.md b/docs/bench-log.md index 304a42264..968212a44 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -2466,8 +2466,16 @@ host's "Error: CudaClientError: Failed to read the response: UnexpectedEof", exi (`floor-pc1-hangcase-2`, upstream's threshold 402,653,184 on the prototype shard on the 4070, 13:10:29Z) hit the job's 5-minute cap with no result line: on that card and point the failure did not reach the hook (a C++ CUDA exception in the sppark NTT code cannot unwind into Rust; or a stream synchronisation that never returns after a -failed launch), or the point was still running; the point's log (the restore job) says which. For a user the app's -budget covers it either way: 120 s, then the server killed and the threshold stepped down. The 4060 Ti 8 GB rows +failed launch), or the point was still running; the point's log (job `floor-pc1-restore`, 13:17:50Z, 16 s: no leftover held +the card, the prover switched on) said which: NOT a dead task. The server allocated the full 2.26 GB trace buffer at +that threshold and filled it (dense 404,750,336 elements, the card at 12,281 MiB in use) and the sampler read 11,937 +MiB at 100% utilisation on all 260 samples: the proof was running inside a card full by a hair, slowly (11 s on the +5090; 301 s was not enough on the 4070 at the limit), finishing or not unknown. The two classes are each covered by +the half that owns them: an allocation the card cannot meet, the hook (seconds); a point that fits by a hair and +crawls, the app's budget (120 s for a v1 shard, then the server killed and the threshold stepped down; its runtime +demonstration needs the 0.3.13 app, an item of the cut's verification run). Rule from it: a playbook that switches +the prover off runs its point under `timeout` inside the script (`POINT_BUDGET_S`, 180 s), under the job's cap, so +the restore tail always runs. The 4060 Ti 8 GB rows from the fleet: alone 2^26 9.6 s at 7,740 MiB, 2^25 compressed 13.5 s at 7,676, core 2^25 5.8 s at 5,916, core 2^26 5.0 s at 7,196, core 2^24 11.2 s at 5,404; 2^27 hung 904 s on v4; beside its miner the compressed 2^26 point hung too (7.7 GB plus the 1.4 GB miner on 8.2 GB), so an 8 GB card proves ALONE (the tier line) and core-only beside the diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index 35a110578..841ec2d38 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -153,6 +153,17 @@ line) at 9,886 MiB on the 5090 in 5.9 s, VERIFIED; the restore script removed it the copy. The signing path on the Mac: `sign-server` and `verify-server embedded` agree, a tampered manifest and a wrong binary are refused. +The hang on a point that does not fit, and the rule it leaves (6 October 2026, 13:00 to 13:20Z; bench-log "prover +floor"): patch v5's panic hook fails an allocation the card cannot meet in seconds with the stage named (the fleet's +3080: 13 s against 568 s of hang), and the app's per-shard budget (`provedefault::shard_budget`, 120 s for a v1 shard) +kills a point that fits by a hair and crawls, stepping the threshold down (`step_down`) for the next shard; PC 1's +second hang case (upstream's threshold on the prototype shard on the 4070) was the second class: 11,937 of 12,282 +MiB at 100% for the job's whole 5-minute cap, not a dead task. Hygiene rule from it (the coordinator, 13:16Z): a job +whose cap can fire while the prover is off runs its point under a budget INSIDE the script, shorter than the cap +(`timeout`, `POINT_BUDGET_S`, 180 s in every prover-floor playbook), so the restore tail (the prover on, the sockets +unlinked) always runs and no PC is left without its prover by a timed-out job; the 0.3.13 verification run exercises +the app's own budget and step-down on PC 2 with the cut's app before the publish. + What the next cut's shipper must do (0.3.13), in order: (1) run the `prover-server` workflow on master (or merge prover-floor and let the push trigger it; about 40 minutes cold on the hosted runner, approximate); (2) on the Mac, `packaging/prover/push-server.sh --run ` (the OTA key, the dl token and the Vercel login as for diff --git a/tools/prover-floor/pc-12gb-card-miner.ps1 b/tools/prover-floor/pc-12gb-card-miner.ps1 index 8db230fa1..345d95383 100644 --- a/tools/prover-floor/pc-12gb-card-miner.ps1 +++ b/tools/prover-floor/pc-12gb-card-miner.ps1 @@ -42,6 +42,8 @@ set -uo pipefail export PATH="$HOME/.cargo/bin:$PATH" CUDA_DEVICE_ORDER=PCI_BUS_ID CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}" stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +# a point that runs past POINT_BUDGET_S (default 180 s) is killed by `timeout` INSIDE the script, so the tail (the +# prover back on, the sockets unlinked) always runs; the job's cap stays the outer guard (6 October 2026, hang case 2) JOB='JOBW_PLACEHOLDER'; KIT='KITW_PLACEHOLDER'; CARD=CARD_PLACEHOLDER; PHASE='PHASE_PLACEHOLDER' FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server; H=/opt/igneum-floor/host/igneum-prove-host echo "RESULT wsl_cards $(nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader,nounits 2>/dev/null | tr '\n' ';')" @@ -69,7 +71,7 @@ runshard() { # name fixture env... nvidia-smi -i $CARD --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc-12gb-card.ps1 b/tools/prover-floor/pc-12gb-card.ps1 index 812c3c18c..847e453bd 100644 --- a/tools/prover-floor/pc-12gb-card.ps1 +++ b/tools/prover-floor/pc-12gb-card.ps1 @@ -42,6 +42,8 @@ set -uo pipefail export PATH="$HOME/.cargo/bin:$PATH" CUDA_DEVICE_ORDER=PCI_BUS_ID CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}" stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +# a point that runs past POINT_BUDGET_S (default 180 s) is killed by `timeout` INSIDE the script, so the tail (the +# prover back on, the sockets unlinked) always runs; the job's cap stays the outer guard (6 October 2026, hang case 2) JOB='JOBW_PLACEHOLDER'; KIT='KITW_PLACEHOLDER'; CARD=CARD_PLACEHOLDER; PHASE='PHASE_PLACEHOLDER' FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server; H=/opt/igneum-floor/host/igneum-prove-host echo "RESULT wsl_cards $(nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader,nounits 2>/dev/null | tr '\n' ';')" @@ -69,7 +71,7 @@ runshard() { # name fixture env... nvidia-smi -i $CARD --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc1-hangcase.ps1 b/tools/prover-floor/pc1-hangcase.ps1 index f67bfd01d..dc8f85a68 100644 --- a/tools/prover-floor/pc1-hangcase.ps1 +++ b/tools/prover-floor/pc1-hangcase.ps1 @@ -42,6 +42,8 @@ set -uo pipefail export PATH="$HOME/.cargo/bin:$PATH" CUDA_DEVICE_ORDER=PCI_BUS_ID CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}" stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +# a point that runs past POINT_BUDGET_S (default 180 s) is killed by `timeout` INSIDE the script, so the tail (the +# prover back on, the sockets unlinked) always runs; the job's cap stays the outer guard (6 October 2026, hang case 2) JOB='JOBW_PLACEHOLDER'; KIT='KITW_PLACEHOLDER'; CARD=CARD_PLACEHOLDER; PHASE='PHASE_PLACEHOLDER' FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server; H=/opt/igneum-floor/host/igneum-prove-host echo "RESULT wsl_cards $(nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader,nounits 2>/dev/null | tr '\n' ';')" @@ -69,7 +71,7 @@ runshard() { # name fixture env... nvidia-smi -i $CARD --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc2-floor-core-alone.ps1 b/tools/prover-floor/pc2-floor-core-alone.ps1 index 10bce978e..e6ca716ee 100644 --- a/tools/prover-floor/pc2-floor-core-alone.ps1 +++ b/tools/prover-floor/pc2-floor-core-alone.ps1 @@ -46,7 +46,7 @@ run() { # name fixture env... nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock @@ -66,7 +66,7 @@ runshard() { # name fixture env... (--mode shard: core then compressed; the co nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc2-floor-core-miner.ps1 b/tools/prover-floor/pc2-floor-core-miner.ps1 index 06b92aa79..c92dc1d8f 100644 --- a/tools/prover-floor/pc2-floor-core-miner.ps1 +++ b/tools/prover-floor/pc2-floor-core-miner.ps1 @@ -46,7 +46,7 @@ run() { # name fixture env... nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock @@ -66,7 +66,7 @@ runshard() { # name fixture env... (--mode shard: core then compressed; the co nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc2-floor-core2-miner.ps1 b/tools/prover-floor/pc2-floor-core2-miner.ps1 index 69f9221b0..45376b565 100644 --- a/tools/prover-floor/pc2-floor-core2-miner.ps1 +++ b/tools/prover-floor/pc2-floor-core2-miner.ps1 @@ -46,7 +46,7 @@ run() { # name fixture env... nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock @@ -66,7 +66,7 @@ runshard() { # name fixture env... (--mode shard: core then compressed; the co nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc2-floor-measure.ps1 b/tools/prover-floor/pc2-floor-measure.ps1 index e7695a5e9..73aaba015 100644 --- a/tools/prover-floor/pc2-floor-measure.ps1 +++ b/tools/prover-floor/pc2-floor-measure.ps1 @@ -28,6 +28,8 @@ set -uo pipefail export PATH="$HOME/.cargo/bin:$PATH" CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}" stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +# a point that runs past POINT_BUDGET_S (default 180 s) is killed by `timeout` INSIDE the script, so the tail (the +# prover back on, the sockets unlinked) always runs; the job's cap stays the outer guard (6 October 2026, hang case 2) JOB='JOBW_PLACEHOLDER'; H=/opt/igneum-pv1/igneum-prove-host; FX="/root/igneum-prove-pv1/proving/fixtures" FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server EMPTY='EMPTY_PLACEHOLDER'; V1="$FX/fees-v1-shards2.json"; FULL="$FX/block-338-shard1.json"; ONE="$FX/block-56-transfers.json" @@ -45,7 +47,7 @@ run() { # name fixture env... nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock @@ -65,7 +67,7 @@ runshard() { # name fixture env... (--mode shard: core then compressed; the co nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc2-floor-sweep1.ps1 b/tools/prover-floor/pc2-floor-sweep1.ps1 index 7a38e1cb9..9918903c7 100644 --- a/tools/prover-floor/pc2-floor-sweep1.ps1 +++ b/tools/prover-floor/pc2-floor-sweep1.ps1 @@ -44,7 +44,7 @@ run() { # name fixture env... nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc2-floor-sweep2.ps1 b/tools/prover-floor/pc2-floor-sweep2.ps1 index 5f9058f70..6988fe749 100644 --- a/tools/prover-floor/pc2-floor-sweep2.ps1 +++ b/tools/prover-floor/pc2-floor-sweep2.ps1 @@ -44,7 +44,7 @@ run() { # name fixture env... nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc2-floor-sweep3.ps1 b/tools/prover-floor/pc2-floor-sweep3.ps1 index 485e0f5c7..7a1712c0e 100644 --- a/tools/prover-floor/pc2-floor-sweep3.ps1 +++ b/tools/prover-floor/pc2-floor-sweep3.ps1 @@ -42,7 +42,7 @@ run() { # name fixture env... nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock diff --git a/tools/prover-floor/pc2-floor-sweep4-miner.ps1 b/tools/prover-floor/pc2-floor-sweep4-miner.ps1 index e8cc60f4c..171c796e5 100644 --- a/tools/prover-floor/pc2-floor-sweep4-miner.ps1 +++ b/tools/prover-floor/pc2-floor-sweep4-miner.ps1 @@ -46,7 +46,7 @@ run() { # name fixture env... nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null & local SMI=$! local t0=$(date +%s) - env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 + timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1 local rc=$? local wall=$(( $(date +%s) - t0 )) pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock