Compare commits
30 commits
master
...
prover-flo
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b5ca42cabf | ||
|
|
500ab8c2fe | ||
|
|
fd000e34b7 | ||
|
|
18055c1196 | ||
|
|
74c5310ff9 | ||
|
|
1df93c707a | ||
|
|
290de6c28b | ||
|
|
2bbc0acef1 | ||
|
|
22050df641 | ||
|
|
ac7ce22f24 | ||
|
|
83e4699aef | ||
|
|
bb0453ba49 | ||
|
|
97f8673e0f | ||
|
|
6b8d00c78e | ||
|
|
5cdcf9be61 | ||
|
|
460780d513 | ||
|
|
8e64da2213 | ||
|
|
e0803ddfc0 | ||
|
|
710e0550eb | ||
|
|
ef9940c617 | ||
|
|
87ba8c6159 | ||
|
|
d1d7d19a06 | ||
|
|
6b45ca7777 | ||
|
|
68fc4c52f0 | ||
|
|
0391de0328 | ||
|
|
5a0fa37bfa | ||
|
|
a78ee4a979 | ||
|
|
b5469f3010 | ||
|
|
1b90c4ecfd | ||
|
|
a6752b0abf |
41 changed files with 2544 additions and 39 deletions
88
.github/workflows/prover-server.yml
vendored
Normal file
88
.github/workflows/prover-server.yml
vendored
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
# The project's GPU prover server (prover floor, 6 October 2026): SP1's sp1-gpu-server 6.8.1 rebuilt from source
|
||||
# with proving/prover-floor/sp1-gpu-6.8.1-floor.patch, which sizes the server's buffers to the shard instead of
|
||||
# to a 24 GB card (bench-log "prover floor": the RTX 4070 12 GB mines and proves at threshold 2^26). One recipe,
|
||||
# packaging/prover/build-server.sh, shared with the PCs' WSL2 path (tools/prover-floor/pc2-build-server.ps1).
|
||||
#
|
||||
# What comes out: the artifact sp1-gpu-server-floor (the binary, its sha256, build.json). Nothing is signed here:
|
||||
# packaging/prover/push-server.sh on the Mac downloads the artifact by run id, writes prover-server.json (the SP1
|
||||
# version and commit, the patch's sha256, the CUDA targets, the binary's sha256 and size, this run's id), signs it
|
||||
# with the OTA key the apps trust and publishes the three files to the downloads host; the Windows build
|
||||
# (windows.yml) fetches and verifies them for the payload; the app verifies them again before use.
|
||||
#
|
||||
# Not byte-reproducible across machines (SP1's prover-types build script stamps the build time into the binary), so
|
||||
# the signed manifest names THIS build; the PC build is a behavioural cross-check (the same fixtures, verified).
|
||||
# The CUDA toolkit comes from Jimver/cuda-toolkit (nvcc, nvtx, cudart only), as SP1's own release workflow does.
|
||||
name: prover-server
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
cuda_archs:
|
||||
description: "CUDA_ARCHS (default 80,86,89,120: 3060/3090 sm_86, 4060 to 4090 sm_89, 5080/5090 sm_120, A100 sm_80)"
|
||||
default: "80,86,89,120"
|
||||
required: false
|
||||
push:
|
||||
branches: [master]
|
||||
paths:
|
||||
- 'proving/prover-floor/sp1-gpu-6.8.1-floor.patch'
|
||||
- 'packaging/prover/build-server.sh'
|
||||
- '.github/workflows/prover-server.yml'
|
||||
permissions:
|
||||
contents: read
|
||||
jobs:
|
||||
build:
|
||||
name: sp1-gpu-server 6.8.1 + floor patch (linux x86_64)
|
||||
runs-on: ubuntu-24.04
|
||||
timeout-minutes: 120
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: free disk (the CUDA toolkit and the SP1 tree need about 20 GB)
|
||||
run: |
|
||||
sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL || true
|
||||
df -h / | tail -1
|
||||
- name: CUDA 12.8.1 (nvcc, nvtx, cudart)
|
||||
uses: Jimver/cuda-toolkit@v0.2.23
|
||||
with:
|
||||
cuda: '12.8.1'
|
||||
method: 'network'
|
||||
sub-packages: '["nvcc", "nvtx", "cudart"]'
|
||||
use-github-cache: false
|
||||
use-local-cache: false
|
||||
- name: Go 1.27 (the server's native-gnark feature)
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version: '1.27.1'
|
||||
- name: protoc, cmake, clang
|
||||
run: |
|
||||
sudo apt-get update -qq
|
||||
sudo apt-get install -y -qq protobuf-compiler cmake clang pkg-config libssl-dev
|
||||
nvcc --version | tail -1; protoc --version; cmake --version | head -1; go version
|
||||
- name: Rust stable
|
||||
run: rustup toolchain install stable --profile minimal && rustup default stable
|
||||
- name: cargo cache
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
/tmp/igneum-sp1-6.8.1/target
|
||||
key: prover-server-${{ runner.os }}-${{ hashFiles('proving/prover-floor/sp1-gpu-6.8.1-floor.patch') }}-${{ github.event.inputs.cuda_archs || '80,86,89,120' }}
|
||||
restore-keys: prover-server-${{ runner.os }}-
|
||||
- name: build (packaging/prover/build-server.sh)
|
||||
env:
|
||||
CUDA_ARCHS: ${{ github.event.inputs.cuda_archs || '80,86,89,120' }}
|
||||
CARGO_JOBS: 4
|
||||
run: |
|
||||
set -euo pipefail
|
||||
bash packaging/prover/build-server.sh build/prover-server /tmp/igneum-sp1-6.8.1
|
||||
cat build/prover-server/build.json
|
||||
cat build/prover-server/sp1-gpu-server.sha256
|
||||
- name: the binary answers --version
|
||||
run: |
|
||||
v="$(build/prover-server/sp1-gpu-server --version)"
|
||||
echo "version: $v"; [ "$v" = "6.8.1" ] || { echo "::error::the server reports $v, not 6.8.1 (the SDK refuses it)"; exit 1; }
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: sp1-gpu-server-floor
|
||||
path: build/prover-server/
|
||||
if-no-files-found: error
|
||||
retention-days: 90
|
||||
16
.github/workflows/windows.yml
vendored
16
.github/workflows/windows.yml
vendored
|
|
@ -152,6 +152,19 @@ jobs:
|
|||
echo "files: $(ls "$HOME/.config/igneum" | tr '\n' ' ')"
|
||||
bash packaging/mac/packaged-config.sh --test
|
||||
|
||||
- name: prover server (sp1-gpu-server from the downloads host, its signed manifest verified; absent = stock server)
|
||||
shell: bash
|
||||
env:
|
||||
DL_TOKEN: ${{ secrets.DL_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
signer="app/igneum-app/target/release/igneum-ota-sign.exe"
|
||||
[ -x "$signer" ] || { echo "::error::$signer was not built by the engine step"; exit 1; }
|
||||
# prover floor (packaging/prover/fetch-server.sh, the same checks): a host without the manifest ships no
|
||||
# server (the app runs SP1's stock one, 24 GB cards); a manifest that does not verify fails the build
|
||||
DL_TOKEN="$DL_TOKEN" bash packaging/prover/fetch-server.sh build/prover-server --signer "$signer"
|
||||
ls -la build/prover-server || true
|
||||
|
||||
- name: payload inputs (payload-inputs.zip from the downloads host, signature, hashes and node commit verified)
|
||||
shell: bash
|
||||
env:
|
||||
|
|
@ -198,6 +211,9 @@ jobs:
|
|||
export IGNEUM_WIN_RELEASE="$PWD/build/inputs"
|
||||
export IGNEUM_WORKERS_DIR="$PWD/build/inputs"
|
||||
export IGNEUM_APP_EXE="$PWD/app/igneum-app/target/release/igneum-app.exe"
|
||||
# prover floor: the verified server folder from the step above (empty = the stock server) and the signer
|
||||
export IGNEUM_PROVER_SERVER="$PWD/build/prover-server"
|
||||
export IGNEUM_OTA_SIGN="$PWD/app/igneum-app/target/release/igneum-ota-sign.exe"
|
||||
packaging/windows/make-payload.sh "$PWD/packaging/windows/dist/igneum-windows-app.zip"
|
||||
test -f "packaging/windows/igneum-windows-app/Igneum Miner.exe" || { echo "::error::the window host did not land in the payload"; exit 1; }
|
||||
|
||||
|
|
|
|||
|
|
@ -24,6 +24,9 @@ mod manifest;
|
|||
mod jobs;
|
||||
#[path = "../inputs.rs"]
|
||||
mod inputs;
|
||||
#[path = "../proverserver.rs"]
|
||||
#[allow(dead_code)]
|
||||
mod proverserver;
|
||||
|
||||
use ed25519_dalek::{Signer, SigningKey};
|
||||
use std::path::Path;
|
||||
|
|
@ -83,6 +86,23 @@ fn main() {
|
|||
Err(e) => die(&e),
|
||||
}
|
||||
}
|
||||
// the project's GPU prover server (src/proverserver.rs): `verify-server <pub|embedded> <prover-server.json>
|
||||
// <prover-server.json.sig> [--binary <sp1-gpu-server>]` checks the signature, parses the manifest and, with
|
||||
// --binary, the binary's sha256 and size; prints the server's version, sha256 and targets
|
||||
Some("verify-server") if args.len() == 4 || args.len() == 6 => {
|
||||
let pk = if args[1] == "embedded" { manifest::OTA_PUBLIC_KEY_HEX.to_string() } else { read_key_arg(&args[1]) };
|
||||
let bytes = std::fs::read(&args[2]).unwrap_or_else(|e| die(&format!("{}: {e}", args[2])));
|
||||
let sig = std::fs::read_to_string(&args[3]).unwrap_or_else(|e| die(&format!("{}: {e}", args[3])));
|
||||
manifest::verify_signature(&bytes, sig.trim(), &pk).unwrap_or_else(|e| die(&e));
|
||||
let m = proverserver::parse(&String::from_utf8_lossy(&bytes)).unwrap_or_else(|e| die(&e));
|
||||
if args.len() == 6 {
|
||||
if args[4] != "--binary" {
|
||||
die("verify-server <pub|embedded> <manifest> <sig> [--binary <file>]");
|
||||
}
|
||||
proverserver::check_binary(&m, Path::new(&args[5])).unwrap_or_else(|e| die(&e));
|
||||
}
|
||||
println!("ok: sp1-gpu-server {} ({} bytes, sha256 {}), SP1 {} {}, patch {}, targets {}, built {} by {}", m.sha256, m.bytes, &m.sha256[..16], m.sp1_version, m.sp1_commit, &m.patch_sha256[..16.min(m.patch_sha256.len())], m.cuda_archs, m.built_at, m.built_by);
|
||||
}
|
||||
Some("embedded") if args.len() == 1 => {
|
||||
println!("{}", manifest::OTA_PUBLIC_KEY_HEX);
|
||||
println!("fingerprint sha256:{}", manifest::fingerprint(manifest::OTA_PUBLIC_KEY_HEX));
|
||||
|
|
@ -142,6 +162,18 @@ fn main() {
|
|||
Err(e) => die(&e),
|
||||
}
|
||||
}
|
||||
// the project's GPU prover server manifest (src/proverserver.rs): parsed first, so a manifest that does not
|
||||
// name the server, its sha256 and its size is never signed
|
||||
Some("sign-server") if args.len() == 3 => {
|
||||
let seed = manifest::hex_decode(&read_key_arg(&args[1])).unwrap_or_else(|| die("private key is not hex"));
|
||||
let seed: [u8; 32] = seed.try_into().unwrap_or_else(|_| die("private key is not 32 bytes"));
|
||||
let sk = SigningKey::from_bytes(&seed);
|
||||
let bytes = std::fs::read(&args[2]).unwrap_or_else(|e| die(&format!("{}: {e}", args[2])));
|
||||
let text = std::str::from_utf8(&bytes).unwrap_or_else(|_| die("prover-server manifest is not UTF-8"));
|
||||
let m = proverserver::parse(text).unwrap_or_else(|e| die(&format!("refusing to sign: {e}")));
|
||||
eprintln!("signing the prover server manifest: sp1-gpu-server {} bytes, sha256 {}, SP1 {} ({}), targets {}, built {} by {}", m.bytes, m.sha256, m.sp1_version, &m.sp1_commit[..12.min(m.sp1_commit.len())], m.cuda_archs, m.built_at, m.built_by);
|
||||
println!("{}", manifest::hex_encode(&sk.sign(&bytes).to_bytes()));
|
||||
}
|
||||
Some("sign-inputs") if args.len() == 3 => {
|
||||
let seed = manifest::hex_decode(&read_key_arg(&args[1])).unwrap_or_else(|| die("private key is not hex"));
|
||||
let seed: [u8; 32] = seed.try_into().unwrap_or_else(|_| die("private key is not 32 bytes"));
|
||||
|
|
@ -189,7 +221,7 @@ fn main() {
|
|||
println!("ok: inputs built {} from node commit {} ({}); checked: {}", m.built_at, m.node_source_commit, m.node_source_branch, checked.join(", "));
|
||||
}
|
||||
_ => {
|
||||
eprintln!("usage: igneum-ota-sign keygen <priv> <pub> | sign <priv> <manifest.json> | verify <pub> <manifest.json> <sig> | embedded | fingerprint <pub> | sha256 <file> | sign-jobs <priv> <jobs.json> | verify-jobs <pub> <jobs.json> <sig> | envelope-jobs <pub> <jobs.json> <sig> | verify-signed-jobs <pub> <jobs.signed.json> | sign-inputs <priv> <payload-inputs.json> | verify-inputs <pub|embedded> <payload-inputs.json> <sig> [--zip z] [--dir d] [--node-commit c]");
|
||||
eprintln!("usage: igneum-ota-sign keygen <priv> <pub> | sign <priv> <manifest.json> | verify <pub> <manifest.json> <sig> | embedded | fingerprint <pub> | sha256 <file> | sign-jobs <priv> <jobs.json> | verify-jobs <pub> <jobs.json> <sig> | envelope-jobs <pub> <jobs.json> <sig> | verify-signed-jobs <pub> <jobs.signed.json> | sign-inputs <priv> <payload-inputs.json> | verify-inputs <pub|embedded> <payload-inputs.json> <sig> [--zip z] [--dir d] [--node-commit c] | sign-server <priv> <prover-server.json> | verify-server <pub|embedded> <prover-server.json> <sig> [--binary f]");
|
||||
std::process::exit(2);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -108,6 +108,10 @@ pub struct Settings {
|
|||
/// on when the machine can prove, never switching an explicit on back off). Older installs apply it at their next start.
|
||||
#[serde(default)]
|
||||
pub prove_default_applied: bool,
|
||||
/// The prover profile on the patched GPU server (src/provedefault.rs `profile`): "auto" (the measured tier from
|
||||
/// the card's VRAM), "2^25", "2^26", "2^27" or "stock". Empty = auto.
|
||||
#[serde(default)]
|
||||
pub prove_profile: String,
|
||||
}
|
||||
|
||||
fn one() -> u32 {
|
||||
|
|
@ -119,7 +123,7 @@ fn yes() -> bool {
|
|||
|
||||
impl Default for Settings {
|
||||
fn default() -> Settings {
|
||||
Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, power_control: false, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false, prove_default_applied: false }
|
||||
Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, power_control: false, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false, prove_default_applied: false, prove_profile: String::new() }
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -283,7 +283,11 @@ impl Shared {
|
|||
}
|
||||
let cards = self.state.lock().unwrap().mining.cards.clone();
|
||||
let wsl = if cfg!(windows) { Some(crate::wslhost::distro_answers()) } else { None };
|
||||
let d = crate::provedefault::decide(&cards, std::env::consts::OS, wsl, crate::detect::total_ram_mb());
|
||||
// the project's GPU server next to the engine (src/proverserver.rs): verified here once, so the gate it moves
|
||||
// (24 GB to 12 GB) applies to the install-time default
|
||||
let bin_dir = std::env::current_exe().ok().and_then(|p| p.parent().map(|d| d.to_path_buf())).unwrap_or_default();
|
||||
let patched = crate::proverserver::shipped(&bin_dir).map(|s| s.is_some()).unwrap_or(false);
|
||||
let d = crate::provedefault::decide_with_server(&cards, std::env::consts::OS, wsl, crate::detect::total_ram_mb(), patched);
|
||||
let on = d.on || already_on;
|
||||
{
|
||||
let mut s = self.settings.lock().unwrap();
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ mod jobrun;
|
|||
mod jobbuild;
|
||||
mod prover;
|
||||
mod provedefault;
|
||||
mod proverserver;
|
||||
mod segments;
|
||||
mod verifier;
|
||||
mod wslhost;
|
||||
|
|
|
|||
|
|
@ -14,6 +14,23 @@
|
|||
//!
|
||||
//! Decided 5 October 2026 (delegated by the project lead: "deploy what is absolute best"), docs/plans/proving-v1.md. The default
|
||||
//! never switches an explicit on back off, and Settings always wins afterwards.
|
||||
//!
|
||||
//! With the project's own build of the GPU server shipped and verified (src/proverserver.rs, 6 October 2026, the
|
||||
//! bench-log entry "prover floor": the patched server sizes its buffers to the shard instead of to a 24 GB card),
|
||||
//! the tiers are measured on the cards themselves and the gate moves to 12 GB:
|
||||
//!
|
||||
//! | Card (the per-card VRAM read, `nvidia-smi` MiB) | Profile (`SP1_GPU_ELEMENT_THRESHOLD` to the host) | Why |
|
||||
//! |---|---|---|
|
||||
//! | 20 GB and more (24 GB, 32 GB) | the server's own sizes (no override) | 16,851 MiB on the v1 shard at upstream's threshold on the 5090, 3.7 GB under the stock server |
|
||||
//! | 14 to 20 GB (16 GB) | 2^27 = 134,217,728 | 12,915 MiB alone, 10.95 GB own plus the miner's 1.7 GB beside it, 17.4 s a v1 shard (the 5090's allocation) |
|
||||
//! | 10 to 14 GB (12 GB) | 2^26 = 67,108,864, mining and proving | the RTX 4070 itself: 9,034 MiB of 12,282 beside its own miner on PC 1 (Windows, 24.1 s a v1 shard), 10.1 to 10.2 GB on a headless Linux 4070 and 5070 (the GPU fleet, 27 to 37 s); 7.6 GB alone |
|
||||
//! | 7 to 10 GB (8 GB, 10 GB) | 2^26, PROVES ALONE (off by default: the app proves beside the miner) | a 3080 10 GB proves alone at 7.9 to 8.2 GB (7.0 to 7.2 s) and a 4060 Ti 8 GB at 7.74 GB of 8.19 (9.6 s), the GPU fleet; beside the miner 7.7 + 1.4 GB is over 8 GB; 2^27 never fits |
|
||||
//! | under 7 GB | off, with the reason | the compressed shard alone is 7.7 GB at the smallest threshold that proves it in a minute |
|
||||
//!
|
||||
//! `Settings` overrides the profile (`prove_profile`: auto, 2^25, 2^26, 2^27, stock). A point that does not fit must
|
||||
//! never sit idle (the fleet's finding: the server hung for 15 minutes at 0%): the server now fails such a shard at
|
||||
//! once (the floor patch's panic hook) and the app gives every shard a wall-clock budget (`shard_budget`) and steps
|
||||
//! the threshold down one notch on a timeout (`step_down`: 2^27 to 2^26 to 2^25) before the next try.
|
||||
|
||||
use crate::state::CardState;
|
||||
|
||||
|
|
@ -37,6 +54,101 @@ fn gb(mb: u64) -> u64 {
|
|||
(mb + 512) / 1024
|
||||
}
|
||||
|
||||
/// The patched server's gate for mining AND proving: a 12 GB card (`nvidia-smi` 12,282 for the RTX 4070, 12,288
|
||||
/// for a 3060). A 10 GB 3080 (10,240) and an 8 GB 4060 Ti (8,188) prove alone (`MIN_VRAM_MB_PROVE_ALONE`).
|
||||
pub const MIN_VRAM_MB_PATCHED: u64 = 11_000;
|
||||
pub const MIN_VRAM_MB_PROVE_ALONE: u64 = 7_000;
|
||||
|
||||
/// The wall-clock budget for one shard: three times the measured time of a v1 shard (22,172 pgas) beside the
|
||||
/// miner on the 12 GB card at the profile's threshold, scaled by the shard's pgas, never under 120 s and never over
|
||||
/// 30 minutes. The point: a shard that does not fit the card is killed and the threshold stepped down instead of
|
||||
/// the card sitting idle (the GPU fleet, 6 October 2026: 15 minutes at 0% on the 8 and 10 GB cards at 2^27).
|
||||
/// Measured bases, the RTX 4070 beside its miner (bench-log "prover floor"): 2^27 17.3 s, 2^26 24.1 s, 2^25
|
||||
/// 40.9 s; the server's own sizes on a 24 GB card about 20 s (the 4090 beside its miner 26.1 s at 2^26, the fleet).
|
||||
pub fn shard_budget(threshold: Option<u64>, pgas: u64) -> std::time::Duration {
|
||||
let base_s: f64 = match threshold {
|
||||
Some(THRESHOLD_2_25) => 40.9,
|
||||
Some(THRESHOLD_2_26) => 24.1,
|
||||
Some(THRESHOLD_2_27) => 17.3,
|
||||
_ => 26.1,
|
||||
};
|
||||
let scale = (pgas as f64 / 22_172.0).max(1.0);
|
||||
let s = (3.0 * base_s * scale).clamp(120.0, 1800.0);
|
||||
std::time::Duration::from_secs(s as u64)
|
||||
}
|
||||
|
||||
/// After a timeout at `threshold`: the next notch down, or None when there is none (2^25 is the last: under it the
|
||||
/// time grows past the deadline, bench-log "route 2" 2^24 at 56.6 s alone). The server's own sizes step to 2^27.
|
||||
pub fn step_down(threshold: Option<u64>) -> Option<Option<u64>> {
|
||||
match threshold {
|
||||
None => Some(Some(THRESHOLD_2_27)),
|
||||
Some(THRESHOLD_2_27) => Some(Some(THRESHOLD_2_26)),
|
||||
Some(THRESHOLD_2_26) => Some(Some(THRESHOLD_2_25)),
|
||||
Some(t) if t > THRESHOLD_2_27 => Some(Some(THRESHOLD_2_27)),
|
||||
Some(t) if t > THRESHOLD_2_26 => Some(Some(THRESHOLD_2_26)),
|
||||
Some(t) if t > THRESHOLD_2_25 => Some(Some(THRESHOLD_2_25)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn threshold_name(t: Option<u64>) -> String {
|
||||
match t {
|
||||
None => "the server's own sizes".into(),
|
||||
Some(THRESHOLD_2_25) => "2^25".into(),
|
||||
Some(THRESHOLD_2_26) => "2^26".into(),
|
||||
Some(THRESHOLD_2_27) => "2^27".into(),
|
||||
Some(v) => v.to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The per-card profile on the patched server: the environment the host gets, and one line for the tile.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct Profile {
|
||||
/// "12gb", "16gb", "24gb", "32gb" or "off".
|
||||
pub tier: &'static str,
|
||||
/// `SP1_GPU_ELEMENT_THRESHOLD` for the host (the server reads it); None = the server's own sizes.
|
||||
pub threshold: Option<u64>,
|
||||
/// Whether the card mines and proves at once on this profile (measured), or proves only.
|
||||
pub mine_and_prove: bool,
|
||||
pub line: String,
|
||||
}
|
||||
|
||||
pub const THRESHOLD_2_25: u64 = 1 << 25;
|
||||
pub const THRESHOLD_2_26: u64 = 1 << 26;
|
||||
pub const THRESHOLD_2_27: u64 = 1 << 27;
|
||||
|
||||
/// The profile from the card's VRAM, as measured (the module's table). `override_name` is Settings' `prove_profile`
|
||||
/// ("auto" or "" = the table; "2^25", "2^26", "2^27" or "stock" force a threshold, the tier line says so).
|
||||
pub fn profile(vram_mb: u64, override_name: &str) -> Profile {
|
||||
let forced = match override_name.trim() {
|
||||
"2^25" | "2**25" | "33554432" => Some(Some(THRESHOLD_2_25)),
|
||||
"2^26" | "2**26" | "67108864" => Some(Some(THRESHOLD_2_26)),
|
||||
"2^27" | "2**27" | "134217728" => Some(Some(THRESHOLD_2_27)),
|
||||
"stock" | "full" => Some(None),
|
||||
_ => None,
|
||||
};
|
||||
let auto = if vram_mb >= 20_000 {
|
||||
Profile { tier: if vram_mb >= VRAM_MB_PROTOTYPE_SHARD { "32gb" } else { "24gb" }, threshold: None, mine_and_prove: true, line: format!("{} GB card: the server's own sizes (16.9 GB measured on the v1 shard at upstream's threshold)", gb(vram_mb)) }
|
||||
} else if vram_mb >= 14_000 {
|
||||
Profile { tier: "16gb", threshold: Some(THRESHOLD_2_27), mine_and_prove: true, line: format!("{} GB card: threshold 2^27 (12.9 GB alone, 10.95 GB plus the miner beside it, measured on the 5090's allocation)", gb(vram_mb)) }
|
||||
} else if vram_mb >= MIN_VRAM_MB_PATCHED {
|
||||
Profile { tier: "12gb", threshold: Some(THRESHOLD_2_26), mine_and_prove: true, line: format!("{} GB card: threshold 2^26, mines and proves (the RTX 4070 measured 9,034 MiB of 12,282 beside its own miner on Windows, 24.1 s a shard; 10.1 to 10.2 GB on a headless Linux 4070 and 5070)", gb(vram_mb)) }
|
||||
} else if vram_mb >= MIN_VRAM_MB_PROVE_ALONE {
|
||||
Profile { tier: "8gb", threshold: Some(THRESHOLD_2_26), mine_and_prove: false, line: format!("{} GB card: threshold 2^26, proves alone (a 3080 10 GB at 7.9 to 8.2 GB and a 4060 Ti 8 GB at 7.74 GB of 8.19, the GPU fleet); beside its miner 7.7 + 1.4 GB is over the card, so proving stays off while it mines; 2^27 never fits", gb(vram_mb)) }
|
||||
} else {
|
||||
Profile { tier: "off", threshold: None, mine_and_prove: false, line: format!("{} GB card: proving off, the compressed shard alone needs 7.7 GB at the smallest threshold that proves it in a minute (measured on 8 and 12 GB cards)", gb(vram_mb)) }
|
||||
};
|
||||
match forced {
|
||||
Some(t) if auto.tier != "off" || t.is_some() => Profile {
|
||||
tier: auto.tier,
|
||||
threshold: t,
|
||||
mine_and_prove: auto.mine_and_prove,
|
||||
line: format!("{} GB card: Settings forces {} (the measured profile would be {})", gb(vram_mb), match t { Some(THRESHOLD_2_25) => "threshold 2^25", Some(THRESHOLD_2_26) => "threshold 2^26", Some(THRESHOLD_2_27) => "threshold 2^27", _ => "the server's own sizes" }, match auto.threshold { Some(THRESHOLD_2_26) => "2^26", Some(THRESHOLD_2_27) => "2^27", _ => "the server's own sizes" }),
|
||||
},
|
||||
_ => auto,
|
||||
}
|
||||
}
|
||||
|
||||
/// Windows machines under this much RAM stay off until measured (consequences review C4, 5 October 2026): PC 2 at
|
||||
/// 63 GB had 25.6 GB in use with the WSL2 VM's working set at 7.9 GB while proving; a 16 GB PC would swap.
|
||||
pub const MIN_RAM_MB_WINDOWS: u64 = 31_000;
|
||||
|
|
@ -50,9 +162,19 @@ pub fn aggregation_card(cards: &[CardState]) -> Option<&CardState> {
|
|||
/// `os` is `std::env::consts::OS` ("windows", "linux", "macos"); `wsl_answers` is read on Windows only; `ram_mb` is the
|
||||
/// machine's RAM when the platform reports it (None = unknown, no gate).
|
||||
pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option<bool>, ram_mb: Option<u64>) -> Decision {
|
||||
decide_with_server(cards, os, wsl_answers, ram_mb, false)
|
||||
}
|
||||
|
||||
/// `patched_server`: the project's verified GPU server is in the payload (src/proverserver.rs), so the gate is the
|
||||
/// measured 12 GB one and the tile names the card's profile; without it, the stock server's 24 GB gate.
|
||||
pub fn decide_with_server(cards: &[CardState], os: &str, wsl_answers: Option<bool>, ram_mb: Option<u64>, patched_server: bool) -> Decision {
|
||||
let nvidia: Vec<&CardState> = cards.iter().filter(|c| c.vendor == "nvidia").collect();
|
||||
// a mining card needs 20 GB (the measured mine-and-prove peak of 16.8 GB), a card that only proves 16 GB
|
||||
let able: Vec<&CardState> = nvidia.iter().copied().filter(|c| c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).collect();
|
||||
// the stock server: a card needs 24 GB (its floor is 13.9 GB for an empty shard, 20.4 for a full one); the
|
||||
// patched server: 12 GB (the RTX 4070's own measurement)
|
||||
// with the patched server a card that only proves ALONE (8 and 10 GB) is not on by default: the app proves
|
||||
// beside the miner, and 7.7 GB plus the miner's 1.4 GB is over such a card (the GPU fleet); Settings may force it
|
||||
let gate = |c: &CardState| if patched_server { MIN_VRAM_MB_PATCHED } else if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY };
|
||||
let able: Vec<&CardState> = nvidia.iter().copied().filter(|c| c.vram_mb >= gate(c)).collect();
|
||||
let off = |line: String| Decision { on: false, line };
|
||||
if os == "macos" {
|
||||
return off("proving stays off on Apple silicon: the M5 Max CPU took 41 to 55 s for an empty shard and minutes for a full one; Settings switches it on (CPU, slow)".into());
|
||||
|
|
@ -63,7 +185,11 @@ pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option<bool>, ram_mb:
|
|||
} else {
|
||||
nvidia.iter().map(|c| format!("{} {} GB{}", c.name, gb(c.vram_mb), if c.enabled { ", mining" } else { "" })).collect::<Vec<_>>().join(", ")
|
||||
};
|
||||
let why = if nvidia.iter().any(|c| c.vram_mb >= 15_872) {
|
||||
let why = if patched_server && nvidia.iter().any(|c| c.vram_mb >= MIN_VRAM_MB_PROVE_ALONE) {
|
||||
"this card proves ALONE on the patched server (7.7 to 8.2 GB a shard at threshold 2^26, measured on a 3080 and a 4060 Ti 8 GB) but not beside its miner, so proving stays off while it mines; Settings switches it on at your own risk"
|
||||
} else if patched_server && !nvidia.is_empty() {
|
||||
"the smallest card that proves is 8 GB (7.7 GB a shard alone at threshold 2^26); this card is under that"
|
||||
} else if nvidia.iter().any(|c| c.vram_mb >= 15_872) {
|
||||
"a full shard needs a 24 GB card (measured 20.4 GB on the adopted shard size, 13.9 GB for an empty one); this card is under that, so Settings would switch proving on at your own risk"
|
||||
} else if nvidia.is_empty() {
|
||||
"this machine mines and does not prove: no zkVM proves on an AMD GPU today, and the CPU prover costs about 5 minutes a shard at a 30 GB RSS (bench-log, the SP1 CPU prover on PC 1); proving needs an NVIDIA card with 24 GB or more"
|
||||
|
|
@ -80,7 +206,9 @@ pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option<bool>, ram_mb:
|
|||
}
|
||||
}
|
||||
}
|
||||
let size_note = if best.vram_mb >= VRAM_MB_PROTOTYPE_SHARD { "" } else { "; until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" };
|
||||
let size_note = if patched_server {
|
||||
format!("; {}", profile(best.vram_mb, "").line)
|
||||
} else if best.vram_mb >= VRAM_MB_PROTOTYPE_SHARD { String::new() } else { "; until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on".to_string() };
|
||||
match os {
|
||||
"windows" => match wsl_answers {
|
||||
Some(true) => Decision { on: true, line: format!("proving on by default: {card} with WSL2 (Ubuntu-24.04 answers){size_note}; Settings switches it off") },
|
||||
|
|
@ -95,6 +223,74 @@ pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option<bool>, ram_mb:
|
|||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn the_profile_follows_the_measured_tiers_and_settings_can_force_one() {
|
||||
let p = profile(12_282, "");
|
||||
assert_eq!((p.tier, p.threshold, p.mine_and_prove), ("12gb", Some(THRESHOLD_2_26), true));
|
||||
assert!(p.line.contains("RTX 4070") && p.line.contains("9,034"), "{}", p.line);
|
||||
assert_eq!(profile(12_288, "auto").threshold, Some(THRESHOLD_2_26), "a 3060 12 GB");
|
||||
let p = profile(16_303, "");
|
||||
assert_eq!((p.tier, p.threshold), ("16gb", Some(THRESHOLD_2_27)));
|
||||
let p = profile(24_564, "");
|
||||
assert_eq!((p.tier, p.threshold), ("24gb", None));
|
||||
assert_eq!(profile(32_607, "").tier, "32gb");
|
||||
// the GPU fleet's rows (6 October 2026): 10 and 8 GB prove alone at 2^26, never 2^27; under 7 GB off
|
||||
let p = profile(10_240, "");
|
||||
assert_eq!((p.tier, p.threshold, p.mine_and_prove), ("8gb", Some(THRESHOLD_2_26), false));
|
||||
assert!(p.line.contains("proves alone") && p.line.contains("2^27 never fits"), "{}", p.line);
|
||||
let p = profile(8_188, "");
|
||||
assert_eq!((p.tier, p.threshold, p.mine_and_prove), ("8gb", Some(THRESHOLD_2_26), false));
|
||||
let p = profile(6_144, "");
|
||||
assert_eq!((p.tier, p.threshold, p.mine_and_prove), ("off", None, false));
|
||||
assert!(p.line.contains("proving off") && p.line.contains("7.7 GB"), "{}", p.line);
|
||||
// Settings: a forced threshold keeps the tier and says so; a forced stock profile on a 12 GB card is allowed too
|
||||
let p = profile(12_282, "2^25");
|
||||
assert_eq!((p.tier, p.threshold), ("12gb", Some(THRESHOLD_2_25)));
|
||||
assert!(p.line.contains("Settings forces threshold 2^25") && p.line.contains("would be 2^26"), "{}", p.line);
|
||||
assert_eq!(profile(24_564, "2^27").threshold, Some(THRESHOLD_2_27));
|
||||
assert_eq!(profile(16_303, "stock").threshold, None);
|
||||
// a 6 GB card forced to a threshold proves at the owner's risk; forced to stock it stays off
|
||||
assert_eq!(profile(6_144, "2^25").threshold, Some(THRESHOLD_2_25));
|
||||
assert_eq!(profile(6_144, "stock").tier, "off");
|
||||
assert_eq!(profile(12_282, "nonsense"), profile(12_282, ""));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_shard_budget_scales_with_the_tier_and_the_shard_and_the_step_down_ends_at_2_25() {
|
||||
use std::time::Duration;
|
||||
// the v1 shard: 3x the measured time, floored at 120 s
|
||||
assert_eq!(shard_budget(Some(THRESHOLD_2_26), 22_172), Duration::from_secs(120));
|
||||
assert_eq!(shard_budget(Some(THRESHOLD_2_25), 22_172), Duration::from_secs(122));
|
||||
assert_eq!(shard_budget(None, 0), Duration::from_secs(120), "an unknown size is the v1 shard");
|
||||
// the prototype shard (6.75 M pgas) at 2^25: 3 x 40.9 x 304 would be hours; capped at 30 minutes
|
||||
assert_eq!(shard_budget(Some(THRESHOLD_2_25), 6_751_568), Duration::from_secs(1800));
|
||||
// a 120,000-pgas block's shard at 2^26: 3 x 24.1 x 5.41 = 391 s
|
||||
assert_eq!(shard_budget(Some(THRESHOLD_2_26), 120_000), Duration::from_secs(391));
|
||||
assert_eq!(step_down(None), Some(Some(THRESHOLD_2_27)));
|
||||
assert_eq!(step_down(Some(THRESHOLD_2_27)), Some(Some(THRESHOLD_2_26)));
|
||||
assert_eq!(step_down(Some(THRESHOLD_2_26)), Some(Some(THRESHOLD_2_25)));
|
||||
assert_eq!(step_down(Some(THRESHOLD_2_25)), None, "2^25 is the last notch");
|
||||
assert_eq!(step_down(Some(100_000_000)), Some(Some(THRESHOLD_2_26)), "an odd value steps to the notch below it");
|
||||
assert_eq!(threshold_name(Some(THRESHOLD_2_26)), "2^26");
|
||||
assert_eq!(threshold_name(None), "the server's own sizes");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn with_the_patched_server_a_12_gb_card_is_on_and_a_10_gb_one_off() {
|
||||
let d = decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 4070", 12_282)], "windows", Some(true), Some(95_902), true);
|
||||
assert!(d.on, "{}", d.line);
|
||||
assert!(d.line.contains("threshold 2^26") && d.line.contains("mines and proves"), "{}", d.line);
|
||||
let d = decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 3080", 10_240)], "linux", None, None, true);
|
||||
assert!(!d.on && d.line.contains("proves ALONE") && d.line.contains("not beside its miner"), "{}", d.line);
|
||||
let d = decide_with_server(&[card("nvidia", "NVIDIA GeForce GTX 1660", 6_144)], "linux", None, None, true);
|
||||
assert!(!d.on && d.line.contains("smallest card that proves is 8 GB"), "{}", d.line);
|
||||
// the same cards without the patched server: the old 24 GB gate
|
||||
assert!(!decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 4070", 12_282)], "linux", None, None, false).on);
|
||||
assert!(decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None, true).on);
|
||||
let d = decide_with_server(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, None, true);
|
||||
assert!(d.on && d.line.contains("the server's own sizes"), "{}", d.line);
|
||||
}
|
||||
|
||||
fn card(vendor: &str, name: &str, vram_mb: u64) -> CardState {
|
||||
CardState { vendor: vendor.into(), name: name.into(), vram_mb, enabled: true, ..Default::default() }
|
||||
}
|
||||
|
|
|
|||
|
|
@ -130,6 +130,20 @@ struct Tools {
|
|||
wsl: bool,
|
||||
setup_script: Option<PathBuf>,
|
||||
cuda: bool,
|
||||
/// The project's patched GPU server shipped in the payload and verified (src/proverserver.rs): its manifest and
|
||||
/// its path as Ubuntu sees it. None: the stock server (the SDK's download) and the 24 GB gate.
|
||||
server: Option<(crate::proverserver::ServerManifest, String)>,
|
||||
/// Why the shipped server is not in use although it was shipped (verification failed, or the fallback fired).
|
||||
server_note: String,
|
||||
}
|
||||
|
||||
/// The shipped server for `Tools`: verified, or the reason it is not used (the stock server then).
|
||||
fn shipped_server(bin_dir: &Path) -> (Option<(crate::proverserver::ServerManifest, String)>, String) {
|
||||
match crate::proverserver::shipped(bin_dir) {
|
||||
Ok(Some((m, bin))) => (Some((m, crate::wslhost::wsl_path(&bin))), String::new()),
|
||||
Ok(None) => (None, String::new()),
|
||||
Err(e) => (None, format!("the shipped prover server is not used: {e}; the stock server (24 GB cards) runs instead")),
|
||||
}
|
||||
}
|
||||
|
||||
fn evm_rpc(shared: &Shared, method: &str, params: Value, timeout: Duration) -> Result<Value, String> {
|
||||
|
|
@ -171,7 +185,8 @@ fn find_tools(bin_dir: &Path) -> Result<Tools, String> {
|
|||
// a Linux path: never PathBuf::join here, which writes a backslash on Windows ("/opt/igneum\\igneum-prove-export"
|
||||
// broke every export on PC 2 under 0.3.7, 5 October 2026)
|
||||
let export = PathBuf::from(format!("{}/igneum-prove-export", host.to_string_lossy().rsplit_once('/').map(|(d, _)| d).unwrap_or("")));
|
||||
Ok(Tools { host, export, miner, wsl: true, setup_script: setup.exists().then_some(setup), cuda: text.contains("cuda") })
|
||||
let (server, server_note) = shipped_server(bin_dir);
|
||||
Ok(Tools { host, export, miner, wsl: true, setup_script: setup.exists().then_some(setup), cuda: text.contains("cuda"), server, server_note })
|
||||
}
|
||||
None => Err(probe_message(bin_dir, answered, setup.exists())),
|
||||
}
|
||||
|
|
@ -181,7 +196,7 @@ fn find_tools(bin_dir: &Path) -> Result<Tools, String> {
|
|||
if !host.exists() || !export.exists() {
|
||||
return Err(format!("igneum-prove-host and igneum-prove-export are not next to the engine ({})", bin_dir.display()));
|
||||
}
|
||||
Ok(Tools { host, export, miner, wsl: false, setup_script: None, cuda: false })
|
||||
Ok(Tools { host, export, miner, wsl: false, setup_script: None, cuda: false, server: None, server_note: String::new() })
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -282,6 +297,100 @@ fn run_tool(shared: &Shared, t: &Tools, exe: &Path, args: &[String], env: &[(&st
|
|||
}
|
||||
}
|
||||
|
||||
/// The threshold the host gets (`SP1_GPU_ELEMENT_THRESHOLD`, as a string) on the patched server: the biggest
|
||||
/// NVIDIA card's profile, with Settings' `prove_profile` override. None on the stock server or at the server's own sizes.
|
||||
fn profile_threshold(shared: &Shared, t: &Tools, stepped_down: Option<Option<u64>>) -> Option<u64> {
|
||||
if t.server.is_none() {
|
||||
return None;
|
||||
}
|
||||
let vram = shared.state.lock().unwrap().mining.cards.iter().filter(|c| c.vendor == "nvidia").map(|c| c.vram_mb).max().unwrap_or(0);
|
||||
let override_name = shared.settings.lock().unwrap().prove_profile.clone();
|
||||
let p = crate::provedefault::profile(vram, &override_name);
|
||||
match stepped_down {
|
||||
Some(th) => {
|
||||
set(shared, |st| st.server_profile = format!("{}: stepped down to {} after a shard ran past its budget (the measured profile is {})", p.tier, crate::provedefault::threshold_name(th), crate::provedefault::threshold_name(p.threshold)));
|
||||
th
|
||||
}
|
||||
None => {
|
||||
set(shared, |st| st.server_profile = format!("{}: {}", p.tier, p.line));
|
||||
p.threshold
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kills the GPU server inside Ubuntu-24.04 after a timeout (the host was killed by `run_tool`; the server it
|
||||
/// spawned normally dies with it, this covers a straggler) and unlinks its socket.
|
||||
fn kill_server(t: &Tools) {
|
||||
if !t.wsl {
|
||||
return;
|
||||
}
|
||||
if let Ok(file) = crate::wslhost::write_script("prove-kill", "pkill -f sp1-gpu-server 2>/dev/null; sleep 1; pkill -9 -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock; echo killed\n") {
|
||||
let _ = crate::platform::quiet(&mut crate::wslhost::command(&crate::platform::tool("wsl"), crate::wslhost::DISTRO, None, &file.path, true, &[])).output();
|
||||
}
|
||||
}
|
||||
|
||||
/// The prefix of the closure's error when a shard ran past its wall-clock budget (the step-down, not a proving error).
|
||||
const TIMEOUT_MARK: &str = "shard-timeout:";
|
||||
|
||||
/// Windows: puts the shipped, verified server where the SDK looks (`~/.sp1/bin/sp1-gpu-server` in Ubuntu-24.04) when
|
||||
/// the one there differs, or restores the stock one after the fallback; records the server in the state either way.
|
||||
fn install_server(shared: &Shared, t: &mut Tools, fallback_to_stock: &mut bool) {
|
||||
if !t.wsl {
|
||||
return;
|
||||
}
|
||||
let Some((m, bin_wsl)) = t.server.clone() else {
|
||||
let note = t.server_note.clone();
|
||||
set(shared, |p| {
|
||||
p.server_version = String::new();
|
||||
p.server_sha256 = String::new();
|
||||
p.server_kind = "stock".into();
|
||||
p.server_note = note;
|
||||
});
|
||||
return;
|
||||
};
|
||||
let body = if *fallback_to_stock { crate::proverserver::restore_stock_script() } else { crate::proverserver::install_script(&bin_wsl, &m.sha256) };
|
||||
let line = match crate::wslhost::write_script("prove-server", &body) {
|
||||
Ok(file) => crate::platform::quiet(&mut crate::wslhost::command(&crate::platform::tool("wsl"), crate::wslhost::DISTRO, None, &file.path, true, &[])).output().map(|o| String::from_utf8_lossy(&o.stdout).lines().find(|l| l.starts_with("RESULT server")).unwrap_or("").to_string()).unwrap_or_default(),
|
||||
Err(e) => format!("RESULT server install FAILED: cannot write the script: {e}"),
|
||||
};
|
||||
shared.log(&format!("prover: {}", if line.is_empty() { "the server install script printed nothing" } else { line.as_str() }));
|
||||
if *fallback_to_stock {
|
||||
t.server_note = "the patched GPU server did not come up on this machine; the stock server (24 GB cards) runs instead".into();
|
||||
t.server = None;
|
||||
*fallback_to_stock = false;
|
||||
let note = t.server_note.clone();
|
||||
set(shared, |p| {
|
||||
p.server_version = String::new();
|
||||
p.server_sha256 = String::new();
|
||||
p.server_kind = "stock".into();
|
||||
p.server_note = note;
|
||||
});
|
||||
return;
|
||||
}
|
||||
let installed = line.contains(" installed") || line.contains(" kept");
|
||||
if !installed {
|
||||
t.server_note = format!("the shipped GPU server could not be installed into Ubuntu-24.04 ({}); the stock server runs instead", if line.is_empty() { "no answer" } else { line.as_str() });
|
||||
t.server = None;
|
||||
}
|
||||
let note = t.server_note.clone();
|
||||
set(shared, |p| {
|
||||
if installed {
|
||||
p.server_version = m.sp1_version.clone();
|
||||
p.server_sha256 = m.sha256.clone();
|
||||
p.server_kind = "patched".into();
|
||||
p.server_note = String::new();
|
||||
} else {
|
||||
p.server_version = String::new();
|
||||
p.server_sha256 = String::new();
|
||||
p.server_kind = "stock".into();
|
||||
p.server_note = note;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The prefix of the closure's error when the GPU server itself did not come up (the fallback, not a proving error).
|
||||
const SERVER_DOWN_MARK: &str = "server-down:";
|
||||
|
||||
fn payout_address(shared: &Shared) -> String {
|
||||
shared.settings.lock().unwrap().address.clone()
|
||||
}
|
||||
|
|
@ -356,6 +465,12 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
let mut held_segments: Vec<HeldSegment> = Vec::new();
|
||||
let mut last_verifier_read = Instant::now() - Duration::from_secs(600);
|
||||
let mut asked_restart = false;
|
||||
// the fallback to the stock server (src/proverserver.rs `fallback`): set after a proof failed because the
|
||||
// patched server did not come up; the next probe restores the stock server and keeps the note
|
||||
let mut fallback_to_stock = false;
|
||||
// a shard that timed out stepped the threshold down (src/provedefault.rs `step_down`); the override holds for
|
||||
// the session and the tile names it (`server_profile`)
|
||||
let mut stepped_down: Option<Option<u64>> = None;
|
||||
// macOS and Linux: the host sits next to the engine, so its pinned ids are read at once, proving on or off
|
||||
// (Windows runs the host inside WSL2, which is probed only once proving is on)
|
||||
if !cfg!(windows) {
|
||||
|
|
@ -388,7 +503,7 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
if tools.is_none() && last_probe.elapsed() >= Duration::from_secs(60) {
|
||||
last_probe = Instant::now();
|
||||
match find_tools(&bin_dir) {
|
||||
Ok(t) => {
|
||||
Ok(mut t) => {
|
||||
shared.log(&format!("prover: host {}{}{}", t.host.display(), if t.wsl { " (WSL2)" } else { "" }, if t.cuda { ", CUDA" } else { ", CPU (slow)" }));
|
||||
set(&shared, |p| {
|
||||
p.available = true;
|
||||
|
|
@ -403,6 +518,7 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
shared.send(crate::engine::Cmd::RestartNode("the WSL2 prover is installed now; the node restarts to verify proof records".into()));
|
||||
}
|
||||
read_ids(&shared, &t);
|
||||
install_server(&shared, &mut t, &mut fallback_to_stock);
|
||||
tools = Some(t);
|
||||
}
|
||||
Err(e) => {
|
||||
|
|
@ -643,7 +759,31 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
}
|
||||
set(&shared, |p| p.message = if t.cuda { "proving on the GPU".into() } else { "CPU prover: about five minutes a shard, 30 GB of RAM, paid only when no card proves first".into() });
|
||||
let prover_env = if t.cuda { "cuda" } else { "cpu" };
|
||||
let (ok, out) = run_tool(&shared, t, &t.host, &[fix_p, "--mode".into(), "compressed".into(), "--shard".into(), w.shard.to_string(), "--prover".into(), payout.clone(), "--out".into(), res_p], &[("SP1_PROVER", prover_env), ("RUST_LOG", "off")], Duration::from_secs(3 * 3600), &dir.join(format!("prove-{}-{}.log", w.number, w.shard)));
|
||||
// the per-card profile on the patched server (src/provedefault.rs `profile`): the threshold the server
|
||||
// sizes its buffers by, from the biggest NVIDIA card's VRAM and Settings' override; none on the stock server
|
||||
let threshold = profile_threshold(&shared, t, stepped_down);
|
||||
let threshold_s = threshold.map(|v| v.to_string());
|
||||
let mut env: Vec<(&str, &str)> = vec![("SP1_PROVER", prover_env), ("RUST_LOG", "off")];
|
||||
if let Some(th) = threshold_s.as_deref() {
|
||||
env.push(("SP1_GPU_ELEMENT_THRESHOLD", th));
|
||||
}
|
||||
// the wall-clock budget (src/provedefault.rs `shard_budget`): a shard that does not fit the card is
|
||||
// killed and the threshold stepped down, never left at 0% (the GPU fleet's finding, 6 October 2026);
|
||||
// the CPU prover keeps its three hours
|
||||
let budget = if t.cuda && t.server.is_some() { crate::provedefault::shard_budget(threshold, w.pgas) } else { Duration::from_secs(3 * 3600) };
|
||||
let (ok, out) = run_tool(&shared, t, &t.host, &[fix_p, "--mode".into(), "compressed".into(), "--shard".into(), w.shard.to_string(), "--prover".into(), payout.clone(), "--out".into(), res_p], &env, budget, &dir.join(format!("prove-{}-{}.log", w.number, w.shard)));
|
||||
if !ok && t.server.is_some() && out.contains("s limit)") {
|
||||
// the budget ran out: the server is killed with the host (the SDK spawned it kill-on-drop; the
|
||||
// install script's pkill covers a straggler) and the threshold steps down for the next shard
|
||||
return Err(format!("{TIMEOUT_MARK}block {} shard {} ({} pgas) ran past its {} s budget at {}", w.number, w.shard, w.pgas, budget.as_secs(), crate::provedefault::threshold_name(threshold)));
|
||||
}
|
||||
if !ok && t.server.is_some() {
|
||||
if let Some(next) = crate::proverserver::fallback(crate::proverserver::Kind::Patched, &out) {
|
||||
// the closure cannot touch the loop's state: the marker in the error is read after it
|
||||
let why = out.lines().find(|l| l.contains("sp1-gpu-server")).unwrap_or("").trim().to_string();
|
||||
return Err(format!("{SERVER_DOWN_MARK}{}: {why}", next.word()));
|
||||
}
|
||||
}
|
||||
if !ok || !results.exists() {
|
||||
let last = out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed").to_string();
|
||||
// the root-socket class (5 October 2026, PC 2 at 20:00Z and 21:25Z): a job that ran the host as root
|
||||
|
|
@ -673,6 +813,48 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
|
|||
}
|
||||
Ok(())
|
||||
})();
|
||||
// the fallback (src/proverserver.rs): the patched server did not come up, so the stock server takes over at
|
||||
// the next probe (install_server restores it) and the tile says why
|
||||
if let Err(e) = &outcome {
|
||||
if let Some(what) = e.strip_prefix(TIMEOUT_MARK) {
|
||||
let current = stepped_down.unwrap_or_else(|| profile_threshold(&shared, t, None));
|
||||
kill_server(t);
|
||||
match crate::provedefault::step_down(current) {
|
||||
Some(next) => {
|
||||
stepped_down = Some(next);
|
||||
shared.log(&format!("prover: {what}; the GPU server was killed and the threshold steps down to {} for the next shard", crate::provedefault::threshold_name(next)));
|
||||
set(&shared, |p| {
|
||||
p.failed += 1;
|
||||
p.status = "waiting".into();
|
||||
p.message = format!("a shard ran past its budget at {}; the next shard runs at {}", crate::provedefault::threshold_name(current), crate::provedefault::threshold_name(next));
|
||||
p.current = String::new();
|
||||
});
|
||||
}
|
||||
None => {
|
||||
shared.log(&format!("prover: {what}; 2^25 is the smallest threshold, so this card cannot hold shards of that size (proving stays on for smaller ones)"));
|
||||
set(&shared, |p| {
|
||||
p.failed += 1;
|
||||
p.status = "waiting".into();
|
||||
p.message = format!("a shard ran past its budget at {}, the smallest threshold; this card cannot hold shards of that size", crate::provedefault::threshold_name(current));
|
||||
p.current = String::new();
|
||||
});
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if let Some(rest) = e.strip_prefix(SERVER_DOWN_MARK) {
|
||||
fallback_to_stock = rest.starts_with("stock");
|
||||
shared.log(&format!("prover: the patched GPU server did not come up ({}); the stock server takes over at the next probe", rest.split_once(": ").map(|(_, w)| w).unwrap_or("")));
|
||||
tools = None;
|
||||
last_probe = Instant::now() - Duration::from_secs(600);
|
||||
set(&shared, |p| {
|
||||
p.status = "setup".into();
|
||||
p.message = "the patched GPU server did not start; switching to SP1's stock server (24 GB cards)".into();
|
||||
p.current = String::new();
|
||||
});
|
||||
continue;
|
||||
}
|
||||
}
|
||||
match outcome {
|
||||
Ok(()) => {
|
||||
shared.event("proving", &format!("block {} shard {} proven and submitted in {:.0} s", w.number, w.shard, started.elapsed().as_secs_f64()));
|
||||
|
|
|
|||
285
app/igneum-app/src/proverserver.rs
Normal file
285
app/igneum-app/src/proverserver.rs
Normal file
|
|
@ -0,0 +1,285 @@
|
|||
//! The project's own build of SP1's GPU prover server (prover floor, 6 October 2026: the stock `sp1-gpu-server`
|
||||
//! 6.8.1 refuses every card under 24 GB before it allocates and sizes every buffer for a 24 GB card; the patched
|
||||
//! one, `proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, sizes them to the shard, so a 12 GB card mines and
|
||||
//! proves: the RTX 4070 in PC 1 proved the v1 shard at 9,034 MiB beside its miner, bench-log "prover floor").
|
||||
//!
|
||||
//! The payload ships it under `wsl2\bin\sp1-gpu-server` next to `wsl2\prover-server.json` and `.sig`: a manifest
|
||||
//! in the format of `payload-inputs.json` (the SP1 version it patches, the patch's sha256, the CUDA targets, the
|
||||
//! binary's sha256 and size) signed on the Mac with the OTA key the app already trusts (`manifest::OTA_PUBLIC_KEY_HEX`).
|
||||
//! Before the server is used the app checks the signature, then the binary's sha256 and size against the manifest
|
||||
//! (`shipped`), and installs it where the SP1 SDK looks (`$HOME/.sp1/bin/sp1-gpu-server` inside Ubuntu-24.04,
|
||||
//! `sp1-cuda-6.8.1/src/server.rs`), replacing whatever is there when the sha256 differs (`install_script`). The
|
||||
//! SDK itself only checks `--version`, which both the stock and the patched server answer with 6.8.1, so the sha256
|
||||
//! is the only thing that says which server a machine runs; `/api/state` carries it (`ProvingState::server_*`).
|
||||
//!
|
||||
//! Fallback: when a proof fails because the server could not start or be reached while the patched server was the
|
||||
//! one installed, the next run goes back to the stock server (`fallback`), the SDK downloads it, and the tier rule
|
||||
//! falls back to the 24 GB gate; the tile says so. Nothing on chain or in the proof format changes with the server.
|
||||
use crate::manifest::{sha256_file, verify_signature, OTA_PUBLIC_KEY_HEX};
|
||||
use std::path::Path;
|
||||
|
||||
/// The manifest's `format` field.
|
||||
pub const FORMAT: &str = "igneum-prover-server/1";
|
||||
/// The stock `sp1-gpu-server` 6.8.1 the SDK downloads (`sp1_gpu_server_v6.8.1_x86_64.tar.gz`), measured on PC 2
|
||||
/// on 5 October 2026 (251,306,680 bytes).
|
||||
pub const STOCK_SHA256_6_8_1: &str = "c2642ad1c42e85d8525159cf0c7cd5200d8766c9be1283f452a1f9bf9fea725c";
|
||||
/// The file names next to the engine (Windows: under `wsl2\`).
|
||||
pub const BINARY: &str = "sp1-gpu-server";
|
||||
pub const MANIFEST: &str = "prover-server.json";
|
||||
pub const SIGNATURE: &str = "prover-server.json.sig";
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct ServerManifest {
|
||||
pub sp1_version: String,
|
||||
pub sp1_commit: String,
|
||||
pub patch_sha256: String,
|
||||
pub cuda_archs: String,
|
||||
pub built_at: String,
|
||||
pub built_by: String,
|
||||
pub sha256: String,
|
||||
pub bytes: u64,
|
||||
}
|
||||
|
||||
/// Which server a sha256 names.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum Kind {
|
||||
Patched,
|
||||
Stock,
|
||||
Unknown,
|
||||
}
|
||||
|
||||
impl Kind {
|
||||
pub fn word(self) -> &'static str {
|
||||
match self {
|
||||
Kind::Patched => "patched",
|
||||
Kind::Stock => "stock",
|
||||
Kind::Unknown => "unknown",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn kind(sha256: &str, shipped: Option<&ServerManifest>) -> Kind {
|
||||
if shipped.map(|m| m.sha256 == sha256).unwrap_or(false) {
|
||||
Kind::Patched
|
||||
} else if sha256 == STOCK_SHA256_6_8_1 {
|
||||
Kind::Stock
|
||||
} else {
|
||||
Kind::Unknown
|
||||
}
|
||||
}
|
||||
|
||||
fn is_hex(s: &str, n: usize) -> bool {
|
||||
s.len() == n && s.bytes().all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase())
|
||||
}
|
||||
|
||||
/// Parses the manifest text (after its signature was checked).
|
||||
pub fn parse(text: &str) -> Result<ServerManifest, String> {
|
||||
let v: serde_json::Value = serde_json::from_str(text).map_err(|e| format!("prover-server.json is not JSON: {e}"))?;
|
||||
let s = |k: &str| v.get(k).and_then(|x| x.as_str()).unwrap_or("").to_string();
|
||||
if s("format") != FORMAT {
|
||||
return Err(format!("prover-server.json: format {:?} is not {FORMAT}", s("format")));
|
||||
}
|
||||
let f = v.get("files").and_then(|f| f.get(BINARY)).ok_or_else(|| format!("prover-server.json names no {BINARY}"))?;
|
||||
let sha256 = f.get("sha256").and_then(|x| x.as_str()).unwrap_or("").to_string();
|
||||
let bytes = f.get("bytes").and_then(|x| x.as_u64()).unwrap_or(0);
|
||||
if !is_hex(&sha256, 64) {
|
||||
return Err("prover-server.json: the server's sha256 is not 64 lowercase hex characters".into());
|
||||
}
|
||||
if bytes == 0 {
|
||||
return Err("prover-server.json: the server's size is missing".into());
|
||||
}
|
||||
let m = ServerManifest {
|
||||
sp1_version: s("sp1_version"),
|
||||
sp1_commit: s("sp1_commit"),
|
||||
patch_sha256: s("patch_sha256"),
|
||||
cuda_archs: s("cuda_archs"),
|
||||
built_at: s("built_at"),
|
||||
built_by: s("built_by"),
|
||||
sha256,
|
||||
bytes,
|
||||
};
|
||||
if m.sp1_version.is_empty() {
|
||||
return Err("prover-server.json: sp1_version is missing".into());
|
||||
}
|
||||
Ok(m)
|
||||
}
|
||||
|
||||
/// The signature with the OTA key, then the parse.
|
||||
pub fn verify_and_parse(manifest_bytes: &[u8], sig_hex: &str) -> Result<ServerManifest, String> {
|
||||
verify_signature(manifest_bytes, sig_hex.trim(), OTA_PUBLIC_KEY_HEX)?;
|
||||
parse(&String::from_utf8_lossy(manifest_bytes))
|
||||
}
|
||||
|
||||
/// The binary on disk must be the one the manifest names: the same sha256, the same size. A modified or truncated
|
||||
/// binary is refused here, before anything runs it.
|
||||
pub fn check_binary(m: &ServerManifest, path: &Path) -> Result<(), String> {
|
||||
let size = std::fs::metadata(path).map_err(|e| format!("{}: {e}", path.display()))?.len();
|
||||
if size != m.bytes {
|
||||
return Err(format!("{}: {size} bytes is not the manifest's {}", path.display(), m.bytes));
|
||||
}
|
||||
let sum = sha256_file(path).map_err(|e| format!("{}: {e}", path.display()))?;
|
||||
if sum != m.sha256 {
|
||||
return Err(format!("{}: sha256 {sum} is not the manifest's {}", path.display(), m.sha256));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The shipped server next to the engine: `<bin_dir>/wsl2/{sp1-gpu-server, prover-server.json, prover-server.json.sig}`
|
||||
/// on Windows (`<bin_dir>/{...}` elsewhere), its manifest verified and the binary checked. `Err` names what is
|
||||
/// wrong; a payload without the three files is `Ok(None)` (the stock server then).
|
||||
pub fn shipped(bin_dir: &Path) -> Result<Option<(ServerManifest, std::path::PathBuf)>, String> {
|
||||
let dir = if cfg!(windows) { bin_dir.join("wsl2") } else { bin_dir.to_path_buf() };
|
||||
let bin = dir.join("bin").join(BINARY);
|
||||
let man = dir.join(MANIFEST);
|
||||
let sig = dir.join(SIGNATURE);
|
||||
if !bin.exists() && !man.exists() {
|
||||
return Ok(None);
|
||||
}
|
||||
if !bin.exists() || !man.exists() || !sig.exists() {
|
||||
return Err(format!("the shipped prover server is incomplete: {} {} {}", present(&bin), present(&man), present(&sig)));
|
||||
}
|
||||
let bytes = std::fs::read(&man).map_err(|e| format!("{}: {e}", man.display()))?;
|
||||
let sig_text = std::fs::read_to_string(&sig).map_err(|e| format!("{}: {e}", sig.display()))?;
|
||||
let m = verify_and_parse(&bytes, &sig_text)?;
|
||||
check_binary(&m, &bin)?;
|
||||
Ok(Some((m, bin)))
|
||||
}
|
||||
|
||||
fn present(p: &Path) -> String {
|
||||
format!("{}={}", p.file_name().map(|n| n.to_string_lossy().to_string()).unwrap_or_default(), if p.exists() { "present" } else { "MISSING" })
|
||||
}
|
||||
|
||||
/// The bash the app runs inside Ubuntu-24.04 (as the app's user) before a proof: installs the shipped server at
|
||||
/// `$HOME/.sp1/bin/sp1-gpu-server` when the one there has another sha256 (or is absent), and prints one line:
|
||||
/// `RESULT server <sha256> <installed|kept>`. `bin_wsl` is the shipped binary's path as seen from Ubuntu
|
||||
/// (`/mnt/<drive>/.../wsl2/bin/sp1-gpu-server`). The copy goes into the Linux file system because the SDK execs
|
||||
/// the path it finds and a server on the Windows drive starts slowly and may not carry the execute bit.
|
||||
pub fn install_script(bin_wsl: &str, sha256: &str) -> String {
|
||||
// the sha256 is validated lowercase hex (`parse`), so it goes into the script bare; the path is quoted
|
||||
let want: String = sha256.chars().filter(|c| c.is_ascii_hexdigit()).collect();
|
||||
format!(
|
||||
"set -u\nD=\"$HOME/.sp1/bin\"; T=\"$D/sp1-gpu-server\"\nmkdir -p \"$D\"\nhave=\"$(sha256sum \"$T\" 2>/dev/null | cut -c1-64)\"\nif [ \"$have\" = {want} ]; then echo \"RESULT server {want} kept\"; exit 0; fi\npkill -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock\ncp {src} \"$T.new\" && chmod +x \"$T.new\" && mv -f \"$T.new\" \"$T\" || {{ echo \"RESULT server {want} install FAILED\"; exit 1; }}\ngot=\"$(sha256sum \"$T\" | cut -c1-64)\"\nif [ \"$got\" = {want} ]; then echo \"RESULT server {want} installed\"; else echo \"RESULT server $got MISMATCH after the copy\"; exit 1; fi\n",
|
||||
src = sq(bin_wsl)
|
||||
)
|
||||
}
|
||||
|
||||
/// A single-quoted bash word (the rule of src/wslhost.rs, repeated here so the signer binary can include this
|
||||
/// module without the WSL code).
|
||||
fn sq(s: &str) -> String {
|
||||
format!("'{}'", s.replace('\'', "'\\''"))
|
||||
}
|
||||
|
||||
/// The bash that puts the stock server back (the fallback): removes the patched binary so the SDK downloads the
|
||||
/// stock release again, and prints `RESULT server stock restored`.
|
||||
pub fn restore_stock_script() -> String {
|
||||
"set -u\nT=\"$HOME/.sp1/bin/sp1-gpu-server\"\npkill -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock\nrm -f \"$T\"\necho \"RESULT server stock restored\"\n".to_string()
|
||||
}
|
||||
|
||||
/// The lines in a failed proof's log that mean the server itself did not come up or answer, as the SDK's client
|
||||
/// words them (`sp1-cuda-6.8.1/src/error.rs`, `client.rs`): not a proving error, not a verify failure.
|
||||
const SERVER_DOWN: [&str; 5] = [
|
||||
"Could not start `sp1-gpu-server`",
|
||||
"Could not connect to `sp1-gpu-server` socket",
|
||||
"Could not check `sp1-gpu-server` version",
|
||||
"Failed to bind to socket addr",
|
||||
"Error running server:",
|
||||
];
|
||||
|
||||
/// After a failed proof: the server kind to use next. The patched server gives way to the stock one when the log
|
||||
/// says the server did not start or answer; a proving error (out of memory, a bad shard, a verify failure) keeps
|
||||
/// the server as it is, and the stock server never changes.
|
||||
pub fn fallback(current: Kind, failed_log: &str) -> Option<Kind> {
|
||||
if current == Kind::Patched && SERVER_DOWN.iter().any(|m| failed_log.contains(m)) {
|
||||
Some(Kind::Stock)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
const SHA_A: &str = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa";
|
||||
|
||||
fn manifest_json(sha: &str, bytes: u64) -> String {
|
||||
format!(
|
||||
r#"{{"format":"{FORMAT}","sp1_version":"6.8.1","sp1_commit":"c84ada1ed5911f28c4d3c9d0ed2f9e6cd7edb824","patch_sha256":"e81cb0d03b291f9fd4bf0a109d6da2d7c897795c9ffd7f797c0ddce723eee2b1","cuda_archs":"80,86,89,120","built_at":"2026-10-06T12:00:00Z","built_by":"ci:1","files":{{"{BINARY}":{{"sha256":"{sha}","bytes":{bytes}}}}}}}"#
|
||||
)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_manifest_parses_and_refuses_the_wrong_format_or_a_bad_hash() {
|
||||
let m = parse(&manifest_json(SHA_A, 10)).unwrap();
|
||||
assert_eq!(m.sp1_version, "6.8.1");
|
||||
assert_eq!(m.bytes, 10);
|
||||
assert_eq!(m.cuda_archs, "80,86,89,120");
|
||||
assert!(parse(&manifest_json("ABCD", 10)).unwrap_err().contains("64 lowercase hex"));
|
||||
assert!(parse(&manifest_json(SHA_A, 0)).unwrap_err().contains("size is missing"));
|
||||
assert!(parse(&manifest_json(SHA_A, 10).replace(FORMAT, "igneum-prover-server/2")).unwrap_err().contains("format"));
|
||||
assert!(parse("{}").unwrap_err().contains("format"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_modified_binary_is_refused_and_the_right_one_accepted() {
|
||||
let dir = std::env::temp_dir().join(format!("igneum-proverserver-{}", std::process::id()));
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
let bin = dir.join(BINARY);
|
||||
std::fs::write(&bin, b"the server bytes").unwrap();
|
||||
let sha = sha256_file(&bin).unwrap();
|
||||
let m = parse(&manifest_json(&sha, 16)).unwrap();
|
||||
check_binary(&m, &bin).unwrap();
|
||||
// one byte changed, same size: refused by the sha256
|
||||
std::fs::write(&bin, b"the server byteS").unwrap();
|
||||
let e = check_binary(&m, &bin).unwrap_err();
|
||||
assert!(e.contains("sha256") && e.contains("is not the manifest's"), "{e}");
|
||||
// truncated: refused by the size before the hash is even read
|
||||
std::fs::write(&bin, b"the server").unwrap();
|
||||
let e = check_binary(&m, &bin).unwrap_err();
|
||||
assert!(e.contains("10 bytes is not the manifest's 16"), "{e}");
|
||||
// missing: refused
|
||||
std::fs::remove_file(&bin).unwrap();
|
||||
assert!(check_binary(&m, &bin).is_err());
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unsigned_or_tampered_manifest_does_not_verify() {
|
||||
let text = manifest_json(SHA_A, 10);
|
||||
// a 64-byte zero signature is not the OTA key's signature of this text
|
||||
let zero_sig = "0".repeat(128);
|
||||
assert!(verify_and_parse(text.as_bytes(), &zero_sig).is_err());
|
||||
assert!(verify_and_parse(text.as_bytes(), "not hex").is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_kind_follows_the_sha256() {
|
||||
let m = parse(&manifest_json(SHA_A, 10)).unwrap();
|
||||
assert_eq!(kind(SHA_A, Some(&m)), Kind::Patched);
|
||||
assert_eq!(kind(STOCK_SHA256_6_8_1, Some(&m)), Kind::Stock);
|
||||
assert_eq!(kind(STOCK_SHA256_6_8_1, None), Kind::Stock);
|
||||
assert_eq!(kind("bbbb", Some(&m)), Kind::Unknown);
|
||||
assert_eq!(Kind::Patched.word(), "patched");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_install_script_quotes_its_paths_and_compares_the_hash_first() {
|
||||
let s = install_script("/mnt/c/Program Files/Igneum Miner/wsl2/bin/sp1-gpu-server", SHA_A);
|
||||
assert!(s.contains("'/mnt/c/Program Files/Igneum Miner/wsl2/bin/sp1-gpu-server'"), "{s}");
|
||||
assert!(s.contains(&format!("if [ \"$have\" = {SHA_A} ]; then echo \"RESULT server {SHA_A} kept\"")), "{s}");
|
||||
assert!(s.contains("pkill -f sp1-gpu-server") && s.contains("rm -f /tmp/sp1-cuda-*.sock"), "the running server is stopped and its socket unlinked before the swap");
|
||||
assert!(s.contains("chmod +x"));
|
||||
assert!(restore_stock_script().contains("rm -f \"$T\""));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_fallback_fires_only_for_a_server_that_did_not_come_up_and_only_from_patched() {
|
||||
let down = "Error: CudaClientError: Connect(Could not start `sp1-gpu-server`: No such file)";
|
||||
assert_eq!(fallback(Kind::Patched, down), Some(Kind::Stock));
|
||||
assert_eq!(fallback(Kind::Patched, "Could not connect to `sp1-gpu-server` socket: Connection refused"), Some(Kind::Stock));
|
||||
assert_eq!(fallback(Kind::Stock, down), None, "the stock server has nothing to fall back to");
|
||||
assert_eq!(fallback(Kind::Unknown, down), None);
|
||||
assert_eq!(fallback(Kind::Patched, "ProverError: CUDA_OUT_OF_MEMORY while proving shard 3"), None, "a proving error keeps the server");
|
||||
assert_eq!(fallback(Kind::Patched, "RESULT compressed shard 0: ... NOT VERIFIED"), None);
|
||||
}
|
||||
}
|
||||
|
|
@ -209,6 +209,15 @@ pub struct ProvingState {
|
|||
pub verifier_reason: String,
|
||||
/// the sentence on the tile for the state above
|
||||
pub verifier_note: String,
|
||||
/// The GPU prover server this machine runs (src/proverserver.rs): its SP1 version, its sha256, and whether it is
|
||||
/// the project's patched build ("patched"), SP1's stock release ("stock") or something else ("unknown");
|
||||
/// `server_profile` is the per-card profile in force (the tier and the threshold the host gets); `server_note`
|
||||
/// says why the stock server is in use when the patched one was shipped (the fallback).
|
||||
pub server_version: String,
|
||||
pub server_sha256: String,
|
||||
pub server_kind: String,
|
||||
pub server_profile: String,
|
||||
pub server_note: String,
|
||||
/// the node's proof pool: records held, verified, rejected
|
||||
pub pool_entries: u64,
|
||||
pub pool_verified: u64,
|
||||
|
|
|
|||
|
|
@ -150,3 +150,90 @@ So after sweep 1 the binding term is the Setup-time keys allocated at full capac
|
|||
trace buffer (keys and shards) to its padded need: `padded_trace_elements` in `jagged_tracegen/src/lib.rs`
|
||||
(each phase pads to the next multiple of 2^21 rows, `generate_jagged_traces`'s "final padding"), applied in
|
||||
`setup_tracegen` and `full_tracegen`, one stacking height of slack, `SP1_GPU_FLOOR_EXACT=0` restoring upstream.
|
||||
|
||||
## Sweep 2 (hung), the diagnosis, patch v3, sweep 3: under 11 GB
|
||||
|
||||
Sweep 2 (`floor-sweep-2`, 23:21Z, the v2 server fb3165d8) hung on its first point. The restore job
|
||||
(`floor-restore-1`, 23:57Z) read the point's log: the v2 server had panicked in a tokio worker at
|
||||
`jagged_tracegen/src/lib.rs:240` ("range end index 37,428,736 out of range for slice of length 36,700,160") and the
|
||||
SDK client waited on the socket for 1,740 s. The arithmetic named the bug: 36,700,160 is a recursion key's buffer
|
||||
as v2 sized it (the padded preprocessed traces, 35,651,584, plus one stacking height), 37,428,736 is that end plus
|
||||
one main trace of 1,777,152 elements: `prove_shard_with_pk` (`shard_prover/src/prover.rs` 332) runs
|
||||
`main_tracegen`, which appends the shard's main traces INTO the key's buffer, which upstream sized for a whole shard.
|
||||
Patch v3 keeps the key preprocessed-sized at Setup and lets `main_tracegen` grow it on first use (`grow_for_main`:
|
||||
a bigger dense buffer and column index, the preprocessed region copied device to device, swapped into the key under
|
||||
its lock; a loud `abort` instead of a hang if a bound were ever short). Build 5 (`floor-build-5`, 120 s):
|
||||
`b37defef9f5de43da06eb99a5aebdb6a0db214f49f485e00b86dd6d6c8b497b4`, 166,752,944 bytes, sm_86/89/120.
|
||||
|
||||
Sweep 3 (`floor-sweep-3`, 00:13:46 to 00:16:57Z, idle 2,089 MiB inside every peak, every proof VERIFIED by the
|
||||
unpatched host): the v1 shard at threshold 2^26 **10,291 MiB (8,202 with the idle subtracted), 5.7 s**; at 2^27
|
||||
12,915 MiB 4.3 s; the empty shard 9,939 to 9,971 MiB; one transfer 10,003 MiB 3.5 s; the 60 M-cycle prototype
|
||||
shard 13,459 MiB 16.8 s at the 12 GB tier (28,295 MiB on the stock server); the 16 GB tier 16,115 MiB; upstream's
|
||||
threshold 16,851 MiB (20,516 stock); the server after Setup 6,535 MiB (v1 9,703, stock 11,623); the grow line
|
||||
`36,700,160 -> 91,226,112 elements` on every recursion key's first use. The full table is in the bench-log entry.
|
||||
|
||||
### The tier line that follows (for the public copy, once a 12 GB card has run the same fixture)
|
||||
|
||||
| Card | Stock SP1 6.8.1 server | The v3 server (this branch), measured on the 5090's allocation | Profile |
|
||||
|---|---|---|---|
|
||||
| 8 GB | refused (the 24 GB panic) | not measured; the empty shard alone is 9.9 GB measured (7.9 GB the server's own), so no | mine only |
|
||||
| 12 GB | refused | MEASURED ON AN RTX 4070 12 GB (PC 1, 6 October): alone 7,553 MiB in 7.7 s at 2^26 (the card held nothing else); beside its own miner (1,449 MiB) 9,034 MiB of 12,282 in 24.1 s; core-only at 2^25 7,242 MiB beside the miner | `SP1_GPU_ELEMENT_THRESHOLD=67108864`: MINES AND PROVES (3.2 GB spare), no hand-off needed |
|
||||
| 16 GB | refused | the v1 shard alone 12.9 GB measured (10.95 own) at 2^27, 4.3 s; beside the miner 10.95 own + 1.7, 17.4 s (sweep 4) | 2^27, mine and prove (2.9 GB spare on paper) |
|
||||
| 24 GB | the v1 shard alone (20.4 GB), never the prototype shard (28.3) | the v1 shard 12.9 GB at 2^27, the prototype shard 13.5 GB (16.8 s) | upstream's 24 GB threshold or 2^27 |
|
||||
| 32 GB | everything (28.3 GB for the prototype shard beside the miner at 30.1) | 16.9 GB at upstream's threshold | unchanged |
|
||||
|
||||
Every row is the 5090's allocation pattern under a budget; the public line keeps "24 GB" until the on-order 12 GB
|
||||
card runs `fees-v1-shards2` shard 0 through this server and its own peak and time are in the bench-log.
|
||||
|
||||
## Sweep 4, beside the miner (job `floor-sweep-4`, 00:35 to 00:38Z)
|
||||
|
||||
The v3 server with the 0.3.11 worker mining on the same card (95%, 338 W, 3,833 MiB resident before the points):
|
||||
2^26: the v1 shard 12,066 MiB (8,233 own) 24.4 s, the empty shard 11,586 MiB (7,753 own) 13.0 s; 2^25: the v1 shard
|
||||
12,066 MiB 43.5 s; 2^27: the v1 shard 14,786 MiB (10,953 own) 17.4 s; every proof VERIFIED. The server's own
|
||||
working set does not move with the miner; the miner costs 4.3x in time; 2^25 is not a lever. The remaining floor is
|
||||
the Setup keys (6,535 MiB in use after Setup with the idle inside, about 4.4 GB own) and the recursion stage's
|
||||
peak (about 3.8 GB): cutting further means fewer keys built at Setup or a smaller recursion program, not a shard
|
||||
knob. The first publish of sweep 4 was refused by the publisher's kit-path-check (the wiped-jobs-folder rule of
|
||||
21:49Z); the measurement template now tests the chain job's kit before use.
|
||||
|
||||
### Route 2 measured (jobs `floor-core-alone`, `floor-core-miner`, 08:47 to 08:53Z)
|
||||
|
||||
Core-only at 2^26 on the v1 shard: 9,874 MiB alone (7,817 own) in 3.1 s and 12,834 MiB beside the miner (7,754 own)
|
||||
in 12.1 s, against the full compressed run's 8,233 own; the core proof 14,379,043 bytes, CPU verify 0.447 s; the
|
||||
aggregator's extra 2.5 s alone and 12.6 s beside a miner. The empty shard 6,377 own. The core proof's hand-off is
|
||||
a prover-protocol change (the pool and `igneum_submitProofRecord` carry compressed proofs today), not one line.
|
||||
The 12 GB mine-and-prove line (9.0 GB own) is not met by core-only at 2^26 (7.8 + 1.7 GB): the levers left are the
|
||||
core threshold at 2^25 and 2^24 for core-only proving and patch v4's `SP1_GPU_MEM_RELEASE_THRESHOLD` (the pool
|
||||
returns freed memory between shards; upstream holds the high-water mark for the process's life), both in
|
||||
`tools/prover-floor/pc2-floor-core2-miner.ps1` for the next PC 2 window. The table is in the bench-log entry.
|
||||
|
||||
### Route 2, second round (jobs `floor-build-6`, `floor-core2-miner`, 09:02 to 09:13Z): the verdict
|
||||
|
||||
Patch v4 adds `SP1_GPU_MEM_RELEASE_THRESHOLD` (`sp1-gpu/crates/cuda/src/task.rs`: upstream's u64::MAX keeps every
|
||||
freed allocation for the process's life). Core-only beside the miner: 2^25 6,409 MiB own in 19.8 s (core proof
|
||||
25.6 MB, CPU verify 0.79 s), 2^24 6,025 own in 56.6 s; the pool's release threshold does not move the peak. So a
|
||||
12 GB card mines and proves as a core-only prover at 2^25 (6.4 + 1.7 GB before the display) under the 9.0 GB line,
|
||||
with the hand-off to a compressing aggregator as the prover-protocol change; the full table and the per-tier
|
||||
consequences are in the bench-log entry. The real card's run decides the public line.
|
||||
|
||||
## Route 1 measured: the RTX 4070 12 GB in PC 1 (jobs `card12-alone`, `card12-miner`, 11:53 to 12:17Z)
|
||||
|
||||
Alone (the card's idle 0 MiB): the compressed v1 shard 7,553 MiB in 7.7 s at 2^26, 10,177 MiB in 5.5 s at 2^27;
|
||||
core-only 5,761 MiB at 2^25. Beside its own miner (1,449 MiB): 9,034 MiB in 24.1 s at 2^26, 11,754 MiB in 17.3 s at
|
||||
2^27, core-only 7,242 MiB at 2^25; a 23.6 M-cycle shard 9,066 MiB in 116.6 s at 2^26. Every proof verified by the
|
||||
unpatched verifier. So a 12 GB card mines and proves compressed shards at 2^26 with 3.2 GB spare, and route 2's
|
||||
core-only hand-off is the reserve, not the requirement. The 5090's allocation pattern overstated the card by about
|
||||
0.65 GB. The full tables and the per-tier consequences are in the bench-log entry; the public line moves to "12 GB
|
||||
mines and proves" when the packaging row ships the server.
|
||||
|
||||
## The packaging row (6 October 2026, afternoon)
|
||||
|
||||
Branch prover-floor carries the whole path: `packaging/prover/build-server.sh` (the one recipe), the CI workflow
|
||||
`prover-server.yml`, `push-server.sh` (the Mac signs the CI build's manifest with the OTA key and publishes),
|
||||
`fetch-server.sh` (verifies and places it for the payload), the `make-payload.sh` step (`wsl2\bin\sp1-gpu-server`,
|
||||
`wsl2\prover-server.json`, `.sig`), the Windows build's fetch step, and the app: `proverserver.rs` (the manifest,
|
||||
the hash check, the install into `~/.sp1/bin`, the fallback), `provedefault::profile` (the tiers from the VRAM,
|
||||
Settings' override), `/api/state`'s `server_*` fields. The tier table and the shipper's steps are in
|
||||
`docs/plans/proving-v1.md`, "The packaged GPU server". Tests: 147 in the app, of which the new ones refuse a
|
||||
modified or truncated binary and an unsigned manifest, choose the tier per VRAM, and fire the fallback only on a
|
||||
server that did not come up.
|
||||
|
|
|
|||
|
|
@ -2223,3 +2223,271 @@ Run b (`segments-pc2-pv1b`, 07:20Z to 07:51Z) claimed nothing in 88 passes: the
|
|||
| The rule as shipped (no switch): fresh refused while the previous segment is pending (known-failed), accepted after it is unproven | 22 passed | 166.2 s |
|
||||
| `--fresh-rule 0`: fresh accepted while the previous segment is pending, `freshAdmissible` true, still refused after a proven one, the second offer a duplicate ("segment already paid") | 23 passed | 139.9 s |
|
||||
|
||||
|
||||
## 5 to 6 October 2026, prover floor: the SP1 6.8.1 GPU server rebuilt for small cards (prover-floor agent)
|
||||
|
||||
Branch `prover-floor` (worktree `igneum-wt-prover-floor`); the model with file and line, the patch and the reading
|
||||
in `docs/analysis/prover-floor.md`; the fork is `proving/prover-floor/sp1-gpu-6.8.1-floor.patch` on SP1 tag v6.8.1
|
||||
(c84ada1e; the clone sits in `vendor/sp1-6.8.1`, gitignored). Every PC 2 row: the RTX 5090 (32,607 MiB) in WSL2
|
||||
Ubuntu-24.04 as root, the miners stopped by the job and the live prover switched off for the run, every
|
||||
`sp1-gpu-server` killed and `/tmp/sp1-cuda-*.sock` unlinked around every point, peak = `nvidia-smi
|
||||
--query-gpu=memory.used` at 1 s (the card's idle 1,755 to 2,060 MiB inside it), the proof `--mode compressed
|
||||
--shard 0` of the UNPATCHED pv1 host `/opt/igneum-pv1/igneum-prove-host` (sha dae6b006...) reaching the patched
|
||||
server through `HOME=/opt/igneum-floor/home` (the SDK spawns `$HOME/.sp1/bin/sp1-gpu-server`,
|
||||
`sp1-cuda-6.8.1/src/server.rs`), so VERIFIED is the unpatched verifier's word. Playbooks `tools/prover-floor/`.
|
||||
|
||||
| What | Measured |
|
||||
|---|---|
|
||||
| Why a 12 GB card proves nothing on the shipped server (source, `sp1-gpu/crates/prover_components/src/builder.rs`) | lines 35 to 39: `gpu_memory_gb = ceil(total / GiB) + 4`, `panic!("Unsupported GPU memory ... must be at least 24GB")` under 24: a 12 GB card reads 16, a 16 GB card 20, refused before any allocation. Lines 41 to 48: the core element threshold is a constant (402,653,184 elements, or 285,212,672 on a card reading 24 to 30) that overwrites the environment's `ELEMENT_THRESHOLD`; every device trace buffer is `threshold + 2^21` elements at 6 bytes each (`jagged_tracegen/src/lib.rs` 484 to 500): 2.26 GiB per shard in flight and the same again in the program's key, whatever the shard holds. The recursion keys 2^27 elements (0.75 GiB each), the allocator pool never returns memory (`cuda/src/task.rs` 152, 196) |
|
||||
| Toolchain on PC 2 (job `floor-toolchain-1`, 22:10Z, 4 s) | nvcc 12.8, cmake 3.28.3, gcc 13.3, clang 18, protoc 3.21.12, cargo 1.99.0, no Go, 16 cores, 30 GB WSL RAM, 925 GB free |
|
||||
| The build (jobs `floor-build-1..3`, 22:17 to 22:32Z) | runs 1 and 2 failed at 2 to 4 min on `gnark-ffi/build.rs:70` (no `go`); run 3 with go1.27.1 (tarball sha256 63d339f0..., checked, unpacked under `/opt/igneum-floor/go`) built in **240 s**: `sp1-gpu-server` 166,768,224 bytes, sha256 `5568108bf7fb9b0e525d8a08926b7046e51136ffaea53f0ca858631d0e938878`, `--version` 6.8.1, `CUDA_ARCHS=86,89,120` (cuobjdump: sm_86, sm_89, sm_120; the stock server lists 80, 86, 89, 90, 100, 120). Patch v1 (sha 700173fe): the panic removed, the threshold from a budget (`SP1_GPU_MEMORY_BUDGET_GB`, `SP1_GPU_ELEMENT_THRESHOLD`), FLOOR memory lines. The live `/root/.sp1/bin` server untouched throughout |
|
||||
| Sweep 1 (job `floor-sweep-1`, 22:34:56 to 22:37:55Z, patched server v1, card idle 1,755 MiB) | control at upstream's sizes: empty shard (block 83616, 280,706 cycles) **13,892 MiB** 2.2 s, v1 shard (fees-v1-shards2 shard 0, 4,717,439 cycles) **20,516 MiB** 4.2 s (the known curve). 12 GB tier (2^27): empty 12,740 MiB 2.4 s, v1 15,396 MiB 4.1 s. 16 GB tier (2^27+2^26): v1 18,628 MiB 3.8 s. Threshold 2^26: empty 12,772 MiB 3.1 s, v1 **12,708 MiB** 5.3 s (4 core shards). Threshold 2^25: v1 12,836 MiB 8.5 s. Normalize cache 1: 15,428 MiB (no change). Every proof VERIFIED, 1,272,897 bytes, verify 0.037 to 0.040 s |
|
||||
| The second floor (the server's own `FLOOR` lines in sweep 1) | `memory after setup: 9,703 MiB` at 2^26 (11,623 at the stock sizes) before any proof: at `Setup` the server pre-builds five recursion keys at the fixed 2^27 capacity (0.75 GB each, 90,177,536 of 134,217,728 elements used: 35.6 M preprocessed, 54.5 M main), the shrink key (2^25) and the core key at the threshold; the v1 shard at 2^26 is 4 core shards (main 44.0 M then 3 x 60.8 M elements) and 4 recursion proofs |
|
||||
| Patch v2 and build 4 (job `floor-build-4`, 23:18:13 to 23:20:14Z, 120 s warm) | every trace buffer sized to its padded need (`padded_trace_elements`: each phase to the next multiple of 2^21, one stacking height of slack, `SP1_GPU_FLOOR_EXACT=0` restores upstream), patch sha 08ce0555; the binary 166,748,808 bytes, sha256 `fb3165d809d2031cc05312c79a545065b2d22ced0a2cd134fb40462262449edf`, sm_86/89/120 |
|
||||
| Sweep 2 (job `floor-sweep-2`, 23:21:17Z, patched server v2) | HUNG on its first point: no row in 17 minutes against 11 to 18 s a point in sweep 1; at 23:40Z the coordinator gave PC 2 to the 0.3.11 update, which killed the job tree. The restore job (`floor-restore-1`, 23:57Z, 23 s) read the point's log: the server had PANICKED in a tokio worker ("range end index 37,428,736 out of range for slice of length 36,700,160", `jagged_tracegen/src/lib.rs:240`) and the SDK client waited on the socket for 1,740 s: a prove path (`main_tracegen`, `prove_shard_with_pk`) appends the shard's main traces INTO the key's buffer, which v2 had sized for the preprocessed phase alone (35,651,584 + 2^20) and upstream for a whole shard; before the panic v2 read 6,567 MiB after Setup. The restore killed nothing (the restart had), unlinked the root socket, switched the prover on; the live server untouched |
|
||||
| Patch v3 and build 5 (job `floor-build-5`, 00:09:55 to 00:11:57Z, 120 s warm) | `main_tracegen` grows a key's buffer to the shard's need on first use (a bigger dense buffer and column index, the preprocessed region copied device to device, swapped into the key; a loud abort instead of a hang if a bound were short), patch sha 3f9d3ab0; the binary 166,752,944 bytes, sha256 `b37defef9f5de43da06eb99a5aebdb6a0db214f49f485e00b86dd6d6c8b497b4`, sm_86/89/120 |
|
||||
| Sweep 3 (job `floor-sweep-3`, 00:13:46 to 00:16:57Z, patched server v3, card idle 2,089 MiB; the table below) | **the v1 shard at threshold 2^26: 10,291 MiB peak (8,202 MiB with the idle subtracted), 5.7 s, VERIFIED**: under the 11.0 GB gate on the 5090's allocation. The server after Setup: **6,535 MiB** (v1 9,703, stock 11,623). The prototype 60 M-cycle shard at the 12 GB tier: 13,459 MiB and 16.8 s (stock 28,295 MiB, 11.4 s). Upstream's threshold on v3: 16,851 MiB (stock 20,516) |
|
||||
|
||||
Sweep 3, every row (the patched server v3 `b37defef...`, sm_86, sm_89, sm_120, the RTX 5090 alone with the miners stopped and the prover off; the idle 2,089 MiB is the display and the other processes on PC 2's card, inside every peak; the unpatched host's VERIFIED on every row):
|
||||
|
||||
| Config (environment to the v3 server) | Fixture | Cycles | Peak MiB (idle inside) | Peak minus idle | Prove s | Verified |
|
||||
|---|---|---|---|---|---|---|
|
||||
| 12 GB tier (threshold 2^27) | empty shard (block 83616) | 280,706 | 9,939 | 7,850 | 2.6 | yes |
|
||||
| 12 GB tier | one transfer (block 56) | 556,369 | 10,003 | 7,914 | 3.5 | yes |
|
||||
| 12 GB tier | v1 shard (fees-v1-shards2 shard 0) | 4,717,439 | 12,915 | 10,826 | 4.3 | yes |
|
||||
| 12 GB tier | the full PROTOTYPE shard (block-338-shard1; 28,295 MiB on the stock server) | 60,415,376 | 13,459 | 11,370 | 16.8 | yes |
|
||||
| threshold 2^26 (`SP1_GPU_ELEMENT_THRESHOLD=67108864`) | v1 shard | 4,717,439 | 10,291 | 8,202 | 5.7 | yes |
|
||||
| threshold 2^26 | empty shard | 280,706 | 9,971 | 7,882 | 3.3 | yes |
|
||||
| 16 GB tier (2^27+2^26) | v1 shard | 4,717,439 | 16,115 | 14,026 | 4.0 | yes |
|
||||
| upstream's threshold (budget 32) | v1 shard (20,516 MiB on v1 and stock) | 4,717,439 | 16,851 | 14,762 | 4.0 | yes |
|
||||
| 12 GB tier + recursion allocation 100,663,296 | v1 shard | 4,717,439 | 12,883 | 10,794 | 4.3 | yes |
|
||||
|
||||
Sweep 4, beside the miner (job `floor-sweep-4`, 00:35:05 to 00:37:48Z, the v3 server, the 0.3.11 worker mining on the
|
||||
same card at 95% and 338 W, 3,808 to 3,833 MiB resident before the points, the resident set inside every peak; the
|
||||
first publish at 00:22Z was refused by the publisher's kit-path-check, the template now tests the chain job's kit first):
|
||||
|
||||
| Config | Fixture | Cycles | Peak MiB (resident inside) | Peak minus resident | Prove s | Verified |
|
||||
|---|---|---|---|---|---|---|
|
||||
| threshold 2^26 | v1 shard | 4,717,439 | 12,066 | 8,233 | 24.4 | yes |
|
||||
| threshold 2^26 | empty shard | 280,706 | 11,586 | 7,753 | 13.0 | yes |
|
||||
| threshold 2^25 | v1 shard | 4,717,439 | 12,066 | 8,233 | 43.5 | yes |
|
||||
| threshold 2^27 | v1 shard | 4,717,439 | 14,786 | 10,953 | 17.4 | yes |
|
||||
|
||||
Reading, 00:20Z. The task's gate (a real shard under 11.0 GB peak at under 60 s, the proof format unchanged) is met
|
||||
on PC 2: the adopted v1 shard proves at 10,291 MiB measured (8.2 GB the server's own) in 5.7 s on the v3 server
|
||||
at threshold 2^26, and the unpatched verifier passes every proof, so the pinned guest ids and the verifying key
|
||||
stand. The 12 GB tier profile is therefore `SP1_GPU_ELEMENT_THRESHOLD=67108864` on the v3 server (5.7 s a v1
|
||||
shard against 4.3 s at 2^27, which the proving agent accepts for the loop); the 16 GB tier takes 2^27 (12.9 GB
|
||||
measured, 10.8 the server's own); the 24 and 32 GB tiers gain 3.7 GB at upstream's own threshold and nothing
|
||||
they needed. What this does NOT say: no 12 GB card has run it (every row is the 5090's allocation pattern, the
|
||||
on-order 3060 or 4070 is the measurement for the public line, which stays 24 GB until then); the mine-and-prove
|
||||
case on a 12 GB card is sweep 4's row: the server's own working set is the same beside the miner (8,233 MiB
|
||||
against 8,202 alone), the miner costs 4.3x in time (24.4 s against 5.7 s a v1 shard, inside the 600 DAA-s deadline
|
||||
by 25x), and 2^25 buys no memory (12,066 MiB again at 43.5 s: the floor is now the Setup keys and the recursion
|
||||
stage, 4.4 GB own after Setup plus about 3.8 GB at the recursion peak). So a 12 GB card (12,288 MiB) mining and
|
||||
proving holds 8.2 GB of server plus 1.7 GB of miner plus its own display (0.5 to 1 GB, not measured): 10.4 to
|
||||
10.9 GB on paper, but 8.2 + 1.7 = 9.9 GB before the display is over the 9.0 GB line the proving agent holds for the
|
||||
tier, so the 12 GB tier reads "proves alone" (10.3 GB, 5.7 s) and mine-and-prove is not claimed there; a 16 GB
|
||||
card mines and proves at 2^27 (10.95 + 1.7 GB = 12.7 plus the display, under its 15.0 GB line) at 17.4 s. All of it the 5090's allocation pattern until the
|
||||
on-order 12 GB card runs the same two points. Shipping it is the project's own signed build of SP1's prover at
|
||||
every SP1 upgrade (the proving plan's packaging row before 0.3.12).
|
||||
|
||||
Route 2, core-only provers (the project lead, 07:40Z: can a 12 GB card mine AND prove; jobs `floor-core-alone` 08:47:07 to
|
||||
08:48:37Z with the miners stopped and `floor-core-miner` 08:50:29 to 08:53:13Z with the 5090 mining at 96%; the v3
|
||||
server b37defef, the pv1 host's `--mode shard` = the core proof, its CPU verify, then the compressed proof in one
|
||||
run, the 1-s sampler split at the core RESULT's timestamp so the core-only peak is the sampler's maximum up to it;
|
||||
every core and compressed proof VERIFIED by the unpatched host; the resident set inside every peak: 2,057 MiB
|
||||
alone, 5,080 MiB with the 0.3.11 miner):
|
||||
|
||||
| Threshold | Fixture | Core-only peak MiB alone (own) | Core s | Core proof bytes | CPU verify s | Compressed s (full peak) | Beside the miner: core peak (own) | Core s | Compressed s |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| 2^26 | v1 shard | 9,874 (7,817) | 3.1 | 14,379,043 | 0.447 | 5.6 (10,290) | 12,834 (7,754) | 12.1 | 24.7 |
|
||||
| 2^26 | empty | 8,434 (6,377) | 1.5 | 6,935,657 | 0.212 | 3.2 (9,938) | 11,458 (6,378) | 5.9 | 12.9 |
|
||||
| 2^27 | v1 shard | 12,850 (10,793) | 2.5 | 10,147,581 | 0.314 | 4.1 (12,850) | 15,874 (10,794) | 8.5 | 17.5 |
|
||||
| 2^27 | empty | 9,394 (7,337) | 1.4 | 5,603,823 | 0.170 | 2.4 (9,970) | 12,482 (7,402) | 4.8 | 9.8 |
|
||||
|
||||
Reading. Stopping at the core proof takes 0.4 GB off the v1 shard's own working set at 2^26 (8.2 to 7.8 GB) and
|
||||
1.5 GB off the empty shard's, not more, because the server still builds the compression keys at Setup (4.4 GB own
|
||||
in use before any shard, mostly the allocator pool's high-water mark from the key commits) and at 2^27 the core
|
||||
phase is the peak itself (the shard's own buffers). The aggregator's extra cost per shard, compressed minus core on
|
||||
the same card: 2.5 s alone and 12.6 s beside a miner at 2^26, 1.6 s at 2^27, 1.0 to 1.7 s for empty shards; the
|
||||
hand-off is 5.6 to 14.4 MB a shard (4 to 11x the 1.27 MB compressed proof), verified by the aggregator's CPU in
|
||||
0.17 to 0.45 s before it compresses. The proof the node verifies does not change (the aggregator posts the
|
||||
compressed proof; the record formats, the guest ids and the verifier stay); the pool does: today
|
||||
`igneum_submitProofRecord` carries compressed proofs and the pool verifies them with `--mode verify`, so a core
|
||||
proof needs a hand-off path from the small card to a compressing aggregator before submission, a prover-protocol
|
||||
change (the proving agent's sizing note), not one line. Per tier, on this measurement: 12 GB mine-and-prove is
|
||||
still over the 9.0 GB line (7.8 + 1.7 = 9.5 GB before the display); 8 GB cards cannot (6.4 GB own for an EMPTY
|
||||
shard); 16 GB mines and proves without route 2; 24 and 32 GB cards as aggregators compress 24 shards a minute alone
|
||||
or 5 beside a miner; a rig puts its small cards on core proofs and one big card on compression. The levers left for
|
||||
the 12 GB line are a smaller core threshold for CORE-ONLY proving (2^25 and 2^24, where the shard's buffers are
|
||||
the peak: not yet measured core-only) and the allocator pool returning freed memory between shards (patch v4,
|
||||
`SP1_GPU_MEM_RELEASE_THRESHOLD`, built but not yet run), both queued for the next PC 2 window.
|
||||
|
||||
Route 2, second round beside the miner (jobs `floor-build-6`, 09:02:05 to 09:04:06Z, patch v4 = v3 plus
|
||||
`SP1_GPU_MEM_RELEASE_THRESHOLD` in `sp1-gpu/crates/cuda/src/task.rs`, the server `68b2f512cd4085c03025f89013b42cfb3f2adec63106a1a228a3c93753d1e1ac`;
|
||||
`floor-core2-miner`, 09:05:45 to 09:13:22Z, the 5090 mining at 95%, 3,801 MiB resident inside every peak; every
|
||||
core and compressed proof VERIFIED; the release engineer's build job ran 08:55 to 09:02Z between the two rounds,
|
||||
serialised by the app's queue, never beside a point):
|
||||
|
||||
| Threshold | Fixture | Core-only peak MiB (own) | Core s | Core proof bytes | CPU verify s | Compressed s (full peak) |
|
||||
|---|---|---|---|---|---|---|
|
||||
| 2^25 | v1 shard | **10,210 (6,409)** | 19.8 | 25,621,187 | 0.789 | 43.3 (12,066) |
|
||||
| 2^25 | empty | 9,922 (6,121) | 8.7 | 10,929,863 | 0.321 | 19.6 (11,618) |
|
||||
| 2^24 | v1 shard | 9,826 (6,025) | 56.6 | 76,657,687 | 2.330 | 127.2 (12,131) |
|
||||
| 2^26, pool release threshold 0 | v1 shard | 11,683 (7,882) | 12.0 | 14,379,043 | 0.446 | 24.3 (12,035) |
|
||||
| 2^25, pool release threshold 0 | v1 shard | 10,147 (6,346) | 19.7 | 25,621,187 | 0.784 | 43.5 (12,003) |
|
||||
|
||||
Reading. For CORE-ONLY proving the core threshold is the lever the full pipeline lacked: 7.8 GB own at 2^26, 6.4
|
||||
at 2^25, 6.0 at 2^24, at 12.1, 19.8 and 56.6 s a v1 shard beside the miner. Returning freed memory to the driver
|
||||
(v4's env) does not lower the peak (the peak is live memory); it stops the prover holding its high-water mark
|
||||
between shards on a shared card. VERDICT on the project lead's question against the 9.0 GB line (the server's own): a 12 GB
|
||||
card can mine and prove as a CORE-ONLY prover at 2^25, 6.4 GB own plus the miner's 1.7 GB = 8.1 GB before the
|
||||
display (8.6 to 9.1 GB of the card's 12.3 with it), 19.8 s a v1 shard (the 600 DAA-s deadline by 30x), handing a
|
||||
25.6 MB core proof to an aggregator that verifies it in 0.8 s on a CPU and compresses it in 23.5 s beside its own
|
||||
miner (about 4 s on a proving-only card, approximate); it cannot mine and prove compressed proofs on the card
|
||||
itself at any threshold (8.2 GB own, sweep 4). Per tier: 8 GB: no (an empty shard's core is 6.1 GB own); 12 GB:
|
||||
mine and prove core-only at 2^25, or prove alone compressed at 2^26 (10.3 GB, 5.7 s); 16 GB: mines and proves
|
||||
compressed at 2^27 (route 2 not needed); 24 and 32 GB: the aggregators, a proving-only card compressing about 15
|
||||
shards a minute (approximate) and a mining one 2 to 3; a rig: the small cards on core proofs at 2^25, one big card
|
||||
compressing. What it costs: the hand-off path (core proof to a compressing aggregator before submission; the pool
|
||||
and `igneum_submitProofRecord` carry compressed proofs today) is a prover-protocol change, the compressor's credit
|
||||
a pool rule, 20x the bytes a shard on the relay; nothing on chain. Every number is the 5090's allocation pattern:
|
||||
the real 12 GB card (route 1, `tools/prover-floor/pc-12gb-card.ps1`, kit `igneum-prove-wsl2-floor.zip`
|
||||
e377c7df...) is the measurement that decides the public line.
|
||||
|
||||
Route 1, THE REAL 12 GB CARD (6 October 2026, the project lead's word at 11:25Z through main; an ASUS Dual RTX 4070 OC 12 GB,
|
||||
12,282 MiB, nvidia-smi index 1, PCIe 4.0 x4 through a USB4 enclosure, mining beside the 5090 in PC 1). PC 1 already
|
||||
had WSL2 Ubuntu-24.04 with nvcc 12.8 and cargo 1.99 (jobs `floor-pc1-wsl2-enable`, elevated, changed nothing, and
|
||||
`floor-pc1-toolchain`, 11:21 to 11:23Z); the v4 server built cold in 353 s (`floor-pc1-build`, 11:25:33 to 11:31:34Z,
|
||||
480 crates, `ed7b1e4b862c451b4a2cefa0e544f7b6a9e04279690c3f8c9881666bd7f8f23a`, sm_86/89/120, sm_89 native for the
|
||||
4070); the kit `igneum-prove-wsl2-floor.zip` (e377c7df...) fetched (`floor-pc1-kit`, 11:32:49Z) and the host built
|
||||
from it on PC 1 (adcad5ea..., the pinned ids matching). Every proof steered to the 4070 by `IGNEUM_CUDA_DEVICE=1`
|
||||
(the host change: the SDK starts the server with `CUDA_VISIBLE_DEVICES=1`); `CUDA_DEVICE_ORDER=PCI_BUS_ID`.
|
||||
|
||||
Alone (job `card12-alone`, 11:53:01 to 12:00:53Z, the miners stopped by the job, the prover off then on; the 4070's
|
||||
own idle **0 MiB** read before and after, so every peak is the server's own; `--mode shard` = the core proof, its
|
||||
CPU verify, then the compressed proof, the sampler split at the core RESULT; every proof VERIFIED):
|
||||
|
||||
| Threshold | Fixture | Cycles | Core-only peak MiB | Full compressed peak MiB | Core s | Compressed s | Core proof bytes | CPU verify s |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
| 2^26 | v1 shard | 4,717,439 | 7,265 | **7,553** | 3.7 | **7.7** | 14,379,043 | 0.432 |
|
||||
| 2^26 | block-72854 (the "empty block first" fixture is NOT empty: a 5x bigger shard) | 23,594,943 | 7,041 | 7,617 | 17.2 | 35.9 | 69,847,451 | 2.204 |
|
||||
| 2^27 | v1 shard | 4.7 M | 10,177 | 10,177 | 3.1 | 5.5 | 10,147,581 | 0.308 |
|
||||
| 2^27 | block-72854 | 23.6 M | 10,017 | 10,241 | 12.9 | 22.8 | 35,910,731 | 1.120 |
|
||||
| 2^25 | v1 shard | 4.7 M | **5,761** | 7,617 | 5.2 | 12.1 | 25,621,187 | 0.793 |
|
||||
| 2^25 | block-72854 | 23.6 M | 5,761 | 7,649 | 31.0 | 74.3 | 164,004,701 | 5.248 |
|
||||
|
||||
Reading. On the card itself a 12 GB card PROVES ALONE: the compressed v1 shard at 2^26 holds 7.55 GB of 12.28 GB
|
||||
(4.7 GB spare) in 7.7 s, at 2^27 10.2 GB (2.1 GB spare) in 5.5 s, and a 23.6 M-cycle shard fits at 2^26 in 36 s.
|
||||
The 5090's allocation pattern overstated the 4070 by about 0.65 GB (8.2 GB "own" against 7.55 measured). The
|
||||
4070 proves the v1 shard 1.35x slower than the 5090 alone (7.7 against 5.7 s at 2^26) through a PCIe 4.0 x4 link.
|
||||
Core-only at 2^25 is 5.76 GB on the card. Consequence for the public line: "12 GB proves (prove-only)" is now
|
||||
measured on a 12 GB card, so the line moves from 24 GB once the patched server ships (the packaging row before
|
||||
0.3.12); mine-and-prove is the beside-the-miner job's row (below, `card12-miner`).
|
||||
|
||||
Beside its own miner (job `card12-miner`, 12:02:36 to 12:17:25Z, the 0.3.12 miner running on both cards, the 4070's
|
||||
working set **1,449 MiB** read on the card before and after, inside every peak; the prover off then on; every
|
||||
proof VERIFIED):
|
||||
|
||||
| Threshold | Fixture | Core-only peak MiB | Full compressed peak MiB | Core s | Compressed s |
|
||||
|---|---|---|---|---|---|
|
||||
| 2^26 | v1 shard | 8,682 | **9,034** | 11.0 | **24.1** |
|
||||
| 2^26 | block-72854 (23.6 M cycles) | 8,426 | 9,066 | 55.8 | 116.6 |
|
||||
| 2^27 | v1 shard | 11,722 | 11,754 | 8.4 | 17.3 |
|
||||
| 2^27 | block-72854 | 11,434 | 11,690 | 37.5 | 68.7 |
|
||||
| 2^25 | v1 shard | **7,242** | 9,098 | 17.0 | 40.9 |
|
||||
| 2^25 | block-72854 | 7,242 | 9,226 | 107.7 | 256.4 |
|
||||
|
||||
VERDICT on the project lead's question, measured on the card itself: **a 12 GB card mines and proves.** Compressed proving at
|
||||
2^26 beside its own miner holds 9.03 GB of the card's 12.28 GB (3.2 GB spare) at 24.1 s a v1 shard (the 600 DAA-s
|
||||
deadline by 25x), with no core-only hand-off, no aggregator and no pool change; a 23.6 M-cycle shard fits the same
|
||||
way at 116.6 s. 2^27 fits with 0.5 GB spare (11.75 GB), so the profile is 2^26. Core-only at 2^25 is 7.24 GB beside
|
||||
the miner (5.0 GB spare), the reserve if the shard budget grows. The miner costs the 4070 3.1x in time (24.1 against
|
||||
7.7 s alone; the 5090 pays 4.3x). The 5090's allocation pattern, and the 9.0 GB line read from it, overstated the
|
||||
real card by about 0.65 GB: the card's own number decides, as the rule says.
|
||||
|
||||
Consequences per tier, now measured on a 12 GB card: **12 GB** (RTX 4070, 3060 12 GB) mines and proves compressed
|
||||
at 2^26 on the patched server (the public line moves to "12 GB mines and proves" once the packaging row ships the
|
||||
server; until then the app's default stays at 24 GB); **16 GB** mines and proves at 2^27 (the 5090's allocation:
|
||||
10.95 own + the miner, 2.9 GB spare, 17.4 s; the card itself not yet run); **8 GB**: not on this server (the
|
||||
core-only v1 shard alone is 5.76 GB on the 4070 at 2^25, so a compressed proof at 7.55 GB does not fit beside a
|
||||
miner on 8 GB; core-only at 2^25 plus a miner is about 7.2 GB and would need the hand-off path: an 8 GB row for a
|
||||
later day, not claimed); **24 and 32 GB**: unchanged, with 3.7 GB more headroom at upstream's threshold; a rig: every
|
||||
card proves its own compressed shards, no aggregator needed. AMD and Apple: outside SP1 (`docs/analysis/amd-proving.md`).
|
||||
What ships this: the packaging row (the project's signed build of SP1's GPU server, rebuilt and re-measured at every
|
||||
SP1 upgrade: `tools/prover-floor/pc2-build-server.ps1` is the recipe, the patch is 4 files and 206 lines on tag
|
||||
v6.8.1), the app's per-card profile (`provedefault.rs`: 12 GB at 2^26, 16 GB at 2^27, 24 GB at upstream's 24 GB
|
||||
threshold, 32 GB at the full one) and the HOME or path switch that points the SDK at the project's server.
|
||||
|
||||
The packaging row (6 October 2026, 12:20 to 12:52Z, branch prover-floor, commits c53aecb to the tip): the server
|
||||
built on CI and the PCs from `packaging/prover/build-server.sh`, signed on the Mac (`igneum-ota-sign sign-server`),
|
||||
carried by the Windows payload (`wsl2\bin\sp1-gpu-server`, `wsl2\prover-server.json`, `.sig`), verified and
|
||||
installed by the app (`src/proverserver.rs`), the tiers from the measured rows (`provedefault::profile`), the stock
|
||||
fallback; 147 app tests. Verification on PC 2 (job `floor-pc2-verify`, 58 s): the manifest's sha256 and size against
|
||||
PC 2's v4 server ok; a tampered copy refused; the app's install script installed it at `~/.sp1/bin` and kept it on
|
||||
the second run; one v1 shard on the 12 GB profile through the installed server: **9,886 MiB, 5.9 s, VERIFIED** (the
|
||||
SDK spawned the shipped server: its `FLOOR opts` line read `element_threshold=67108864`); the restore script removed
|
||||
it and PC 2's stock server came back from the copy (c2642ad1). Consequence: the public line "12 GB mines and proves
|
||||
(2^26), 16 GB and up at 2^27" ships with 0.3.13 on this path; the Windows installer grows about 60 MB zipped; no
|
||||
Mac or Linux user gets the server yet (no CUDA on a Mac; the Linux app proves on the CPU).
|
||||
|
||||
The hang on a point that does not fit (6 October 2026, 13:00 to 13:1xZ; the GPU fleet's finding on rented 8 and 10 GB
|
||||
cards, `docs/analysis/prover-tiers-real-cards.md`: threshold 2^27 on the 3080 and the 4060 Ti left the patched server
|
||||
at the card's memory limit at 0% for 15 minutes until killed). The cause is the one sweep 2 hit on 5 October: an
|
||||
allocation the card cannot meet makes `cudaMallocAsync` fail (`sp1-gpu/crates/cuda/src/stream.rs` 322 to 335, an
|
||||
`AllocError`), the buffer constructor panics inside a tokio worker task, and the request's future never answers; the
|
||||
client waits on its socket. Two fixes on prover-floor: (1) patch v5 (sha256 7fb7f9e886ff6b0f...): a panic hook in
|
||||
`sp1-gpu/crates/server/src/main.rs` names the stage from the panic's location (trace generation, the commit, LogUp
|
||||
GKR, the zerocheck, the jagged sumcheck, the recursion, a device allocation), says "lower SP1_GPU_ELEMENT_THRESHOLD
|
||||
one notch" when the message is an allocation, and exits 70, so the host's proof fails at once with that line; (2) the
|
||||
app (`prover.rs`, `provedefault.rs`): every shard gets a wall-clock budget, three times the measured time of a v1
|
||||
shard beside the miner at the profile's threshold (the 4070: 2^27 17.3 s, 2^26 24.1 s, 2^25 40.9 s) scaled by the
|
||||
shard's pgas, never under 120 s nor over 30 minutes (`shard_budget`); a shard past it is killed with the server and
|
||||
the threshold steps down 2^27 to 2^26 to 2^25 for the next shard (`step_down`), the tile naming the step; tests for
|
||||
both rules (148 in the app). The fleet's rows go into the tiers: 24 GB cards keep the server's own sizes (the stock
|
||||
server proves them, the 4090 at 17.4 GB in 5.6 s); 12 GB 2^26 mines and proves (10.1 to 10.2 GB beside the miner on
|
||||
headless Linux, 9.0 GB on PC 1's Windows); 10 and 8 GB prove ALONE at 2^26 (8,158 MiB own on the 3080 in 7.1 s, 7.74
|
||||
GB of 8.19 on the 4060 Ti in 9.6 s; 2^27 never; beside the miner the 3080 reaches 9,412 of 10,240 on headless Linux,
|
||||
0.8 GB spare, too close for a desktop, so the app leaves them off by default while mining and Settings can force
|
||||
it); the "empty block first" fixture (23.6 M cycles) is out of the tier line. Patch v5 built on PC 2 (`floor-build-7`,
|
||||
e911facb...) and PC 1 (`floor-pc1-build-2`, 028fda68...) in about 2 minutes each. The first known-failed candidate
|
||||
did not fail: job `floor-pc1-hangcase` (13:06:29 to 13:08:05Z, the miners stopped, the 4070 alone) proved the 60
|
||||
M-cycle prototype shard at 2^27 at **10,785 MiB of 12,282 in 24.8 s** (core 15.1 s, 39.7 MB; every proof verified),
|
||||
where the 5090's allocation had said 13,459 MiB: so a 12 GB card proves even the biggest devnet shard alone at 2^27,
|
||||
and the 5090's pattern overstates a 12 GB card by about 2.7 GB on big shards (0.65 GB on the v1 shard). The
|
||||
known-failed case of the gate, on a real card: the GPU fleet's rented 3080 10 GB (driver 570.211, sm_86), the server
|
||||
rebuilt from patch v5 in 86 s (binary 5d4f92a3..., 108,212,616 bytes), the point that hung 568 s on v4 (2^27 on the
|
||||
v1 shard, alone): **the host failed in 13 s wall** (setup 10.2 s, execute 0.3 s, then the abort), the server's line
|
||||
in its log "FLOOR abort: a device allocation failed at slop/crates/tensor/src/inner.rs:51: called `Result::unwrap()`
|
||||
on an `Err` value: AllocError { layout: Layout { size: 486586112, align: 4 } } (the card's memory could not meet an
|
||||
allocation: lower SP1_GPU_ELEMENT_THRESHOLD one notch)", the tracegen line before it at 8,872 MiB in use, the
|
||||
host's "Error: CudaClientError: Failed to read the response: UnexpectedEof", exit 1. The second candidate on PC 1
|
||||
(`floor-pc1-hangcase-2`, upstream's threshold 402,653,184 on the prototype shard on the 4070, 13:10:29Z) hit the
|
||||
job's 5-minute cap with no result line: on that card and point the failure did not reach the hook (a C++ CUDA
|
||||
exception in the sppark NTT code cannot unwind into Rust; or a stream synchronisation that never returns after a
|
||||
failed launch), or the point was still running; the point's log (job `floor-pc1-restore`, 13:17:50Z, 16 s: no leftover held
|
||||
the card, the prover switched on) said which: NOT a dead task. The server allocated the full 2.26 GB trace buffer at
|
||||
that threshold and filled it (dense 404,750,336 elements, the card at 12,281 MiB in use) and the sampler read 11,937
|
||||
MiB at 100% utilisation on all 260 samples: the proof was running inside a card full by a hair, slowly (11 s on the
|
||||
5090; 301 s was not enough on the 4070 at the limit), finishing or not unknown. The two classes are each covered by
|
||||
the half that owns them: an allocation the card cannot meet, the hook (seconds); a point that fits by a hair and
|
||||
crawls, the app's budget (120 s for a v1 shard, then the server killed and the threshold stepped down; its runtime
|
||||
demonstration needs the 0.3.13 app, an item of the cut's verification run). Rule from it: a playbook that switches
|
||||
the prover off runs its point under `timeout` inside the script (`POINT_BUDGET_S`, 180 s), under the job's cap, so
|
||||
the restore tail always runs. The 4060 Ti 8 GB rows
|
||||
from the fleet: alone 2^26 9.6 s at 7,740 MiB, 2^25 compressed 13.5 s at 7,676, core 2^25 5.8 s at 5,916, core 2^26
|
||||
5.0 s at 7,196, core 2^24 11.2 s at 5,404; 2^27 hung 904 s on v4; beside its miner the compressed 2^26 point hung too
|
||||
(7.7 GB plus the 1.4 GB miner on 8.2 GB), so an 8 GB card proves ALONE (the tier line) and core-only beside the
|
||||
miner is its open row.
|
||||
|
||||
Consequences (the rule of 5 October 2026), as they stood after sweep 1 (superseded by the reading above for the 12 and 16 GB tiers): the shipped SP1 GPU server refuses every
|
||||
card under 24 GB before allocating, so 8, 12 and 16 GB NVIDIA cards cannot prove on it whatever the shard; the v1
|
||||
patch takes the shard's term out (20.5 to 12.7 GB on the v1 shard) at a 1.26x time cost (5.3 s against 4.2 s,
|
||||
which the proving agent accepts for the 12 GB profile: the loop's carriage is 25 to 30 s around any proof and the
|
||||
deadline 600 s), but 12.7 GB measured (10.95 GB the server's own) still does not fit a 12 GB card (12,288 MiB,
|
||||
minus the display), so the 12 GB tier stays "mine only" and the public line stays 24 GB; the 16 GB tier (16,384
|
||||
MiB) proves the v1 shard alone on the v1 patch at 12.7 GB with 3.7 GB spare but not beside the miner (1.7 to
|
||||
1.8 GB measured on the 5090) without the v2 cut; the 24 GB tier gains nothing it needs; the 32 GB tier is
|
||||
unchanged. Patch v2 is the cut that decides the 12 GB tier and it is unmeasured tonight. AMD and Apple cards are
|
||||
outside this: SP1 has no HIP or Metal path (`docs/analysis/amd-proving.md`).
|
||||
|
|
|
|||
|
|
@ -116,9 +116,64 @@ The resume path (5 October 2026, the 0.3.11 app): `POST /api/resume` on 0.3.9 re
|
|||
|
||||
The prover-floor agent's first sweep (job `floor-sweep-1`, 22:34 to 22:38Z, PC 2's 5090, the miners stopped, this plan's per-point recipe, its patched `sp1-gpu-server` 5568108b built for sm_86, sm_89 and sm_120, every proof VERIFIED by the unpatched pv1 host): the control at upstream's sizes reproduces the curve above (empty shard 13,892 MiB and 2.2 s; the v1 shard 20,516 MiB and 4.2 s); with the core element threshold at 2^26 the v1 shard proves as four core shards in 5.3 s at **12,708 MiB** and the empty shard at 12,772 MiB; 2^25 gives 12,836 MiB at 8.5 s; 2^27 gives 15,396 MiB. The 12.7 GB left is the server's Setup (five recursion keys pre-built at a fixed 2^27 capacity plus the shrink and core keys: 9.7 GB before the first shard), which its patch v2 sizes to the need. Decided for the 12 GB profile: the split that lands under 11 GB wins (5.3 s a shard is inside the loop's own 25 to 30 s of carriage and 100x inside T); 2^27 is the second profile only if v2 leaves it under 11 GB with the miner's 1.8 GB beside it. The 12 GB row stays OPEN until the final pair (alone and beside the miner) lands and the on-order 3060 runs it.
|
||||
|
||||
### A self-built CUDA server (the 12 GB path), before 0.3.12 (consequences C26)
|
||||
### The packaged GPU server (the 12 GB path): the packaging row for 0.3.13 (6 October 2026, branch prover-floor)
|
||||
|
||||
If the prover-floor agent's rebuilt `sp1-gpu-server` (the Setup sizes cut, built on PC 2 under WSL2) proves a shard under 11 GB, it becomes a shipped artefact and needs its own row of rules before 0.3.12: it is built from a pinned SP1 source tag with `CUDA_ARCHS` covering sm_86, sm_89 and sm_120 (the 12 and 16 GB tiers are Ampere and Ada, not only the 5090's Blackwell; one card family per measured row), by the packaging path that builds the Windows payload (PC 1's build job for the Linux binary, the Mac signs the manifest as it does the DMG), lands in the DMG and the WSL2 package beside the host as `wsl2/bin/sp1-gpu-server` with its sha256 in `payload-inputs.json`, is named in `evidence.md` beside the prover rows ("prover built from SP1 <tag> at <sha>"), is rebuilt and re-measured at every SP1 upgrade, and ships only after `--mode verify-segment` and `--mode verify` on proofs it made show the pinned verifying keys unchanged (the server changes allocation, not the circuit; the ids `0x2b1a81cb...` and `0x474678f3...` must still verify them). The 12 GB claim itself waits for the on-order RTX 3060 to run that server on the same fixtures and recipe as the curve; until then the public line stays at 24 GB.
|
||||
The measurement that decides it is in the bench-log entry "prover floor": SP1 6.8.1's stock `sp1-gpu-server` refuses
|
||||
every card under 24 GB before it allocates (`sp1-gpu/crates/prover_components/src/builder.rs` 35 to 39) and sizes
|
||||
every buffer for a 24 GB card; the project's patch (`proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, 4 files, 206
|
||||
lines on tag v6.8.1) sizes them to the shard, and the RTX 4070 12 GB in PC 1 proved the v1 shard at 9,034 MiB of
|
||||
12,282 beside its own miner in 24.1 s (7,553 MiB alone in 7.7 s), every proof verified by the unpatched verifier.
|
||||
Nothing on chain, in the proof format, the pinned guest ids or the verifying key changes with the server.
|
||||
|
||||
| Piece | Where | What it does |
|
||||
|---|---|---|
|
||||
| The build | `packaging/prover/build-server.sh` (one recipe), run by CI `.github/workflows/prover-server.yml` (ubuntu-24.04, CUDA 12.8.1 from Jimver/cuda-toolkit, Go 1.27.1, `CUDA_ARCHS=80,86,89,120`) and by the PCs' WSL2 path `tools/prover-floor/pc2-build-server.ps1` | clones SP1 at v6.8.1 (c84ada1e), applies the patch, builds `sp1-gpu-server`, writes its sha256 and `build.json`. Not byte-reproducible across machines (SP1's `prover-types` build script stamps the build time in), so the signed manifest names ONE build, CI's; a PC build is a behavioural cross-check on the same fixtures |
|
||||
| The signing | `packaging/prover/push-server.sh --run <ci run id>` on the Mac | downloads the artifact, checks its recorded sha256, writes `prover-server.json` (format `igneum-prover-server/1`, `app/igneum-app/src/proverserver.rs`: the SP1 version and commit it patches, the patch's sha256, the CUDA targets, the binary's sha256 and size, the run id), signs it with the OTA key the apps already trust (`igneum-ota-sign sign`), publishes the binary, the manifest and `prover-server.json.sig` to the downloads host |
|
||||
| The payload | `packaging/prover/fetch-server.sh` (CI's windows.yml and the Mac), `packaging/windows/make-payload.sh` | fetches the three files, verifies the signature with the embedded key and the binary's sha256 and size (`igneum-ota-sign verify-server embedded ... --binary`), and the payload carries them as `wsl2in\sp1-gpu-server`, `wsl2\prover-server.json` and `.sig`. The Windows installer grows by about 60 MB zipped (167 MB raw). A host without the manifest ships no server (the stock one then); a manifest that does not verify fails the build. The Mac DMG carries nothing of it (a Linux binary no Mac runs) and the Linux app proves on the CPU today, so the Hive package waits for the CUDA path on Linux |
|
||||
| The app | `app/igneum-app/src/proverserver.rs`, `prover.rs`, `provedefault.rs`, `config.rs`, `state.rs` | at the prover's probe the three files are verified again (signature, then sha256 and size; a modified binary is refused before anything runs it, tested), the server is installed into Ubuntu-24.04 at `~/.sp1/bin/sp1-gpu-server` where the SDK looks (replacing whatever is there when the sha256 differs; the stock server, which answers `--version` 6.8.1 too, is told apart by its sha256 c2642ad1), and `/api/state` carries `server_version`, `server_sha256`, `server_kind` (patched, stock, unknown), `server_profile` and `server_note`. Each proof gets `SP1_GPU_ELEMENT_THRESHOLD` from the card's profile. If a proof fails because the server did not come up, the stock server is restored at the next probe and the tile says so (tested on the log lines the SDK writes) |
|
||||
| The tiers (`provedefault::profile`, from the per-card VRAM; Settings `prove_profile` overrides: auto, 2^25, 2^26, 2^27, stock) | 12 GB (10 to 14 GB read): 2^26, mines and proves; 16 GB (14 to 20): 2^27; 24 and 32 GB: the server's own sizes; under 10 GB: off, with the reason (the compressed shard alone is 7.6 GB on the 4070). The install-time default (`decide_with_server`) moves its gate from 24 GB to 12 GB only when the verified server is in the payload | measured rows: the 4070 itself (12 GB); the 5090's allocation for 16, 24 and 32 GB |
|
||||
|
||||
The tier table, measured (bench-log "prover floor"; alone / beside the card's own miner; the v1 shard, 4.7 M cycles):
|
||||
|
||||
| Card | Server | Peak MiB alone (time) | Peak MiB beside the miner (time) | Row |
|
||||
|---|---|---|---|---|
|
||||
| RTX 4070 12 GB (12,282), the card itself | patched, 2^26 | 7,553 (7.7 s) | 9,034 (24.1 s) | mines and proves |
|
||||
| RTX 4070 12 GB, the card itself | patched, 2^27 | 10,177 (5.5 s) | 11,754 (17.3 s) | fits with 0.5 GB spare: not the profile |
|
||||
| RTX 4070 12 GB, the card itself | patched, 2^25 core-only | 5,761 | 7,242 | the reserve (a hand-off path, route 2) |
|
||||
| 16 GB (the 5090's allocation) | patched, 2^27 | 12,915 (4.3 s) | 10,953 own + the miner (17.4 s) | mines and proves, the card itself not yet run |
|
||||
| 24 and 32 GB (the 5090) | patched, upstream's threshold | 16,851 (4.0 s) | | unchanged, 3.7 GB more headroom than stock |
|
||||
| any card under 24 GB | stock | refused before allocating | | the 24 GB gate until the server ships |
|
||||
|
||||
Verified on PC 2 (job `floor-pc2-verify`, 12:50:54 to 12:51:51Z, the manifest for PC 2's v4 server 68b2f512 signed
|
||||
on the Mac with the OTA key, fetched as `floor-pc2-manifest`): the manifest's sha256 and size against the binary ok;
|
||||
a tampered copy (one byte appended) refused; the app's install script put the server at `~/.sp1/bin/sp1-gpu-server`
|
||||
("installed", then "kept" on the second run), `--version` 6.8.1; one v1 shard on the 12 GB profile through the
|
||||
installed server with the default HOME (the SDK spawned the shipped one: `element_threshold=67108864` in its own
|
||||
line) at 9,886 MiB on the 5090 in 5.9 s, VERIFIED; the restore script removed it and the live server came back from
|
||||
the copy. The signing path on the Mac: `sign-server` and `verify-server embedded` agree, a tampered manifest and a
|
||||
wrong binary are refused.
|
||||
|
||||
The hang on a point that does not fit, and the rule it leaves (6 October 2026, 13:00 to 13:20Z; bench-log "prover
|
||||
floor"): patch v5's panic hook fails an allocation the card cannot meet in seconds with the stage named (the fleet's
|
||||
3080: 13 s against 568 s of hang), and the app's per-shard budget (`provedefault::shard_budget`, 120 s for a v1 shard)
|
||||
kills a point that fits by a hair and crawls, stepping the threshold down (`step_down`) for the next shard; PC 1's
|
||||
second hang case (upstream's threshold on the prototype shard on the 4070) was the second class: 11,937 of 12,282
|
||||
MiB at 100% for the job's whole 5-minute cap, not a dead task. Hygiene rule from it (the coordinator, 13:16Z): a job
|
||||
whose cap can fire while the prover is off runs its point under a budget INSIDE the script, shorter than the cap
|
||||
(`timeout`, `POINT_BUDGET_S`, 180 s in every prover-floor playbook), so the restore tail (the prover on, the sockets
|
||||
unlinked) always runs and no PC is left without its prover by a timed-out job; the 0.3.13 verification run exercises
|
||||
the app's own budget and step-down on PC 2 with the cut's app before the publish.
|
||||
|
||||
What the next cut's shipper must do (0.3.13), in order: (1) run the `prover-server` workflow on master (or merge
|
||||
prover-floor and let the push trigger it; about 40 minutes cold on the hosted runner, approximate); (2) on the Mac,
|
||||
`packaging/prover/push-server.sh --run <run id>` (the OTA key, the dl token and the Vercel login as for
|
||||
push-inputs.sh), which publishes and deploys the three files; (3) the Windows build then picks them up on its own
|
||||
(the new `prover server` step before `payload inputs`); on a Mac-made payload, `packaging/prover/fetch-server.sh`
|
||||
first; (4) the verification run on PC 2 before the publish: the app on the cut proves one shard with
|
||||
`/api/state` showing `server_kind: patched` and the card's `server_profile`, and `--mode id` unchanged; (5) the
|
||||
public line moves to "12 GB mines and proves (2^26), 16 GB and up at 2^27" with the cut, not before. Every SP1
|
||||
upgrade repeats (1) to (4) with the patch rebased (the patch is against the tag; `git apply` fails loudly when it
|
||||
no longer fits).
|
||||
|
||||
The root-socket class on PC 2, the two times: 20:00:56Z (my chain job's root run; the live prover failed with Connect(PermissionDenied) until the socket was gone) and 21:25:24Z (the aggregation-cost job's root run; the prover stayed dark through the 0.3.10 restart at 21:49:41Z until `socketfix-pc2-pv1` removed the root-owned `/tmp/sp1-cuda-0.sock` at 22:01:16Z; the next shard, block 89011, was proven at 22:02:13Z and paid, and every shard since). The permanent fix in the 0.3.11 app tree: every committed playbook that runs a prove mode as root carries `pkill -f sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock` at its start and end, `tools/ci/prover-socket-check.sh` (in `ci.yml`) fails a playbook without them, and the app's prover names the cause in its log line when the host reports PermissionDenied. The app itself cannot remove a socket another user owns, so a job written outside the tree must still follow the rule.
|
||||
|
||||
|
|
|
|||
64
packaging/prover/build-server.sh
Executable file
64
packaging/prover/build-server.sh
Executable file
|
|
@ -0,0 +1,64 @@
|
|||
#!/usr/bin/env bash
|
||||
# Builds the project's sp1-gpu-server (prover floor, 6 October 2026): SP1 at the pinned tag with the floor patch,
|
||||
# which sizes the server's buffers to the shard instead of to a 24 GB card (bench-log "prover floor"; the RTX 4070
|
||||
# mines and proves at 2^26). The one recipe CI (.github/workflows/prover-server.yml) and the PCs' WSL2 path
|
||||
# (tools/prover-floor/pc2-build-server.ps1) share.
|
||||
#
|
||||
# packaging/prover/build-server.sh <out dir> [source dir]
|
||||
#
|
||||
# Needs: nvcc (CUDA 12.8), cmake, gcc or clang, protoc, cargo, git, and Go (the server's native-gnark feature;
|
||||
# GO_TARBALL_URL and GO_SHA256 below fetch go1.27.1 when `go` is missing). CUDA_ARCHS (default 80,86,89,120: the
|
||||
# 12 GB tier is sm_86 (3060) and sm_89 (4070), 16 GB sm_89 and sm_120, the 24 GB 3090 sm_86, the 4090 sm_89, the 5090
|
||||
# sm_120; the stock server lists 80, 86, 89, 90, 100, 120; sm_75 (Turing) is not in SP1's kernels, CUDA_ARCHS=75
|
||||
# fails in the sys crate's build). Writes <out>/sp1-gpu-server, <out>/sp1-gpu-server.sha256 and <out>/build.json
|
||||
# (the descriptor the signed manifest is made from: SP1 version and commit, the patch's sha256, the archs, the
|
||||
# toolchain, the time). Not byte-reproducible across machines: SP1's prover-types build script stamps the build time
|
||||
# into the binary, so the signed manifest names ONE build (CI's); a PC build is a behavioural cross-check.
|
||||
set -euo pipefail
|
||||
OUT="${1:?out dir}"; SRC="${2:-${TMPDIR:-/tmp}/igneum-sp1-6.8.1}"
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"; ROOT="$(cd "$HERE/../.." && pwd)"
|
||||
PATCH="$ROOT/proving/prover-floor/sp1-gpu-6.8.1-floor.patch"
|
||||
SP1_TAG="v6.8.1"; SP1_COMMIT_EXPECTED="c84ada1ed5911f28c4d3c9d0ed2f9e6cd7edb824"
|
||||
CUDA_ARCHS="${CUDA_ARCHS:-80,86,89,120}"
|
||||
GO_TARBALL_URL="https://go.dev/dl/go1.27.1.linux-amd64.tar.gz"; GO_SHA256="63d339f0da5ab53635a56f2490a7984dfe12dfcff22ad749f63edaf590168445"
|
||||
JOBS="${CARGO_JOBS:-$(nproc)}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
mkdir -p "$OUT"
|
||||
[ -f "$PATCH" ] || { echo "no patch at $PATCH" >&2; exit 2; }
|
||||
PATCH_SHA="$(sha256sum "$PATCH" | cut -c1-64)"
|
||||
CUDA_DIR="${CUDA_PATH:-$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)}"
|
||||
[ -n "$CUDA_DIR" ] && export CUDA_PATH="$CUDA_DIR" CUDACXX="$CUDA_DIR/bin/nvcc" PATH="$CUDA_DIR/bin:$PATH" LD_LIBRARY_PATH="$CUDA_DIR/lib64:${LD_LIBRARY_PATH:-}"
|
||||
command -v nvcc >/dev/null || { echo "no nvcc on the PATH (CUDA 12.8 toolkit)" >&2; exit 2; }
|
||||
if ! command -v go >/dev/null; then
|
||||
GODIR="${GO_DIR:-$OUT/go-toolchain}"
|
||||
if [ ! -x "$GODIR/go/bin/go" ]; then
|
||||
echo "fetching Go ($GO_TARBALL_URL)"
|
||||
mkdir -p "$GODIR"; curl -sSL -o "$GODIR/go.tgz" "$GO_TARBALL_URL"
|
||||
echo "$GO_SHA256 $GODIR/go.tgz" | sha256sum -c - >/dev/null || { echo "Go tarball sha256 mismatch" >&2; exit 2; }
|
||||
tar -xzf "$GODIR/go.tgz" -C "$GODIR" && rm -f "$GODIR/go.tgz"
|
||||
fi
|
||||
export PATH="$GODIR/go/bin:$PATH" GOPATH="$GODIR/gopath" GOCACHE="$GODIR/gocache" GOFLAGS=-mod=mod
|
||||
fi
|
||||
echo "toolchain: nvcc $(nvcc --version | grep -o 'release [0-9.]*'), $(cargo --version), $(go version), protoc $(protoc --version 2>/dev/null || echo MISSING), cmake $(cmake --version | head -1 || echo MISSING), jobs $JOBS, CUDA_ARCHS $CUDA_ARCHS"
|
||||
if [ ! -d "$SRC/.git" ]; then git clone -q --depth 1 --branch "$SP1_TAG" https://github.com/succinctlabs/sp1 "$SRC"; fi
|
||||
cd "$SRC"
|
||||
git checkout -q -- . && git clean -qfd sp1-gpu/crates >/dev/null 2>&1 || true
|
||||
SP1_COMMIT="$(git rev-parse HEAD)"
|
||||
[ "$SP1_COMMIT" = "$SP1_COMMIT_EXPECTED" ] || { echo "SP1 $SP1_TAG is $SP1_COMMIT, expected $SP1_COMMIT_EXPECTED" >&2; exit 2; }
|
||||
git apply "$PATCH"
|
||||
# the patched files are re-stamped (cargo rebuilds by mtime; the clone may be older than the target dir)
|
||||
git diff --name-only | xargs touch
|
||||
echo "patched: $(git diff --stat | tail -1) (sha256 $PATCH_SHA)"
|
||||
export CUDA_ARCHS
|
||||
T0=$(date +%s)
|
||||
cargo build --release --bin sp1-gpu-server -j "$JOBS"
|
||||
BIN="${CARGO_TARGET_DIR:-$SRC/target}/release/sp1-gpu-server"
|
||||
cp "$BIN" "$OUT/sp1-gpu-server"; chmod +x "$OUT/sp1-gpu-server"
|
||||
SHA="$(sha256sum "$OUT/sp1-gpu-server" | cut -c1-64)"; BYTES="$(stat -c %s "$OUT/sp1-gpu-server")"
|
||||
echo "$SHA sp1-gpu-server" > "$OUT/sp1-gpu-server.sha256"
|
||||
VERSION="$("$OUT/sp1-gpu-server" --version 2>/dev/null || echo unknown)"
|
||||
ARCHS_IN="$(cuobjdump --list-elf "$OUT/sp1-gpu-server" 2>/dev/null | grep -o 'sm_[0-9]*' | sort -u | tr '\n' ' ' | sed 's/ $//')"
|
||||
cat > "$OUT/build.json" <<JSON
|
||||
{"sp1_version":"$VERSION","sp1_tag":"$SP1_TAG","sp1_commit":"$SP1_COMMIT","patch_sha256":"$PATCH_SHA","cuda_archs":"$CUDA_ARCHS","elf_targets":"$ARCHS_IN","sha256":"$SHA","bytes":$BYTES,"built_at":"$(stamp)","build_seconds":$(( $(date +%s) - T0 )),"nvcc":"$(nvcc --version | grep -o 'release [0-9.]*')","host":"$(uname -srm)"}
|
||||
JSON
|
||||
echo "built: sp1-gpu-server $BYTES bytes, sha256 $SHA, version $VERSION, targets $ARCHS_IN, $(( $(date +%s) - T0 )) s"
|
||||
26
packaging/prover/fetch-server.sh
Executable file
26
packaging/prover/fetch-server.sh
Executable file
|
|
@ -0,0 +1,26 @@
|
|||
#!/usr/bin/env bash
|
||||
# Fetches the published GPU prover server (packaging/prover/push-server.sh) into a folder for the payload: the
|
||||
# manifest, its signature and the binary from the downloads host, the signature checked with the key the apps carry
|
||||
# (igneum-ota-sign verify-server embedded) and the binary's sha256 and size checked against the manifest.
|
||||
#
|
||||
# packaging/prover/fetch-server.sh [out dir, default proving/prover-floor/server] [--signer path]
|
||||
#
|
||||
# A host without the three files (404) is reported and leaves the folder empty: the payload then ships no server and
|
||||
# the app runs SP1's stock one (24 GB cards). A manifest that is there but does not verify FAILS the fetch.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"; ROOT="$(cd "$HERE/../.." && pwd)"
|
||||
OUT="$ROOT/proving/prover-floor/server"; SIGNER=""
|
||||
while [ $# -gt 0 ]; do case "$1" in --signer) SIGNER="$2"; shift 2 ;; *) OUT="$1"; shift ;; esac; done
|
||||
TOKEN="${DL_TOKEN:-$(tr -d '[:space:]' < "$HOME/.config/igneum/dl-token")}"
|
||||
BASE="https://dl.igneum.network/dl/$TOKEN"
|
||||
[ -n "$SIGNER" ] || SIGNER="$ROOT/app/igneum-app/target/release/igneum-ota-sign"
|
||||
[ -x "$SIGNER" ] || [ -x "$SIGNER.exe" ] || { echo "no igneum-ota-sign at $SIGNER (cargo build --release --bin igneum-ota-sign)" >&2; exit 2; }
|
||||
mkdir -p "$OUT"
|
||||
if ! curl -fsSL --retry 3 -o "$OUT/prover-server.json" "$BASE/prover-server.json"; then
|
||||
echo "no prover-server.json on the downloads host: the payload ships no GPU server (the app uses SP1's stock one, 24 GB cards)"; rm -f "$OUT/prover-server.json"; exit 0
|
||||
fi
|
||||
curl -fsSL --retry 3 -o "$OUT/prover-server.json.sig" "$BASE/prover-server.json.sig"
|
||||
curl -fsSL --retry 3 -o "$OUT/sp1-gpu-server" "$BASE/sp1-gpu-server"
|
||||
"$SIGNER" verify-server embedded "$OUT/prover-server.json" "$OUT/prover-server.json.sig" --binary "$OUT/sp1-gpu-server"
|
||||
chmod +x "$OUT/sp1-gpu-server"
|
||||
echo "fetched into $OUT: $(ls -la "$OUT" | tail -n +2 | awk '{print $NF, $5}' | tr '\n' ' ')"
|
||||
63
packaging/prover/push-server.sh
Executable file
63
packaging/prover/push-server.sh
Executable file
|
|
@ -0,0 +1,63 @@
|
|||
#!/usr/bin/env bash
|
||||
# Publishes the project's GPU prover server (prover floor, 6 October 2026): the binary CI built
|
||||
# (.github/workflows/prover-server.yml, artifact `sp1-gpu-server-floor`) or a given build folder, its signed manifest
|
||||
# prover-server.json (format app/igneum-app/src/proverserver.rs: the SP1 version and commit it patches, the patch's
|
||||
# sha256, the CUDA targets, the binary's sha256 and size, the build) and prover-server.json.sig, the detached Ed25519
|
||||
# signature made on this Mac with the OTA key the apps already trust, into the downloads folder (dl/<token>/), and
|
||||
# deploys it. The Windows build (.github/workflows/windows.yml) and packaging/prover/fetch-server.sh fetch the three
|
||||
# files from there and verify them before the payload carries them (wsl2\bin\sp1-gpu-server, wsl2\prover-server.json
|
||||
# and .sig); the app verifies them again before use.
|
||||
#
|
||||
# packaging/prover/push-server.sh --run <ci run id> the CI artifact (gh run download; gh must be logged in as igneum-labs)
|
||||
# packaging/prover/push-server.sh --dir <folder> a folder with sp1-gpu-server, sp1-gpu-server.sha256 and build.json
|
||||
# [--no-deploy]
|
||||
#
|
||||
# Same secrets layout as push-inputs.sh: ~/.config/igneum/{dl-token, dlsite-dir, vercel, ota-signing-key(.pub)}.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"; ROOT="$(cd "$HERE/../.." && pwd)"
|
||||
RUN=""; DIR=""; DEPLOY=1
|
||||
while [ $# -gt 0 ]; do case "$1" in --run) RUN="$2"; shift 2 ;; --dir) DIR="$2"; shift 2 ;; --no-deploy) DEPLOY=0; shift ;; *) echo "unknown argument $1" >&2; exit 2 ;; esac; done
|
||||
[ -n "$RUN" ] || [ -n "$DIR" ] || { echo "usage: push-server.sh --run <ci run id> | --dir <folder> [--no-deploy]" >&2; exit 2; }
|
||||
TOKEN="$(tr -d '[:space:]' < "$HOME/.config/igneum/dl-token")"
|
||||
DLSITE="${IGNEUM_DLSITE:-}"; [ -n "$DLSITE" ] || DLSITE="$(tr -d '[:space:]' < "$HOME/.config/igneum/dlsite-dir")"
|
||||
[ -d "$DLSITE/dl/$TOKEN" ] || { echo "no downloads folder at $DLSITE/dl/<token>" >&2; exit 1; }
|
||||
KEY="$HOME/.config/igneum/ota-signing-key"; PUB="$HOME/.config/igneum/ota-signing-key.pub"
|
||||
SIGNER="$ROOT/app/igneum-app/target/release/igneum-ota-sign"
|
||||
if [ ! -x "$SIGNER" ]; then echo "building igneum-ota-sign"; (cd "$ROOT/app/igneum-app" && nice -n 19 cargo build --release -j 4 --bin igneum-ota-sign --quiet); fi
|
||||
if [ -n "$RUN" ]; then
|
||||
DIR="$(mktemp -d)/server"; mkdir -p "$DIR"
|
||||
gh auth status 2>/dev/null | grep -q "igneum-labs" || echo "warning: gh's active account is not igneum-labs (gh auth switch --user igneum-labs)"
|
||||
gh run download "$RUN" --repo igneum-network/igneum --name sp1-gpu-server-floor --dir "$DIR"
|
||||
BUILT_BY="ci:$RUN"
|
||||
else
|
||||
BUILT_BY="dir:$(hostname -s)"
|
||||
fi
|
||||
BIN="$DIR/sp1-gpu-server"; [ -f "$BIN" ] || { echo "no sp1-gpu-server in $DIR" >&2; exit 1; }
|
||||
[ -f "$DIR/build.json" ] || { echo "no build.json in $DIR (build-server.sh writes it)" >&2; exit 1; }
|
||||
# the hash the builder recorded must be the file's hash now (a corrupt download is refused here)
|
||||
SHA="$(shasum -a 256 "$BIN" | cut -c1-64)"; BYTES="$(stat -f %z "$BIN")"
|
||||
REC="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1]))["sha256"])' "$DIR/build.json")"
|
||||
[ "$SHA" = "$REC" ] || { echo "the binary's sha256 $SHA is not build.json's $REC" >&2; exit 1; }
|
||||
[ -f "$DIR/sp1-gpu-server.sha256" ] && { grep -q "^$SHA " "$DIR/sp1-gpu-server.sha256" || { echo "sp1-gpu-server.sha256 disagrees" >&2; exit 1; }; }
|
||||
DEST="$DLSITE/dl/$TOKEN"
|
||||
python3 - "$DIR/build.json" "$DEST/prover-server.json" "$BUILT_BY" "$SHA" "$BYTES" <<'PY'
|
||||
import json, sys, datetime
|
||||
b = json.load(open(sys.argv[1]))
|
||||
m = {"format": "igneum-prover-server/1", "sp1_version": b["sp1_version"], "sp1_commit": b["sp1_commit"], "patch_sha256": b["patch_sha256"],
|
||||
"cuda_archs": b.get("elf_targets") or b["cuda_archs"], "built_at": b["built_at"], "built_by": sys.argv[3],
|
||||
"published_at": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
||||
"files": {"sp1-gpu-server": {"sha256": sys.argv[4], "bytes": int(sys.argv[5])}}}
|
||||
open(sys.argv[2], "w").write(json.dumps(m, sort_keys=True, separators=(",", ":")) + "\n")
|
||||
PY
|
||||
"$SIGNER" sign-server "$KEY" "$DEST/prover-server.json" > "$DEST/prover-server.json.sig"
|
||||
cp "$BIN" "$DEST/sp1-gpu-server"
|
||||
"$SIGNER" verify-server "$PUB" "$DEST/prover-server.json" "$DEST/prover-server.json.sig" --binary "$DEST/sp1-gpu-server"
|
||||
"$SIGNER" verify-server embedded "$DEST/prover-server.json" "$DEST/prover-server.json.sig" >/dev/null || { echo "the embedded key does not verify this signature: the OTA key on this Mac is not the one the apps carry" >&2; exit 1; }
|
||||
cat "$DEST/prover-server.json"; echo "signature: $(cut -c1-16 "$DEST/prover-server.json.sig")..."
|
||||
if [ "$DEPLOY" = 1 ]; then
|
||||
echo "deploying $DLSITE"
|
||||
(cd "$DLSITE" && npx vercel@latest --global-config "$HOME/.config/igneum/vercel" deploy --prod --yes 2>&1 | grep -v "$TOKEN" || true)
|
||||
echo "published: https://dl.igneum.network/dl/<token>/{sp1-gpu-server,prover-server.json,prover-server.json.sig}"
|
||||
else
|
||||
echo "not deployed (--no-deploy); run the Vercel deploy from $DLSITE when ready"
|
||||
fi
|
||||
|
|
@ -97,6 +97,20 @@ if [ -f "$PROVE_LINUX/igneum-prove-host" ] && [ -f "$PROVE_LINUX/igneum-prove-ex
|
|||
cp "$PROVE_LINUX/igneum-prove-host" "$PROVE_LINUX/igneum-prove-export" "$STAGE/wsl2/bin/"
|
||||
echo "prover (WSL2): igneum-prove-host $(stat -f %z "$PROVE_LINUX/igneum-prove-host") bytes, igneum-prove-export $(stat -f %z "$PROVE_LINUX/igneum-prove-export") bytes"
|
||||
else echo "warning: no Linux igneum-prove-host/igneum-prove-export in $PROVE_LINUX (cargo zigbuild --target x86_64-unknown-linux-gnu.2.36 --features igneum-prove-host/cuda); the Proving tile will ask for the WSL2 setup, which builds them"; fi
|
||||
# prover floor (6 October 2026, app/igneum-app/src/proverserver.rs): the project's build of SP1's GPU server, which
|
||||
# lets 12 GB and 16 GB cards prove (the stock one refuses them). The three files come from packaging/prover/
|
||||
# fetch-server.sh (the downloads host, published by push-server.sh; IGNEUM_PROVER_SERVER overrides the folder) and
|
||||
# are verified here with the key the apps carry before they go in; the app verifies them again before use. Without
|
||||
# them the app runs SP1's stock server and the 24 GB gate. The binary is 167 MB (about 60 MB zipped).
|
||||
SERVER_DIR="${IGNEUM_PROVER_SERVER:-$ROOT/proving/prover-floor/server}"
|
||||
SIGNER_BIN="${IGNEUM_OTA_SIGN:-$ROOT/app/igneum-app/target/release/igneum-ota-sign}"
|
||||
if [ -f "$SERVER_DIR/sp1-gpu-server" ] && [ -f "$SERVER_DIR/prover-server.json" ] && [ -f "$SERVER_DIR/prover-server.json.sig" ]; then
|
||||
if [ -x "$SIGNER_BIN" ] || [ -x "$SIGNER_BIN.exe" ]; then
|
||||
"$SIGNER_BIN" verify-server embedded "$SERVER_DIR/prover-server.json" "$SERVER_DIR/prover-server.json.sig" --binary "$SERVER_DIR/sp1-gpu-server" || { echo "the prover server in $SERVER_DIR does not verify; not shipping it"; exit 1; }
|
||||
else echo "warning: no igneum-ota-sign at $SIGNER_BIN; the server goes in unverified here (the app verifies it before use)"; fi
|
||||
cp "$SERVER_DIR/sp1-gpu-server" "$STAGE/wsl2/bin/" && cp "$SERVER_DIR/prover-server.json" "$SERVER_DIR/prover-server.json.sig" "$STAGE/wsl2/"
|
||||
echo "prover server (WSL2): sp1-gpu-server $(stat -f %z "$SERVER_DIR/sp1-gpu-server" 2>/dev/null || stat -c %s "$SERVER_DIR/sp1-gpu-server") bytes, sha256 $(python3 -c 'import json,sys; print(json.load(open(sys.argv[1]))["files"]["sp1-gpu-server"]["sha256"][:16])' "$SERVER_DIR/prover-server.json")..., signed manifest wsl2/prover-server.json"
|
||||
else echo "warning: no verified prover server in $SERVER_DIR (packaging/prover/fetch-server.sh); the app will run SP1's stock server, which refuses cards under 24 GB"; fi
|
||||
cp "$ROOT"/proving/windows-wsl2/*.sh "$ROOT"/proving/windows-wsl2/*.ps1 "$ROOT"/proving/windows-wsl2/*.bat "$ROOT/proving/windows-wsl2/README.txt" "$STAGE/wsl2/"
|
||||
cp "$ROOT"/proving/fixtures/*.json "$STAGE/wsl2/fixtures/"
|
||||
|
||||
|
|
|
|||
|
|
@ -264,10 +264,13 @@ fn run() -> Result<()> {
|
|||
|
||||
let r = match mode.as_str() {
|
||||
"execute" => run_execute(&sp1, &shards, block.env.parent_hash, &mut results),
|
||||
"shard" => run_shard(&sp1, &shards, shard_index, out_path.as_deref(), &mut results),
|
||||
"shard" => run_shard(&sp1, &shards, shard_index, out_path.as_deref(), &mut results, false),
|
||||
// Prover floor, route 2 (6 October 2026): the core proof alone, verified, no compression: what a small card
|
||||
// would hand to an aggregator. The peak of this run is the core-only prover's floor.
|
||||
"core" => run_shard(&sp1, &shards, shard_index, out_path.as_deref(), &mut results, true),
|
||||
"compressed" => run_compressed(&sp1, &shards, shard_index, out_path.as_deref(), &mut results),
|
||||
"block" => run_block(&sp1, &shards, &claim, out_path.as_deref(), &mut results),
|
||||
"all" => run_shard(&sp1, &shards, shard_index, out_path.as_deref(), &mut results).and_then(|_| run_block(&sp1, &shards, &claim, out_path.as_deref(), &mut results)),
|
||||
"all" => run_shard(&sp1, &shards, shard_index, out_path.as_deref(), &mut results, false).and_then(|_| run_block(&sp1, &shards, &claim, out_path.as_deref(), &mut results)),
|
||||
other => Err(anyhow!("unknown mode {other}")),
|
||||
};
|
||||
// The proof system (and with it the CUDA client) is dropped here, inside the runtime guard of `main`.
|
||||
|
|
@ -381,7 +384,7 @@ fn save_proof<T: serde::Serialize>(proof: &T, path: std::path::PathBuf) {
|
|||
}
|
||||
|
||||
|
||||
fn run_shard(sp1: &Sp1ProofSystem, shards: &[BuiltShard], index: usize, out_dir: Option<&str>, results: &mut serde_json::Map<String, serde_json::Value>) -> Result<()> {
|
||||
fn run_shard(sp1: &Sp1ProofSystem, shards: &[BuiltShard], index: usize, out_dir: Option<&str>, results: &mut serde_json::Map<String, serde_json::Value>, core_only: bool) -> Result<()> {
|
||||
let s = shards.get(index).ok_or_else(|| anyhow!("shard {index} is not in the plan ({} shards)", shards.len()))?;
|
||||
let i = s.output.shard_index;
|
||||
results.insert("shard_index".into(), i.into());
|
||||
|
|
@ -408,6 +411,10 @@ fn run_shard(sp1: &Sp1ProofSystem, shards: &[BuiltShard], index: usize, out_dir:
|
|||
if let Some(dir) = out_dir.and_then(|p| std::path::Path::new(p).parent()) {
|
||||
save_proof(&core, dir.join(format!("block-{}-shard-{i}-core.bin", s.input.env.number)));
|
||||
}
|
||||
if core_only {
|
||||
println!("RESULT core only: shard {i} stops after the core proof ({bytes} bytes); the aggregator compresses it");
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
stage(&format!("compressed shard {i}"));
|
||||
let proof = sp1.prove_shard(&ShardWitness { input: s.input.clone() })?;
|
||||
|
|
|
|||
|
|
@ -171,10 +171,26 @@ pub struct Sp1SegmentProof {
|
|||
pub struct Sp1WrappedProof(pub SP1ProofWithPublicValues);
|
||||
|
||||
impl Sp1ProofSystem {
|
||||
/// The SDK's prover from the environment, with one addition for machines that hold more than one card
|
||||
/// (prover floor, 6 October 2026): `IGNEUM_CUDA_DEVICE=<n>` picks the CUDA device the proof runs on when
|
||||
/// `SP1_PROVER=cuda`. The SDK's own `from_env` always takes device 0 and the server it spawns is given
|
||||
/// `CUDA_VISIBLE_DEVICES=<n>` by the client, so the host's environment cannot steer it any other way.
|
||||
fn client_from_env() -> sp1_sdk::blocking::EnvProver {
|
||||
#[cfg(feature = "cuda")]
|
||||
{
|
||||
let cuda = std::env::var("SP1_PROVER").map(|p| p.to_lowercase() == "cuda").unwrap_or(false);
|
||||
if let (true, Some(id)) = (cuda, std::env::var("IGNEUM_CUDA_DEVICE").ok().and_then(|s| s.parse::<u32>().ok())) {
|
||||
println!("RESULT cuda device: {id} (IGNEUM_CUDA_DEVICE)");
|
||||
return sp1_sdk::blocking::EnvProver::Cuda(ProverClient::builder().cuda().with_device_id(id).build());
|
||||
}
|
||||
}
|
||||
ProverClient::from_env()
|
||||
}
|
||||
|
||||
/// Builds the prover from the environment (`SP1_PROVER` = cpu, cuda or mock) and runs both key setups.
|
||||
pub fn from_env(shard_elf: Elf, agg_elf: Elf) -> Result<Self> {
|
||||
let t = Instant::now();
|
||||
let client = ProverClient::from_env();
|
||||
let client = Self::client_from_env();
|
||||
let client_dt = t.elapsed();
|
||||
let t = Instant::now();
|
||||
let shard_pk = client.setup(shard_elf)?;
|
||||
|
|
|
|||
|
|
@ -1,5 +1,26 @@
|
|||
diff --git a/sp1-gpu/crates/cuda/src/task.rs b/sp1-gpu/crates/cuda/src/task.rs
|
||||
index a503a86..813016b 100644
|
||||
--- a/sp1-gpu/crates/cuda/src/task.rs
|
||||
+++ b/sp1-gpu/crates/cuda/src/task.rs
|
||||
@@ -149,7 +149,15 @@ pub enum GlobalTaskPoolBuildError {
|
||||
|
||||
impl TaskPoolBuilder {
|
||||
pub fn new() -> Self {
|
||||
- Self { capacity: None, device: CudaDevice(0), mem_release_threshold: u64::MAX }
|
||||
+ // Igneum prover-floor patch: upstream keeps every freed device allocation in the pool for the process's
|
||||
+ // life (threshold u64::MAX), so the prover holds its high-water mark between shards on a card it shares
|
||||
+ // with a miner. `SP1_GPU_MEM_RELEASE_THRESHOLD=<bytes>` sets the pool's release threshold (0 returns
|
||||
+ // freed memory to the driver at once); unset, upstream's behaviour.
|
||||
+ let mem_release_threshold = std::env::var("SP1_GPU_MEM_RELEASE_THRESHOLD")
|
||||
+ .ok()
|
||||
+ .and_then(|s| s.parse::<u64>().ok())
|
||||
+ .unwrap_or(u64::MAX);
|
||||
+ Self { capacity: None, device: CudaDevice(0), mem_release_threshold }
|
||||
}
|
||||
|
||||
pub fn num_tasks(mut self, num_tasks: usize) -> Self {
|
||||
diff --git a/sp1-gpu/crates/jagged_tracegen/src/lib.rs b/sp1-gpu/crates/jagged_tracegen/src/lib.rs
|
||||
index 579f70a..09e73e8 100644
|
||||
index 579f70a..2264044 100644
|
||||
--- a/sp1-gpu/crates/jagged_tracegen/src/lib.rs
|
||||
+++ b/sp1-gpu/crates/jagged_tracegen/src/lib.rs
|
||||
@@ -481,6 +481,33 @@ async fn device_preprocessed_tracegen<A: CudaTracegenAir<Felt>>(
|
||||
|
|
@ -64,7 +85,81 @@ index 579f70a..09e73e8 100644
|
|||
log_stacking_height,
|
||||
max_log_row_count,
|
||||
backend,
|
||||
@@ -984,9 +1021,15 @@ pub async fn full_tracegen<A: CudaTracegenAir<Felt>>(
|
||||
@@ -906,6 +943,11 @@ pub async fn main_tracegen<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
|
||||
|
||||
log_chip_stats(machine, &chip_set, &traces);
|
||||
|
||||
+ // Igneum prover-floor patch: the key's buffer is sized to its preprocessed traces at setup (upstream sized it
|
||||
+ // for a whole shard), so grow it here to what this shard needs before the main traces are appended: a bigger
|
||||
+ // dense buffer and column index, the preprocessed region copied device to device, swapped into the key.
|
||||
+ grow_for_main(&mut jagged_traces.preprocessed_traces, &traces, log_stacking_height, backend);
|
||||
+
|
||||
copy_main_jagged_traces(
|
||||
traces,
|
||||
&mut jagged_traces.preprocessed_traces,
|
||||
@@ -918,6 +960,61 @@ pub async fn main_tracegen<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
|
||||
(public_values, chip_set, permit)
|
||||
}
|
||||
|
||||
+/// Igneum prover-floor patch: see `main_tracegen`. The need is the preprocessed phase as laid out (its padded
|
||||
+/// end, `preprocessed_offset`) plus the main traces padded to the stacking height plus one stacking height of
|
||||
+/// slack; a buffer at least that big is left alone. The process aborts, loudly, if the copy cannot be made,
|
||||
+/// because a panic inside a prover task is what left sweep 2 hanging on the client's socket.
|
||||
+fn grow_for_main(
|
||||
+ jagged: &mut JaggedTraceMle<Felt, TaskScope>,
|
||||
+ main_traces: &BTreeMap<String, Trace<TaskScope>>,
|
||||
+ log_stacking_height: u32,
|
||||
+ backend: &TaskScope,
|
||||
+) {
|
||||
+ let pre_end = jagged.dense().preprocessed_offset;
|
||||
+ let needed = pre_end
|
||||
+ + padded_trace_elements(main_traces, log_stacking_height)
|
||||
+ + (1 << log_stacking_height);
|
||||
+ let have = jagged.dense().dense.capacity();
|
||||
+ if have >= needed {
|
||||
+ return;
|
||||
+ }
|
||||
+ let mut new_dense: Buffer<Felt, TaskScope> = Buffer::with_capacity_in(needed, backend.clone());
|
||||
+ let mut new_col_index: Buffer<u32, TaskScope> =
|
||||
+ Buffer::with_capacity_in(needed >> 1, backend.clone());
|
||||
+ unsafe {
|
||||
+ new_dense.assume_init();
|
||||
+ new_col_index.assume_init();
|
||||
+ }
|
||||
+ {
|
||||
+ let JaggedMle { dense_data, col_index, .. } = &mut **jagged;
|
||||
+ let src_dense: &Slice<_, _> = &dense_data.dense[..pre_end];
|
||||
+ let dst_dense: &mut Slice<_, _> = &mut new_dense[..pre_end];
|
||||
+ let src_col: &Slice<_, _> = &col_index[..pre_end >> 1];
|
||||
+ let dst_col: &mut Slice<_, _> = &mut new_col_index[..pre_end >> 1];
|
||||
+ unsafe {
|
||||
+ if dst_dense.copy_from_slice(src_dense, backend).is_err()
|
||||
+ || dst_col.copy_from_slice(src_col, backend).is_err()
|
||||
+ {
|
||||
+ eprintln!("FLOOR grow FAILED: could not copy the preprocessed region ({pre_end} elements) into the grown buffer ({needed} elements); aborting instead of hanging");
|
||||
+ std::process::abort();
|
||||
+ }
|
||||
+ }
|
||||
+ }
|
||||
+ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() {
|
||||
+ eprintln!(
|
||||
+ "FLOOR grow key buffer {have} -> {needed} elements (preprocessed {pre_end}, {} bytes)",
|
||||
+ needed * 6
|
||||
+ );
|
||||
+ }
|
||||
+ let JaggedMle { dense_data, col_index, .. } = &mut **jagged;
|
||||
+ dense_data.dense = new_dense;
|
||||
+ *col_index = new_col_index;
|
||||
+ unsafe {
|
||||
+ dense_data.dense.set_len(pre_end);
|
||||
+ col_index.set_len(pre_end >> 1);
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub async fn main_tracegen_permit<GC: IopCtx<F = Felt>, A: CudaTracegenAir<Felt>>(
|
||||
machine: &Machine<Felt, A>,
|
||||
@@ -984,9 +1081,15 @@ pub async fn full_tracegen<A: CudaTracegenAir<Felt>>(
|
||||
|
||||
log_chip_stats(machine, &chip_set, &main_traces);
|
||||
|
||||
|
|
@ -81,7 +176,7 @@ index 579f70a..09e73e8 100644
|
|||
log_stacking_height,
|
||||
max_log_row_count,
|
||||
backend,
|
||||
@@ -1002,6 +1045,18 @@ pub async fn full_tracegen<A: CudaTracegenAir<Felt>>(
|
||||
@@ -1002,6 +1105,18 @@ pub async fn full_tracegen<A: CudaTracegenAir<Felt>>(
|
||||
)
|
||||
.await;
|
||||
|
||||
|
|
@ -202,6 +297,86 @@ index 5dccd9d..574d4fa 100644
|
|||
.await,
|
||||
recursion_verifier,
|
||||
)
|
||||
diff --git a/sp1-gpu/crates/server/src/main.rs b/sp1-gpu/crates/server/src/main.rs
|
||||
index 65e94f5..498357c 100644
|
||||
--- a/sp1-gpu/crates/server/src/main.rs
|
||||
+++ b/sp1-gpu/crates/server/src/main.rs
|
||||
@@ -17,9 +17,55 @@ struct Args {
|
||||
version: bool,
|
||||
}
|
||||
|
||||
+/// Igneum prover-floor patch (6 October 2026, the GPU fleet's finding): a panic inside a prover task (an allocation
|
||||
+/// the card cannot meet, `cudaMallocAsync` failing and `Buffer::with_capacity_in` panicking in a tokio worker) left
|
||||
+/// the request's future waiting for ever, the client on its socket, the card at 0% for the 15 minutes until someone
|
||||
+/// killed it (the 8 GB and 10 GB cards at threshold 2^27). The server must fail the shard instead: this hook names
|
||||
+/// the stage from the panic's location and exits, so the client's proof fails at once with the server's last line.
|
||||
+fn stage_of(location: &str) -> &'static str {
|
||||
+ let l = location.to_ascii_lowercase();
|
||||
+ if l.contains("jagged_tracegen") || l.contains("/tracegen") {
|
||||
+ "trace generation"
|
||||
+ } else if l.contains("commit") || l.contains("basefold") || l.contains("merkle") {
|
||||
+ "the commit (codewords and Merkle trees)"
|
||||
+ } else if l.contains("logup_gkr") {
|
||||
+ "LogUp GKR"
|
||||
+ } else if l.contains("zerocheck") {
|
||||
+ "the zerocheck"
|
||||
+ } else if l.contains("jagged") {
|
||||
+ "the jagged sumcheck"
|
||||
+ } else if l.contains("prover_components") || l.contains("recursion") || l.contains("sp1-prover") || l.contains("sp1_prover") {
|
||||
+ "the recursion (compression)"
|
||||
+ } else if l.contains("cuda") || l.contains("slop") {
|
||||
+ "a device allocation"
|
||||
+ } else {
|
||||
+ "the prover"
|
||||
+ }
|
||||
+}
|
||||
+
|
||||
+fn install_abort_on_panic() {
|
||||
+ std::panic::set_hook(Box::new(|info| {
|
||||
+ let location = info.location().map(|l| format!("{}:{}", l.file(), l.line())).unwrap_or_else(|| "unknown".into());
|
||||
+ let message = info
|
||||
+ .payload()
|
||||
+ .downcast_ref::<&str>()
|
||||
+ .map(|s| s.to_string())
|
||||
+ .or_else(|| info.payload().downcast_ref::<String>().cloned())
|
||||
+ .unwrap_or_default();
|
||||
+ let oom = message.to_ascii_lowercase().contains("alloc") || message.contains("MemoryAllocation") || message.contains("OUT_OF_MEMORY");
|
||||
+ eprintln!(
|
||||
+ "FLOOR abort: {} failed at {location}: {message}{}; the server exits so the client's proof fails instead of waiting",
|
||||
+ stage_of(&location),
|
||||
+ if oom { " (the card's memory could not meet an allocation: lower SP1_GPU_ELEMENT_THRESHOLD one notch)" } else { "" }
|
||||
+ );
|
||||
+ std::process::exit(70);
|
||||
+ }));
|
||||
+}
|
||||
+
|
||||
#[tokio::main]
|
||||
#[allow(clippy::print_stdout)]
|
||||
async fn main() {
|
||||
+ install_abort_on_panic();
|
||||
tracing_subscriber::fmt::init();
|
||||
|
||||
let args = Args::parse();
|
||||
@@ -40,3 +86,19 @@ async fn main() {
|
||||
eprintln!("Error running server: {e}");
|
||||
}
|
||||
}
|
||||
+
|
||||
+#[cfg(test)]
|
||||
+mod floor_tests {
|
||||
+ use super::stage_of;
|
||||
+
|
||||
+ #[test]
|
||||
+ fn the_stage_is_named_from_the_panic_location() {
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/jagged_tracegen/src/lib.rs:240"), "trace generation");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/basefold/src/fri.rs:97"), "the commit (codewords and Merkle trees)");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/logup_gkr/src/tracegen.rs:72"), "LogUp GKR");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/zerocheck/src/prover.rs:1163"), "the zerocheck");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/prover_components/src/builder.rs:70"), "the recursion (compression)");
|
||||
+ assert_eq!(stage_of("sp1-gpu/crates/cuda/src/stream.rs:330"), "a device allocation");
|
||||
+ assert_eq!(stage_of("somewhere/else.rs:1"), "the prover");
|
||||
+ }
|
||||
+}
|
||||
diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs
|
||||
index 4035f1f..0d0d907 100644
|
||||
--- a/sp1-gpu/crates/server/src/server.rs
|
||||
|
|
|
|||
|
|
@ -39,7 +39,7 @@ cd "\$SRC"
|
|||
git checkout -q -- . && git clean -qfd sp1-gpu/crates >/dev/null 2>&1
|
||||
echo "RESULT source \$(git describe --tags --always) \$(git rev-parse HEAD)"
|
||||
git apply "\$PATCH" || { echo "RESULT build_failed patch does not apply"; exit 2; }
|
||||
touch sp1-gpu/crates/prover_components/src/builder.rs sp1-gpu/crates/jagged_tracegen/src/lib.rs sp1-gpu/crates/server/src/server.rs
|
||||
git diff --name-only | xargs touch
|
||||
echo "RESULT patched \$(git diff --stat | tail -1)"
|
||||
# Go, as the release workflow installs it (the server's native-gnark feature compiles the gnark library with go;
|
||||
# the wrap path is never run by a compressed proof, but the feature is upstream's and stays): a pinned tarball
|
||||
|
|
|
|||
106
tools/prover-floor/pc-12gb-card-miner.ps1
Normal file
106
tools/prover-floor/pc-12gb-card-miner.ps1
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
# Prover floor, route 1 (6 October 2026, the project lead: a real 12 GB card lands today; can it mine AND prove?). The same
|
||||
# fixtures as sweeps 3 and 4 on the card itself: `--mode shard` (the core proof, its verify, then the compressed
|
||||
# proof, in one run; the 1-s sampler split at the core RESULT gives the core-only peak) at thresholds 2^26, 2^27
|
||||
# and 2^25 (the core-only mine-and-prove profile), the v1 shard and an empty block. Two phases, each its own job: FLOOR_PHASE=alone (publish WITH
|
||||
# --stop-miners; measures the card's own idle first) and FLOOR_PHASE=miner (publish WITHOUT --stop-miners; measures
|
||||
# the miner's working set on the card first). Works on PC 1 or PC 2: the card is found by its memory (under
|
||||
# 13,000 MiB; FLOOR_CARD_INDEX overrides) and every proof runs on it through IGNEUM_CUDA_DEVICE (the host passes
|
||||
# the index to the SDK, which starts the server with CUDA_VISIBLE_DEVICES=<index>; CUDA_DEVICE_ORDER=PCI_BUS_ID
|
||||
# keeps nvidia-smi's and CUDA's numbering the same). The kit: /opt/igneum-floor/home/.sp1/bin/sp1-gpu-server (the
|
||||
# v3 build, pc2-build-server.ps1 on this PC) and the host built from the fetched package igneum-prove-wsl2-floor
|
||||
# (jobs\floor-pc1-kit\, the fetch job floor-pc1-kit with --extract); each is tested first and named if missing. Never touches the
|
||||
# app's /opt/igneum host or /root/.sp1 server; leaves the prover ON; the runner restores the miners.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
$phase = 'miner' # the beside-the-miner phase: publish WITHOUT --stop-miners
|
||||
"RESULT start $(Stamp) phase=$phase machine=$env:COMPUTERNAME"
|
||||
"RESULT cards $((& nvidia-smi --query-gpu=index,name,memory.total,memory.used,pci.bus_id --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
# the 12 GB card: the first index whose memory.total is under 13,000 MiB (a 3060 12 GB reads 12,288; a 4070 12,282)
|
||||
$cards = & nvidia-smi --query-gpu=index,memory.total --format=csv,noheader,nounits 2>$null | ForEach-Object { $p = $_ -split ',\s*'; [pscustomobject]@{ index = [int]$p[0]; total = [int]$p[1] } }
|
||||
$card = if ($env:FLOOR_CARD_INDEX) { [int]$env:FLOOR_CARD_INDEX } else { ($cards | Where-Object { $_.total -lt 13000 } | Select-Object -First 1).index }
|
||||
if ($null -eq $card) { "RESULT measure_failed no card under 13,000 MiB on this PC (set FLOOR_CARD_INDEX to force one)"; "RESULT end $(Stamp)"; exit 2 }
|
||||
"RESULT card index=$card total_mib=$(($cards | Where-Object { $_.index -eq $card }).total)"
|
||||
"RESULT prover_off $(Stamp) $(Prove $false)"
|
||||
Start-Sleep -Seconds 30
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-card' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
# the kit (the wiped-jobs-folder class): the fetched package under the jobs folder, tested before use
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
# the app stores a fetch under the FETCH JOB's id (jobs\floor-pc1-kit\, 11:32Z: the --to name became the file's name), so
|
||||
# the kit folder is the fetch id; FLOOR_KIT_DIR names another (a fetch to PC 2 would have its own id)
|
||||
$kit = Join-Path $jobs $(if ($env:FLOOR_KIT_DIR) { $env:FLOOR_KIT_DIR } else { 'floor-pc1-kit' })
|
||||
$kitOk = Test-Path (Join-Path $kit 'igneum-prove-wsl2\package\proving\igneum-prove\Cargo.toml')
|
||||
"RESULT kit $(if ($kitOk) { "present $kit" } else { "MISSING: publish the fetch job of igneum-prove-wsl2-floor.zip to this machine (--dir jobs --extract, the job id names the folder) first" })"
|
||||
$kitW = if ($kitOk) { WslPath (Join-Path $kit 'igneum-prove-wsl2\package') } else { '/nonexistent' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH" CUDA_DEVICE_ORDER=PCI_BUS_ID
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
# a point that runs past POINT_BUDGET_S (default 180 s) is killed by `timeout` INSIDE the script, so the tail (the
|
||||
# prover back on, the sockets unlinked) always runs; the job's cap stays the outer guard (6 October 2026, hang case 2)
|
||||
JOB='JOBW_PLACEHOLDER'; KIT='KITW_PLACEHOLDER'; CARD=CARD_PLACEHOLDER; PHASE='PHASE_PLACEHOLDER'
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server; H=/opt/igneum-floor/host/igneum-prove-host
|
||||
echo "RESULT wsl_cards $(nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader,nounits 2>/dev/null | tr '\n' ';')"
|
||||
[ -x "$SRV" ] || { echo "RESULT measure_failed no patched server at $SRV: run the build job (tools/prover-floor/pc2-build-server.ps1) on this PC first (25 to 45 min cold)"; exit 2; }
|
||||
echo "RESULT patched_server sha256=$(sha256sum $SRV | cut -c1-16) version=$($SRV --version 2>/dev/null)"
|
||||
if [ ! -x "$H" ]; then
|
||||
[ -f "$KIT/proving/igneum-prove/Cargo.toml" ] || { echo "RESULT measure_failed no host and no kit: fetch igneum-prove-wsl2-floor.zip first"; exit 2; }
|
||||
echo "STAGE host build $(stamp)"
|
||||
mkdir -p /opt/igneum-floor/host-src && rsync -a "$KIT/" /opt/igneum-floor/host-src/ && find /opt/igneum-floor/host-src -type f -exec touch {} +
|
||||
TD=/root/igneum-prove/proving/igneum-prove/target; [ -d "$TD" ] || TD=/opt/igneum-floor/host-target
|
||||
( cd /opt/igneum-floor/host-src/proving/igneum-prove && CARGO_TARGET_DIR=$TD nice -n 19 cargo build --release -p igneum-prove-host --features igneum-prove-host/cuda > $JOB/host-build.log 2>&1 ) || { echo "RESULT measure_failed host build; tail:"; tail -n 30 $JOB/host-build.log; exit 2; }
|
||||
mkdir -p /opt/igneum-floor/host && cp $TD/release/igneum-prove-host /opt/igneum-floor/host/ && echo "RESULT host built sha256=$(sha256sum $H | cut -c1-16)"
|
||||
fi
|
||||
FX="/opt/igneum-floor/host-src/proving/fixtures"; [ -d "$FX" ] || FX="$KIT/proving/fixtures"
|
||||
V1="$FX/fees-v1-shards2.json"; EMPTY="$FX/block-72854-empty-block-first.json"
|
||||
echo "RESULT host_ids $($H --mode id 2>/dev/null | tr '\n' ' ' | cut -c1-200)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
smi() { nvidia-smi -i $CARD --query-gpu=memory.used --format=csv,noheader,nounits | head -1; }
|
||||
echo "RESULT card_before phase=$PHASE used_mib=$(smi) (alone: the card's own idle; miner: idle + the miner's working set on this card)"
|
||||
runshard() { # name fixture env...
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi -i $CARD --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local n=$(wc -l < "$csv")
|
||||
local core=$(grep -E "^RESULT core shard" "$log" | tail -1); local comp=$(grep -E "^RESULT compressed shard" "$log" | tail -1)
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local core_at=$(echo "$core" | sed -E 's/.* at ([0-9T:.-]+Z?)$/\1/'); local core_epoch=$(date -u -d "${core_at}" +%s 2>/dev/null || echo 0)
|
||||
local split=$(( core_epoch - t0 )); [ $split -lt 1 ] && split=$n
|
||||
local corepeak=$(awk -F', *' -v s=$split 'NR<=s+1 { if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local cl=$(echo "$core" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/core_s=\1 core_bytes=\2 core_verify_s=\3 core_\4/')
|
||||
local pl=$(echo "$comp" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/compressed_s=\1 compressed_bytes=\2 compressed_verify_s=\3 compressed_\4/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT card12 phase=$PHASE cfg=$name fixture=$(basename $fx .json) card=$CARD core_peak_mib=$corepeak full_peak_mib=$peak split_s=$split samples=$n wall_s=$wall cycles=${cyc:-na} ${cl:-no_core_result} ${pl:-no_compressed_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR (memory|opts|grow)|^RESULT cuda device" "$log" | sed "s/^/RESULT floorline phase=$PHASE cfg=$name fixture=$(basename $fx .json) /" | head -10
|
||||
}
|
||||
runshard e26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
runshard e26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
runshard e27 "$V1" SP1_GPU_ELEMENT_THRESHOLD=134217728
|
||||
runshard e27 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=134217728
|
||||
runshard e25 "$V1" SP1_GPU_ELEMENT_THRESHOLD=33554432
|
||||
runshard e25 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=33554432
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT card_after phase=$PHASE used_mib=$(smi)"
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('JOBW_PLACEHOLDER', $jobW).Replace('KITW_PLACEHOLDER', $kitW).Replace('CARD_PLACEHOLDER', "$card").Replace('PHASE_PLACEHOLDER', $phase)
|
||||
$bashFile = Join-Path $job 'card12.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT prover_on $(Stamp) $(Prove $true)"
|
||||
"RESULT end $(Stamp)"
|
||||
106
tools/prover-floor/pc-12gb-card.ps1
Normal file
106
tools/prover-floor/pc-12gb-card.ps1
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
# Prover floor, route 1 (6 October 2026, the project lead: a real 12 GB card lands today; can it mine AND prove?). The same
|
||||
# fixtures as sweeps 3 and 4 on the card itself: `--mode shard` (the core proof, its verify, then the compressed
|
||||
# proof, in one run; the 1-s sampler split at the core RESULT gives the core-only peak) at thresholds 2^26, 2^27
|
||||
# and 2^25 (the core-only mine-and-prove profile), the v1 shard and an empty block. Two phases, each its own job: FLOOR_PHASE=alone (publish WITH
|
||||
# --stop-miners; measures the card's own idle first) and FLOOR_PHASE=miner (publish WITHOUT --stop-miners; measures
|
||||
# the miner's working set on the card first). Works on PC 1 or PC 2: the card is found by its memory (under
|
||||
# 13,000 MiB; FLOOR_CARD_INDEX overrides) and every proof runs on it through IGNEUM_CUDA_DEVICE (the host passes
|
||||
# the index to the SDK, which starts the server with CUDA_VISIBLE_DEVICES=<index>; CUDA_DEVICE_ORDER=PCI_BUS_ID
|
||||
# keeps nvidia-smi's and CUDA's numbering the same). The kit: /opt/igneum-floor/home/.sp1/bin/sp1-gpu-server (the
|
||||
# v3 build, pc2-build-server.ps1 on this PC) and the host built from the fetched package igneum-prove-wsl2-floor
|
||||
# (jobs\floor-pc1-kit\, the fetch job floor-pc1-kit with --extract); each is tested first and named if missing. Never touches the
|
||||
# app's /opt/igneum host or /root/.sp1 server; leaves the prover ON; the runner restores the miners.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
$phase = if ($env:FLOOR_PHASE) { $env:FLOOR_PHASE } else { 'alone' }
|
||||
"RESULT start $(Stamp) phase=$phase machine=$env:COMPUTERNAME"
|
||||
"RESULT cards $((& nvidia-smi --query-gpu=index,name,memory.total,memory.used,pci.bus_id --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
# the 12 GB card: the first index whose memory.total is under 13,000 MiB (a 3060 12 GB reads 12,288; a 4070 12,282)
|
||||
$cards = & nvidia-smi --query-gpu=index,memory.total --format=csv,noheader,nounits 2>$null | ForEach-Object { $p = $_ -split ',\s*'; [pscustomobject]@{ index = [int]$p[0]; total = [int]$p[1] } }
|
||||
$card = if ($env:FLOOR_CARD_INDEX) { [int]$env:FLOOR_CARD_INDEX } else { ($cards | Where-Object { $_.total -lt 13000 } | Select-Object -First 1).index }
|
||||
if ($null -eq $card) { "RESULT measure_failed no card under 13,000 MiB on this PC (set FLOOR_CARD_INDEX to force one)"; "RESULT end $(Stamp)"; exit 2 }
|
||||
"RESULT card index=$card total_mib=$(($cards | Where-Object { $_.index -eq $card }).total)"
|
||||
"RESULT prover_off $(Stamp) $(Prove $false)"
|
||||
Start-Sleep -Seconds 30
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-card' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
# the kit (the wiped-jobs-folder class): the fetched package under the jobs folder, tested before use
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
# the app stores a fetch under the FETCH JOB's id (jobs\floor-pc1-kit\, 11:32Z: the --to name became the file's name), so
|
||||
# the kit folder is the fetch id; FLOOR_KIT_DIR names another (a fetch to PC 2 would have its own id)
|
||||
$kit = Join-Path $jobs $(if ($env:FLOOR_KIT_DIR) { $env:FLOOR_KIT_DIR } else { 'floor-pc1-kit' })
|
||||
$kitOk = Test-Path (Join-Path $kit 'igneum-prove-wsl2\package\proving\igneum-prove\Cargo.toml')
|
||||
"RESULT kit $(if ($kitOk) { "present $kit" } else { "MISSING: publish the fetch job of igneum-prove-wsl2-floor.zip to this machine (--dir jobs --extract, the job id names the folder) first" })"
|
||||
$kitW = if ($kitOk) { WslPath (Join-Path $kit 'igneum-prove-wsl2\package') } else { '/nonexistent' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH" CUDA_DEVICE_ORDER=PCI_BUS_ID
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
# a point that runs past POINT_BUDGET_S (default 180 s) is killed by `timeout` INSIDE the script, so the tail (the
|
||||
# prover back on, the sockets unlinked) always runs; the job's cap stays the outer guard (6 October 2026, hang case 2)
|
||||
JOB='JOBW_PLACEHOLDER'; KIT='KITW_PLACEHOLDER'; CARD=CARD_PLACEHOLDER; PHASE='PHASE_PLACEHOLDER'
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server; H=/opt/igneum-floor/host/igneum-prove-host
|
||||
echo "RESULT wsl_cards $(nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader,nounits 2>/dev/null | tr '\n' ';')"
|
||||
[ -x "$SRV" ] || { echo "RESULT measure_failed no patched server at $SRV: run the build job (tools/prover-floor/pc2-build-server.ps1) on this PC first (25 to 45 min cold)"; exit 2; }
|
||||
echo "RESULT patched_server sha256=$(sha256sum $SRV | cut -c1-16) version=$($SRV --version 2>/dev/null)"
|
||||
if [ ! -x "$H" ]; then
|
||||
[ -f "$KIT/proving/igneum-prove/Cargo.toml" ] || { echo "RESULT measure_failed no host and no kit: fetch igneum-prove-wsl2-floor.zip first"; exit 2; }
|
||||
echo "STAGE host build $(stamp)"
|
||||
mkdir -p /opt/igneum-floor/host-src && rsync -a "$KIT/" /opt/igneum-floor/host-src/ && find /opt/igneum-floor/host-src -type f -exec touch {} +
|
||||
TD=/root/igneum-prove/proving/igneum-prove/target; [ -d "$TD" ] || TD=/opt/igneum-floor/host-target
|
||||
( cd /opt/igneum-floor/host-src/proving/igneum-prove && CARGO_TARGET_DIR=$TD nice -n 19 cargo build --release -p igneum-prove-host --features igneum-prove-host/cuda > $JOB/host-build.log 2>&1 ) || { echo "RESULT measure_failed host build; tail:"; tail -n 30 $JOB/host-build.log; exit 2; }
|
||||
mkdir -p /opt/igneum-floor/host && cp $TD/release/igneum-prove-host /opt/igneum-floor/host/ && echo "RESULT host built sha256=$(sha256sum $H | cut -c1-16)"
|
||||
fi
|
||||
FX="/opt/igneum-floor/host-src/proving/fixtures"; [ -d "$FX" ] || FX="$KIT/proving/fixtures"
|
||||
V1="$FX/fees-v1-shards2.json"; EMPTY="$FX/block-72854-empty-block-first.json"
|
||||
echo "RESULT host_ids $($H --mode id 2>/dev/null | tr '\n' ' ' | cut -c1-200)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
smi() { nvidia-smi -i $CARD --query-gpu=memory.used --format=csv,noheader,nounits | head -1; }
|
||||
echo "RESULT card_before phase=$PHASE used_mib=$(smi) (alone: the card's own idle; miner: idle + the miner's working set on this card)"
|
||||
runshard() { # name fixture env...
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi -i $CARD --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local n=$(wc -l < "$csv")
|
||||
local core=$(grep -E "^RESULT core shard" "$log" | tail -1); local comp=$(grep -E "^RESULT compressed shard" "$log" | tail -1)
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local core_at=$(echo "$core" | sed -E 's/.* at ([0-9T:.-]+Z?)$/\1/'); local core_epoch=$(date -u -d "${core_at}" +%s 2>/dev/null || echo 0)
|
||||
local split=$(( core_epoch - t0 )); [ $split -lt 1 ] && split=$n
|
||||
local corepeak=$(awk -F', *' -v s=$split 'NR<=s+1 { if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local cl=$(echo "$core" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/core_s=\1 core_bytes=\2 core_verify_s=\3 core_\4/')
|
||||
local pl=$(echo "$comp" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/compressed_s=\1 compressed_bytes=\2 compressed_verify_s=\3 compressed_\4/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT card12 phase=$PHASE cfg=$name fixture=$(basename $fx .json) card=$CARD core_peak_mib=$corepeak full_peak_mib=$peak split_s=$split samples=$n wall_s=$wall cycles=${cyc:-na} ${cl:-no_core_result} ${pl:-no_compressed_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR (memory|opts|grow)|^RESULT cuda device" "$log" | sed "s/^/RESULT floorline phase=$PHASE cfg=$name fixture=$(basename $fx .json) /" | head -10
|
||||
}
|
||||
runshard e26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
runshard e26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
runshard e27 "$V1" SP1_GPU_ELEMENT_THRESHOLD=134217728
|
||||
runshard e27 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=134217728
|
||||
runshard e25 "$V1" SP1_GPU_ELEMENT_THRESHOLD=33554432
|
||||
runshard e25 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=33554432
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT card_after phase=$PHASE used_mib=$(smi)"
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('JOBW_PLACEHOLDER', $jobW).Replace('KITW_PLACEHOLDER', $kitW).Replace('CARD_PLACEHOLDER', "$card").Replace('PHASE_PLACEHOLDER', $phase)
|
||||
$bashFile = Join-Path $job 'card12.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT prover_on $(Stamp) $(Prove $true)"
|
||||
"RESULT end $(Stamp)"
|
||||
34
tools/prover-floor/pc1-hangcase-restore.ps1
Normal file
34
tools/prover-floor/pc1-hangcase-restore.ps1
Normal file
|
|
@ -0,0 +1,34 @@
|
|||
# Prover floor: after floor-pc1-hangcase-2 hit its cap (13:15:30Z) the script's tail never ran: the prover stays off
|
||||
# and a host or server may still hold the 4070. This reads the point's log and samples first (the diagnosis), kills
|
||||
# the leftovers inside WSL2, unlinks the sockets, and switches the prover on. The app and the miners untouched
|
||||
# (the runner restored the miners at the cap).
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
"RESULT start $(Stamp)"
|
||||
"RESULT gpus_before $((& nvidia-smi --query-gpu=index,memory.used,utilization.gpu --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$prev = Join-Path $jobs 'floor-pc1-hangcase-2'
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$prevW = if (Test-Path $prev) { WslPath $prev } else { '/nonexistent' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
P='PREV_PLACEHOLDER'
|
||||
echo "RESULT procs_before $(pgrep -a 'sp1-gpu-server|igneum-prove-host|nvidia-smi' | tr '\n' ';' | cut -c1-300)"
|
||||
for f in "$P"/log-*.txt; do [ -f "$f" ] || continue; echo "RESULT log $(basename $f) lines=$(wc -l < "$f")"; grep -E "^(FLOOR|RESULT|Error|error|thread)" "$f" | tail -12 | sed "s/^/RESULT logline /" | cut -c1-300; echo "RESULT logtail $(tail -n 3 "$f" | tr '\n' '|' | cut -c1-400)"; done
|
||||
for f in "$P"/smi-*.csv; do [ -f "$f" ] || continue; echo "RESULT smi $(basename $f) samples=$(wc -l < "$f") peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$f") last3=$(tail -n 3 "$f" | awk -F', *' '{printf "%s/%s%% ", $2, $3}')"; done
|
||||
pkill -f igneum-prove-host 2>/dev/null; pkill -f sp1-gpu-server 2>/dev/null; pkill -f 'nvidia-smi --query-gpu=timestamp' 2>/dev/null; sleep 2; pkill -9 -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT procs_after $(pgrep -a 'sp1-gpu-server|igneum-prove-host' | tr '\n' ';' || echo none)"
|
||||
'@
|
||||
$bash = $bash.Replace('PREV_PLACEHOLDER', $prevW)
|
||||
$job = $env:IGNEUM_JOB_DIR; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
$bashFile = Join-Path $job 'restore.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT prover_on $(Stamp) $(Prove $true)"
|
||||
Start-Sleep -Seconds 10
|
||||
"RESULT gpus_after $((& nvidia-smi --query-gpu=index,memory.used,utilization.gpu --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
"RESULT end $(Stamp)"
|
||||
111
tools/prover-floor/pc1-hangcase.ps1
Normal file
111
tools/prover-floor/pc1-hangcase.ps1
Normal file
|
|
@ -0,0 +1,111 @@
|
|||
# Prover floor, route 1 (6 October 2026, the project lead: a real 12 GB card lands today; can it mine AND prove?). The same
|
||||
# fixtures as sweeps 3 and 4 on the card itself: `--mode shard` (the core proof, its verify, then the compressed
|
||||
# proof, in one run; the 1-s sampler split at the core RESULT gives the core-only peak) at thresholds 2^26, 2^27
|
||||
# and 2^25 (the core-only mine-and-prove profile), the v1 shard and an empty block. Two phases, each its own job: FLOOR_PHASE=alone (publish WITH
|
||||
# --stop-miners; measures the card's own idle first) and FLOOR_PHASE=miner (publish WITHOUT --stop-miners; measures
|
||||
# the miner's working set on the card first). Works on PC 1 or PC 2: the card is found by its memory (under
|
||||
# 13,000 MiB; FLOOR_CARD_INDEX overrides) and every proof runs on it through IGNEUM_CUDA_DEVICE (the host passes
|
||||
# the index to the SDK, which starts the server with CUDA_VISIBLE_DEVICES=<index>; CUDA_DEVICE_ORDER=PCI_BUS_ID
|
||||
# keeps nvidia-smi's and CUDA's numbering the same). The kit: /opt/igneum-floor/home/.sp1/bin/sp1-gpu-server (the
|
||||
# v3 build, pc2-build-server.ps1 on this PC) and the host built from the fetched package igneum-prove-wsl2-floor
|
||||
# (jobs\floor-pc1-kit\, the fetch job floor-pc1-kit with --extract); each is tested first and named if missing. Never touches the
|
||||
# app's /opt/igneum host or /root/.sp1 server; leaves the prover ON; the runner restores the miners.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
$phase = 'hangcase' # the known-failed case: one point that cannot fit, must FAIL in seconds on the v5 server (publish WITH --stop-miners)
|
||||
"RESULT start $(Stamp) phase=$phase machine=$env:COMPUTERNAME"
|
||||
"RESULT cards $((& nvidia-smi --query-gpu=index,name,memory.total,memory.used,pci.bus_id --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
# the 12 GB card: the first index whose memory.total is under 13,000 MiB (a 3060 12 GB reads 12,288; a 4070 12,282)
|
||||
$cards = & nvidia-smi --query-gpu=index,memory.total --format=csv,noheader,nounits 2>$null | ForEach-Object { $p = $_ -split ',\s*'; [pscustomobject]@{ index = [int]$p[0]; total = [int]$p[1] } }
|
||||
$card = if ($env:FLOOR_CARD_INDEX) { [int]$env:FLOOR_CARD_INDEX } else { ($cards | Where-Object { $_.total -lt 13000 } | Select-Object -First 1).index }
|
||||
if ($null -eq $card) { "RESULT measure_failed no card under 13,000 MiB on this PC (set FLOOR_CARD_INDEX to force one)"; "RESULT end $(Stamp)"; exit 2 }
|
||||
"RESULT card index=$card total_mib=$(($cards | Where-Object { $_.index -eq $card }).total)"
|
||||
"RESULT prover_off $(Stamp) $(Prove $false)"
|
||||
Start-Sleep -Seconds 30
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-card' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
# the kit (the wiped-jobs-folder class): the fetched package under the jobs folder, tested before use
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
# the app stores a fetch under the FETCH JOB's id (jobs\floor-pc1-kit\, 11:32Z: the --to name became the file's name), so
|
||||
# the kit folder is the fetch id; FLOOR_KIT_DIR names another (a fetch to PC 2 would have its own id)
|
||||
$kit = Join-Path $jobs $(if ($env:FLOOR_KIT_DIR) { $env:FLOOR_KIT_DIR } else { 'floor-pc1-kit' })
|
||||
$kitOk = Test-Path (Join-Path $kit 'igneum-prove-wsl2\package\proving\igneum-prove\Cargo.toml')
|
||||
"RESULT kit $(if ($kitOk) { "present $kit" } else { "MISSING: publish the fetch job of igneum-prove-wsl2-floor.zip to this machine (--dir jobs --extract, the job id names the folder) first" })"
|
||||
$kitW = if ($kitOk) { WslPath (Join-Path $kit 'igneum-prove-wsl2\package') } else { '/nonexistent' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH" CUDA_DEVICE_ORDER=PCI_BUS_ID
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
# a point that runs past POINT_BUDGET_S (default 180 s) is killed by `timeout` INSIDE the script, so the tail (the
|
||||
# prover back on, the sockets unlinked) always runs; the job's cap stays the outer guard (6 October 2026, hang case 2)
|
||||
JOB='JOBW_PLACEHOLDER'; KIT='KITW_PLACEHOLDER'; CARD=CARD_PLACEHOLDER; PHASE='PHASE_PLACEHOLDER'
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server; H=/opt/igneum-floor/host/igneum-prove-host
|
||||
echo "RESULT wsl_cards $(nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader,nounits 2>/dev/null | tr '\n' ';')"
|
||||
[ -x "$SRV" ] || { echo "RESULT measure_failed no patched server at $SRV: run the build job (tools/prover-floor/pc2-build-server.ps1) on this PC first (25 to 45 min cold)"; exit 2; }
|
||||
echo "RESULT patched_server sha256=$(sha256sum $SRV | cut -c1-16) version=$($SRV --version 2>/dev/null)"
|
||||
if [ ! -x "$H" ]; then
|
||||
[ -f "$KIT/proving/igneum-prove/Cargo.toml" ] || { echo "RESULT measure_failed no host and no kit: fetch igneum-prove-wsl2-floor.zip first"; exit 2; }
|
||||
echo "STAGE host build $(stamp)"
|
||||
mkdir -p /opt/igneum-floor/host-src && rsync -a "$KIT/" /opt/igneum-floor/host-src/ && find /opt/igneum-floor/host-src -type f -exec touch {} +
|
||||
TD=/root/igneum-prove/proving/igneum-prove/target; [ -d "$TD" ] || TD=/opt/igneum-floor/host-target
|
||||
( cd /opt/igneum-floor/host-src/proving/igneum-prove && CARGO_TARGET_DIR=$TD nice -n 19 cargo build --release -p igneum-prove-host --features igneum-prove-host/cuda > $JOB/host-build.log 2>&1 ) || { echo "RESULT measure_failed host build; tail:"; tail -n 30 $JOB/host-build.log; exit 2; }
|
||||
mkdir -p /opt/igneum-floor/host && cp $TD/release/igneum-prove-host /opt/igneum-floor/host/ && echo "RESULT host built sha256=$(sha256sum $H | cut -c1-16)"
|
||||
fi
|
||||
FX="/opt/igneum-floor/host-src/proving/fixtures"; [ -d "$FX" ] || FX="$KIT/proving/fixtures"
|
||||
V1="$FX/fees-v1-shards2.json"; EMPTY="$FX/block-72854-empty-block-first.json"
|
||||
echo "RESULT host_ids $($H --mode id 2>/dev/null | tr '\n' ' ' | cut -c1-200)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
smi() { nvidia-smi -i $CARD --query-gpu=memory.used --format=csv,noheader,nounits | head -1; }
|
||||
echo "RESULT card_before phase=$PHASE used_mib=$(smi) (alone: the card's own idle; miner: idle + the miner's working set on this card)"
|
||||
runshard() { # name fixture env...
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi -i $CARD --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda IGNEUM_CUDA_DEVICE=$CARD RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local n=$(wc -l < "$csv")
|
||||
local core=$(grep -E "^RESULT core shard" "$log" | tail -1); local comp=$(grep -E "^RESULT compressed shard" "$log" | tail -1)
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local core_at=$(echo "$core" | sed -E 's/.* at ([0-9T:.-]+Z?)$/\1/'); local core_epoch=$(date -u -d "${core_at}" +%s 2>/dev/null || echo 0)
|
||||
local split=$(( core_epoch - t0 )); [ $split -lt 1 ] && split=$n
|
||||
local corepeak=$(awk -F', *' -v s=$split 'NR<=s+1 { if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local cl=$(echo "$core" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/core_s=\1 core_bytes=\2 core_verify_s=\3 core_\4/')
|
||||
local pl=$(echo "$comp" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/compressed_s=\1 compressed_bytes=\2 compressed_verify_s=\3 compressed_\4/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT card12 phase=$PHASE cfg=$name fixture=$(basename $fx .json) card=$CARD core_peak_mib=$corepeak full_peak_mib=$peak split_s=$split samples=$n wall_s=$wall cycles=${cyc:-na} ${cl:-no_core_result} ${pl:-no_compressed_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR (memory|opts|grow)|^RESULT cuda device" "$log" | sed "s/^/RESULT floorline phase=$PHASE cfg=$name fixture=$(basename $fx .json) /" | head -10
|
||||
}
|
||||
# the known-failed case (6 October 2026, the GPU fleet's finding: a point that does not fit left the old server at
|
||||
# 0% for 15 minutes). The first try, 2^27 on the 60 M-cycle prototype shard, FIT the 4070 (10,785 MiB, 24.8 s, run
|
||||
# floor-pc1-hangcase 13:06Z), so the point is upstream's own threshold (402,653,184, the 32 GB tier) on that shard:
|
||||
# 28,307 MiB on the 5090, impossible on 12,282. The v5 server (the panic hook) must fail it within seconds and name
|
||||
# the stage; a timeout here (the job's cap) is the FAILED verdict for the gate.
|
||||
FULL="$FX/block-338-shard1.json"
|
||||
echo "RESULT server_sha256 $(sha256sum /opt/igneum-floor/home/.sp1/bin/sp1-gpu-server | cut -c1-64) (v5 expected)"
|
||||
t_case=$(date +%s)
|
||||
runshard efull "$FULL" SP1_GPU_ELEMENT_THRESHOLD=402653184
|
||||
echo "RESULT hangcase seconds_to_close=$(( $(date +%s) - t_case ))"
|
||||
grep -E "FLOOR abort|Error|error|CudaClientError|panicked" "$JOB/log-efull-block-338-shard1.txt" | head -4 | sed 's/^/RESULT hangline /' | cut -c1-400
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT card_after phase=$PHASE used_mib=$(smi)"
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('JOBW_PLACEHOLDER', $jobW).Replace('KITW_PLACEHOLDER', $kitW).Replace('CARD_PLACEHOLDER', "$card").Replace('PHASE_PLACEHOLDER', $phase)
|
||||
$bashFile = Join-Path $job 'card12.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT prover_on $(Stamp) $(Prove $true)"
|
||||
"RESULT end $(Stamp)"
|
||||
28
tools/prover-floor/pc1-wsl2-enable.ps1
Normal file
28
tools/prover-floor/pc1-wsl2-enable.ps1
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
# Prover floor, route 1, step 1 (6 October 2026, the project lead at 11:25Z: YES to WSL2 on PC 1, one administrator prompt plus
|
||||
# a reboot). Published --elevated: the UAC prompt is the job's first step. Enables the two Windows features WSL2
|
||||
# needs, no distribution, no restart inside the job (the project lead reboots on the coordinator's word). The RESULT line says
|
||||
# "reboot needed" when Windows asks for one. Touches nothing else: the installed app, the miners and the node run on.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
"RESULT start $(Stamp) machine=$env:COMPUTERNAME elevated=$(([Security.Principal.WindowsPrincipal][Security.Principal.WindowsIdentity]::GetCurrent()).IsInRole([Security.Principal.WindowsBuiltInRole]::Administrator))"
|
||||
"RESULT windows $((Get-CimInstance Win32_OperatingSystem).Caption) build $((Get-CimInstance Win32_OperatingSystem).BuildNumber)"
|
||||
"RESULT gpus $((& nvidia-smi --query-gpu=index,name,memory.total,memory.used --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
$features = @('Microsoft-Windows-Subsystem-Linux', 'VirtualMachinePlatform')
|
||||
$restart = $false
|
||||
foreach ($f in $features) {
|
||||
$before = (Get-WindowsOptionalFeature -Online -FeatureName $f -ErrorAction SilentlyContinue).State
|
||||
"RESULT feature $f before=$before"
|
||||
if ("$before" -eq 'Enabled') { continue }
|
||||
try {
|
||||
$r = Enable-WindowsOptionalFeature -Online -FeatureName $f -NoRestart -ErrorAction Stop
|
||||
$after = (Get-WindowsOptionalFeature -Online -FeatureName $f -ErrorAction SilentlyContinue).State
|
||||
"RESULT feature $f enable=ok after=$after restart_needed=$($r.RestartNeeded)"
|
||||
if ($r.RestartNeeded) { $restart = $true }
|
||||
} catch {
|
||||
"RESULT feature $f enable=FAILED $($_.Exception.Message)"
|
||||
}
|
||||
}
|
||||
$wslState = try { (& wsl.exe --status 2>&1 | Out-String) -replace "`0", '' } catch { "wsl.exe not callable: $_" }
|
||||
"RESULT wsl_status $(($wslState -split "`r?`n" | Where-Object { $_.Trim() } | Select-Object -First 3) -join ' | ')"
|
||||
if ($restart) { "RESULT reboot needed: the features are enabled and Windows asks for a restart before WSL2 can start" } else { "RESULT no reboot needed (the features were already enabled or Windows asked for none)" }
|
||||
"RESULT end $(Stamp)"
|
||||
File diff suppressed because one or more lines are too long
95
tools/prover-floor/pc2-floor-core-alone.ps1
Normal file
95
tools/prover-floor/pc2-floor-core-alone.ps1
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
# Route 2, core-only provers: --mode shard = core proof then compressed in one run; the sampler split at the core RESULT gives the core-only peak. Publish WITH --stop-miners (alone).
|
||||
# Prover floor (5 October 2026): the peak GPU memory and the time of one compressed shard proof through the PATCHED
|
||||
# sp1-gpu-server (/opt/igneum-floor/home/.sp1/bin, reached by HOME=/opt/igneum-floor/home: the SDK spawns the
|
||||
# server it finds under $HOME/.sp1/bin, sp1-cuda-6.8.1/src/server.rs) on PC 2's RTX 5090, the miners STOPPED by the
|
||||
# job (--stop-miners) and the live prover switched off for the run (its server would otherwise own the socket).
|
||||
# Every point: every server killed and its socket unlinked, a 1-s nvidia-smi sampler, one `--mode compressed
|
||||
# --shard 0` of the pv1 host (/opt/igneum-pv1, the UNPATCHED SDK and verifier: its VERIFIED is the unpatched
|
||||
# verifier's word on the patched server's proof), the peak, the time, the FLOOR lines the server prints.
|
||||
# The point list comes from the FLOOR_POINTS environment the job carries, else the default sweep below.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
"RESULT start $(Stamp) prover off for the run: $(Prove $false)"
|
||||
Start-Sleep -Seconds 45
|
||||
"RESULT gpus $(Stamp) $((& nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu,power.draw --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-measure' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
# The empty live shard is a kit left by the chain job (the wiped-jobs-folder class: an app update may clear it), so
|
||||
# its presence is tested first; when it is gone the EMPTY points report measure_failed rows instead of killing the run.
|
||||
$emptyPath = Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json'
|
||||
if (Test-Path $emptyPath) { $emptyW = WslPath $emptyPath; "RESULT empty_fixture present $emptyPath" } else { $emptyW = '/nonexistent/block-83616.json'; "RESULT empty_fixture MISSING (kit wiped): the EMPTY points will report measure_failed" }
|
||||
$points = if ($env:FLOOR_POINTS) { $env:FLOOR_POINTS } else { 'runshard ce26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864;runshard ce26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864;runshard ce27 "$V1" SP1_GPU_ELEMENT_THRESHOLD=134217728;runshard ce27 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=134217728' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
JOB='JOBW_PLACEHOLDER'; H=/opt/igneum-pv1/igneum-prove-host; FX="/root/igneum-prove-pv1/proving/fixtures"
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server
|
||||
EMPTY='EMPTY_PLACEHOLDER'; V1="$FX/fees-v1-shards2.json"; FULL="$FX/block-338-shard1.json"; ONE="$FX/block-56-transfers.json"
|
||||
[ -x "$SRV" ] || { echo "RESULT measure_failed no patched server at $SRV"; exit 2; }
|
||||
[ -x "$H" ] || { echo "RESULT measure_failed no pv1 host at $H"; exit 2; }
|
||||
echo "RESULT patched_server sha256=$(sha256sum $SRV | cut -c1-64) version=$($SRV --version 2>/dev/null) host=$(sha256sum $H | cut -c1-16)"
|
||||
echo "RESULT live_server sha256=$(sha256sum /root/.sp1/bin/sp1-gpu-server | cut -c1-16) untouched"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT idle_mib $(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits | head -1)"
|
||||
run() { # name fixture env...
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local n=$(wc -l < "$csv")
|
||||
local line=$(grep -E "^RESULT compressed shard" "$log" | tail -1 | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/prove_s=\1 bytes=\2 verify_s=\3 \4/')
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT floor cfg=$name fixture=$(basename $fx .json) peak_mib=$peak samples=$n wall_s=$wall cycles=${cyc:-na} ${line:-no_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -40
|
||||
}
|
||||
runshard() { # name fixture env... (--mode shard: core then compressed; the core-phase peak read from the sampler by time)
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local n=$(wc -l < "$csv")
|
||||
local core=$(grep -E "^RESULT core shard" "$log" | tail -1)
|
||||
local comp=$(grep -E "^RESULT compressed shard" "$log" | tail -1)
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local core_at=$(echo "$core" | sed -E 's/.* at ([0-9T:.-]+Z?)$/\1/'); local core_epoch=$(date -u -d "${core_at}" +%s 2>/dev/null || echo 0)
|
||||
local split=$(( core_epoch - t0 )); [ $split -lt 1 ] && split=$n
|
||||
local corepeak=$(awk -F', *' -v s=$split 'NR<=s+1 { if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local cl=$(echo "$core" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/core_s=\1 core_bytes=\2 core_verify_s=\3 core_\4/')
|
||||
local pl=$(echo "$comp" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/compressed_s=\1 compressed_bytes=\2 compressed_verify_s=\3 compressed_\4/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT core cfg=$name fixture=$(basename $fx .json) core_peak_mib=$corepeak full_peak_mib=$peak split_s=$split samples=$n wall_s=$wall cycles=${cyc:-na} ${cl:-no_core_result} ${pl:-no_compressed_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR (memory|opts|grow)" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -12
|
||||
}
|
||||
POINTS_PLACEHOLDER_BASH
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('JOBW_PLACEHOLDER', $jobW).Replace('EMPTY_PLACEHOLDER', $emptyW).Replace('POINTS_PLACEHOLDER_BASH', ($points -replace ';', "`n"))
|
||||
$bashFile = Join-Path $job 'measure.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT end $(Stamp) prover back on: $(Prove $true)"
|
||||
95
tools/prover-floor/pc2-floor-core-miner.ps1
Normal file
95
tools/prover-floor/pc2-floor-core-miner.ps1
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
# Route 2, core-only provers, beside the miner: publish WITHOUT --stop-miners.
|
||||
# Prover floor (5 October 2026): the peak GPU memory and the time of one compressed shard proof through the PATCHED
|
||||
# sp1-gpu-server (/opt/igneum-floor/home/.sp1/bin, reached by HOME=/opt/igneum-floor/home: the SDK spawns the
|
||||
# server it finds under $HOME/.sp1/bin, sp1-cuda-6.8.1/src/server.rs) on PC 2's RTX 5090, the miners STOPPED by the
|
||||
# job (--stop-miners) and the live prover switched off for the run (its server would otherwise own the socket).
|
||||
# Every point: every server killed and its socket unlinked, a 1-s nvidia-smi sampler, one `--mode compressed
|
||||
# --shard 0` of the pv1 host (/opt/igneum-pv1, the UNPATCHED SDK and verifier: its VERIFIED is the unpatched
|
||||
# verifier's word on the patched server's proof), the peak, the time, the FLOOR lines the server prints.
|
||||
# The point list comes from the FLOOR_POINTS environment the job carries, else the default sweep below.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
"RESULT start $(Stamp) prover off for the run: $(Prove $false)"
|
||||
Start-Sleep -Seconds 45
|
||||
"RESULT gpus $(Stamp) $((& nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu,power.draw --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-measure' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
# The empty live shard is a kit left by the chain job (the wiped-jobs-folder class: an app update may clear it), so
|
||||
# its presence is tested first; when it is gone the EMPTY points report measure_failed rows instead of killing the run.
|
||||
$emptyPath = Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json'
|
||||
if (Test-Path $emptyPath) { $emptyW = WslPath $emptyPath; "RESULT empty_fixture present $emptyPath" } else { $emptyW = '/nonexistent/block-83616.json'; "RESULT empty_fixture MISSING (kit wiped): the EMPTY points will report measure_failed" }
|
||||
$points = if ($env:FLOOR_POINTS) { $env:FLOOR_POINTS } else { 'runshard ce26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864;runshard ce26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864;runshard ce27 "$V1" SP1_GPU_ELEMENT_THRESHOLD=134217728;runshard ce27 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=134217728' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
JOB='JOBW_PLACEHOLDER'; H=/opt/igneum-pv1/igneum-prove-host; FX="/root/igneum-prove-pv1/proving/fixtures"
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server
|
||||
EMPTY='EMPTY_PLACEHOLDER'; V1="$FX/fees-v1-shards2.json"; FULL="$FX/block-338-shard1.json"; ONE="$FX/block-56-transfers.json"
|
||||
[ -x "$SRV" ] || { echo "RESULT measure_failed no patched server at $SRV"; exit 2; }
|
||||
[ -x "$H" ] || { echo "RESULT measure_failed no pv1 host at $H"; exit 2; }
|
||||
echo "RESULT patched_server sha256=$(sha256sum $SRV | cut -c1-64) version=$($SRV --version 2>/dev/null) host=$(sha256sum $H | cut -c1-16)"
|
||||
echo "RESULT live_server sha256=$(sha256sum /root/.sp1/bin/sp1-gpu-server | cut -c1-16) untouched"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT idle_mib $(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits | head -1)"
|
||||
run() { # name fixture env...
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local n=$(wc -l < "$csv")
|
||||
local line=$(grep -E "^RESULT compressed shard" "$log" | tail -1 | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/prove_s=\1 bytes=\2 verify_s=\3 \4/')
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT floor cfg=$name fixture=$(basename $fx .json) peak_mib=$peak samples=$n wall_s=$wall cycles=${cyc:-na} ${line:-no_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -40
|
||||
}
|
||||
runshard() { # name fixture env... (--mode shard: core then compressed; the core-phase peak read from the sampler by time)
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local n=$(wc -l < "$csv")
|
||||
local core=$(grep -E "^RESULT core shard" "$log" | tail -1)
|
||||
local comp=$(grep -E "^RESULT compressed shard" "$log" | tail -1)
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local core_at=$(echo "$core" | sed -E 's/.* at ([0-9T:.-]+Z?)$/\1/'); local core_epoch=$(date -u -d "${core_at}" +%s 2>/dev/null || echo 0)
|
||||
local split=$(( core_epoch - t0 )); [ $split -lt 1 ] && split=$n
|
||||
local corepeak=$(awk -F', *' -v s=$split 'NR<=s+1 { if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local cl=$(echo "$core" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/core_s=\1 core_bytes=\2 core_verify_s=\3 core_\4/')
|
||||
local pl=$(echo "$comp" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/compressed_s=\1 compressed_bytes=\2 compressed_verify_s=\3 compressed_\4/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT core cfg=$name fixture=$(basename $fx .json) core_peak_mib=$corepeak full_peak_mib=$peak split_s=$split samples=$n wall_s=$wall cycles=${cyc:-na} ${cl:-no_core_result} ${pl:-no_compressed_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR (memory|opts|grow)" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -12
|
||||
}
|
||||
POINTS_PLACEHOLDER_BASH
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('JOBW_PLACEHOLDER', $jobW).Replace('EMPTY_PLACEHOLDER', $emptyW).Replace('POINTS_PLACEHOLDER_BASH', ($points -replace ';', "`n"))
|
||||
$bashFile = Join-Path $job 'measure.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT end $(Stamp) prover back on: $(Prove $true)"
|
||||
95
tools/prover-floor/pc2-floor-core2-miner.ps1
Normal file
95
tools/prover-floor/pc2-floor-core2-miner.ps1
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
# Route 2, second round, beside the miner (publish WITHOUT --stop-miners): core-only at 2^25 and 2^24, and the pool release threshold 0 (v4 server).
|
||||
# Prover floor (5 October 2026): the peak GPU memory and the time of one compressed shard proof through the PATCHED
|
||||
# sp1-gpu-server (/opt/igneum-floor/home/.sp1/bin, reached by HOME=/opt/igneum-floor/home: the SDK spawns the
|
||||
# server it finds under $HOME/.sp1/bin, sp1-cuda-6.8.1/src/server.rs) on PC 2's RTX 5090, the miners STOPPED by the
|
||||
# job (--stop-miners) and the live prover switched off for the run (its server would otherwise own the socket).
|
||||
# Every point: every server killed and its socket unlinked, a 1-s nvidia-smi sampler, one `--mode compressed
|
||||
# --shard 0` of the pv1 host (/opt/igneum-pv1, the UNPATCHED SDK and verifier: its VERIFIED is the unpatched
|
||||
# verifier's word on the patched server's proof), the peak, the time, the FLOOR lines the server prints.
|
||||
# The point list comes from the FLOOR_POINTS environment the job carries, else the default sweep below.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
"RESULT start $(Stamp) prover off for the run: $(Prove $false)"
|
||||
Start-Sleep -Seconds 45
|
||||
"RESULT gpus $(Stamp) $((& nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu,power.draw --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-measure' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
# The empty live shard is a kit left by the chain job (the wiped-jobs-folder class: an app update may clear it), so
|
||||
# its presence is tested first; when it is gone the EMPTY points report measure_failed rows instead of killing the run.
|
||||
$emptyPath = Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json'
|
||||
if (Test-Path $emptyPath) { $emptyW = WslPath $emptyPath; "RESULT empty_fixture present $emptyPath" } else { $emptyW = '/nonexistent/block-83616.json'; "RESULT empty_fixture MISSING (kit wiped): the EMPTY points will report measure_failed" }
|
||||
$points = if ($env:FLOOR_POINTS) { $env:FLOOR_POINTS } else { 'runshard ce25 "$V1" SP1_GPU_ELEMENT_THRESHOLD=33554432;runshard ce25 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=33554432;runshard ce24 "$V1" SP1_GPU_ELEMENT_THRESHOLD=16777216;runshard ce26r0 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864 SP1_GPU_MEM_RELEASE_THRESHOLD=0;runshard ce25r0 "$V1" SP1_GPU_ELEMENT_THRESHOLD=33554432 SP1_GPU_MEM_RELEASE_THRESHOLD=0' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
JOB='JOBW_PLACEHOLDER'; H=/opt/igneum-pv1/igneum-prove-host; FX="/root/igneum-prove-pv1/proving/fixtures"
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server
|
||||
EMPTY='EMPTY_PLACEHOLDER'; V1="$FX/fees-v1-shards2.json"; FULL="$FX/block-338-shard1.json"; ONE="$FX/block-56-transfers.json"
|
||||
[ -x "$SRV" ] || { echo "RESULT measure_failed no patched server at $SRV"; exit 2; }
|
||||
[ -x "$H" ] || { echo "RESULT measure_failed no pv1 host at $H"; exit 2; }
|
||||
echo "RESULT patched_server sha256=$(sha256sum $SRV | cut -c1-64) version=$($SRV --version 2>/dev/null) host=$(sha256sum $H | cut -c1-16)"
|
||||
echo "RESULT live_server sha256=$(sha256sum /root/.sp1/bin/sp1-gpu-server | cut -c1-16) untouched"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT idle_mib $(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits | head -1)"
|
||||
run() { # name fixture env...
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local n=$(wc -l < "$csv")
|
||||
local line=$(grep -E "^RESULT compressed shard" "$log" | tail -1 | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/prove_s=\1 bytes=\2 verify_s=\3 \4/')
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT floor cfg=$name fixture=$(basename $fx .json) peak_mib=$peak samples=$n wall_s=$wall cycles=${cyc:-na} ${line:-no_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -40
|
||||
}
|
||||
runshard() { # name fixture env... (--mode shard: core then compressed; the core-phase peak read from the sampler by time)
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local n=$(wc -l < "$csv")
|
||||
local core=$(grep -E "^RESULT core shard" "$log" | tail -1)
|
||||
local comp=$(grep -E "^RESULT compressed shard" "$log" | tail -1)
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local core_at=$(echo "$core" | sed -E 's/.* at ([0-9T:.-]+Z?)$/\1/'); local core_epoch=$(date -u -d "${core_at}" +%s 2>/dev/null || echo 0)
|
||||
local split=$(( core_epoch - t0 )); [ $split -lt 1 ] && split=$n
|
||||
local corepeak=$(awk -F', *' -v s=$split 'NR<=s+1 { if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local cl=$(echo "$core" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/core_s=\1 core_bytes=\2 core_verify_s=\3 core_\4/')
|
||||
local pl=$(echo "$comp" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/compressed_s=\1 compressed_bytes=\2 compressed_verify_s=\3 compressed_\4/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT core cfg=$name fixture=$(basename $fx .json) core_peak_mib=$corepeak full_peak_mib=$peak split_s=$split samples=$n wall_s=$wall cycles=${cyc:-na} ${cl:-no_core_result} ${pl:-no_compressed_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR (memory|opts|grow)" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -12
|
||||
}
|
||||
POINTS_PLACEHOLDER_BASH
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('JOBW_PLACEHOLDER', $jobW).Replace('EMPTY_PLACEHOLDER', $emptyW).Replace('POINTS_PLACEHOLDER_BASH', ($points -replace ';', "`n"))
|
||||
$bashFile = Join-Path $job 'measure.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT end $(Stamp) prover back on: $(Prove $true)"
|
||||
|
|
@ -18,15 +18,18 @@ Start-Sleep -Seconds 45
|
|||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-measure' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
$kitFile = Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json'
|
||||
if (-not (Test-Path $kitFile)) { Write-Output "RESULT kit missing: $kitFile; republish the fetch after any app update (C32)"; exit 2 }
|
||||
$emptyW = WslPath $kitFile
|
||||
# The empty live shard is a kit left by the chain job (the wiped-jobs-folder class: an app update may clear it), so
|
||||
# its presence is tested first; when it is gone the EMPTY points report measure_failed rows instead of killing the run.
|
||||
$emptyPath = Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json'
|
||||
if (Test-Path $emptyPath) { $emptyW = WslPath $emptyPath; "RESULT empty_fixture present $emptyPath" } else { $emptyW = '/nonexistent/block-83616.json'; "RESULT empty_fixture MISSING (kit wiped): the EMPTY points will report measure_failed" }
|
||||
$points = if ($env:FLOOR_POINTS) { $env:FLOOR_POINTS } else { 'POINTS_PLACEHOLDER' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
# a point that runs past POINT_BUDGET_S (default 180 s) is killed by `timeout` INSIDE the script, so the tail (the
|
||||
# prover back on, the sockets unlinked) always runs; the job's cap stays the outer guard (6 October 2026, hang case 2)
|
||||
JOB='JOBW_PLACEHOLDER'; H=/opt/igneum-pv1/igneum-prove-host; FX="/root/igneum-prove-pv1/proving/fixtures"
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server
|
||||
EMPTY='EMPTY_PLACEHOLDER'; V1="$FX/fees-v1-shards2.json"; FULL="$FX/block-338-shard1.json"; ONE="$FX/block-56-transfers.json"
|
||||
|
|
@ -44,7 +47,7 @@ run() { # name fixture env...
|
|||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
|
|
@ -56,6 +59,32 @@ run() { # name fixture env...
|
|||
echo "RESULT floor cfg=$name fixture=$(basename $fx .json) peak_mib=$peak samples=$n wall_s=$wall cycles=${cyc:-na} ${line:-no_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -40
|
||||
}
|
||||
runshard() { # name fixture env... (--mode shard: core then compressed; the core-phase peak read from the sampler by time)
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode shard --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local n=$(wc -l < "$csv")
|
||||
local core=$(grep -E "^RESULT core shard" "$log" | tail -1)
|
||||
local comp=$(grep -E "^RESULT compressed shard" "$log" | tail -1)
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local core_at=$(echo "$core" | sed -E 's/.* at ([0-9T:.-]+Z?)$/\1/'); local core_epoch=$(date -u -d "${core_at}" +%s 2>/dev/null || echo 0)
|
||||
local split=$(( core_epoch - t0 )); [ $split -lt 1 ] && split=$n
|
||||
local corepeak=$(awk -F', *' -v s=$split 'NR<=s+1 { if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local cl=$(echo "$core" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/core_s=\1 core_bytes=\2 core_verify_s=\3 core_\4/')
|
||||
local pl=$(echo "$comp" | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/compressed_s=\1 compressed_bytes=\2 compressed_verify_s=\3 compressed_\4/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT core cfg=$name fixture=$(basename $fx .json) core_peak_mib=$corepeak full_peak_mib=$peak split_s=$split samples=$n wall_s=$wall cycles=${cyc:-na} ${cl:-no_core_result} ${pl:-no_compressed_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR (memory|opts|grow)" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -12
|
||||
}
|
||||
POINTS_PLACEHOLDER_BASH
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
|
|
|
|||
37
tools/prover-floor/pc2-floor-restore.ps1
Normal file
37
tools/prover-floor/pc2-floor-restore.ps1
Normal file
|
|
@ -0,0 +1,37 @@
|
|||
# Prover floor: the restore after floor-sweep-2 hung on its first point and the 0.3.11 update killed the job tree
|
||||
# (the runner's finally block never ran). Reads what hung first (the point's log, the processes, the card), then
|
||||
# kills every leftover of the sweep inside WSL2 (host, patched server, sampler), unlinks the sockets, and switches
|
||||
# the live prover on. The live /root/.sp1/bin server is never touched. Miners: the app starts them itself.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
"RESULT start $(Stamp)"
|
||||
"RESULT gpus_before $(Stamp) $((& nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
$sweepDir = Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\floor-sweep-2'
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$sweepW = WslPath $sweepDir
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
SW='SWEEP_PLACEHOLDER'
|
||||
echo "RESULT procs_before $(pgrep -a 'sp1-gpu-server|igneum-prove-host|nvidia-smi' | tr '\n' ';' | cut -c1-400)"
|
||||
echo "RESULT sockets_before $(ls -l /tmp/sp1-cuda-*.sock 2>/dev/null | tr '\n' ';')"
|
||||
for f in "$SW"/log-*.txt; do [ -f "$f" ] || continue; echo "RESULT hung_log $(basename $f) lines=$(wc -l < "$f")"; grep -E "^(FLOOR|RESULT|thread|Error|error|panicked)" "$f" | head -30 | sed "s/^/RESULT hung_line /"; echo "RESULT hung_tail $(tail -n 5 "$f" | tr '\n' '|' | cut -c1-600)"; done
|
||||
for f in "$SW"/smi-*.csv; do [ -f "$f" ] || continue; echo "RESULT hung_smi $(basename $f) samples=$(wc -l < "$f") peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$f") last=$(tail -n 1 "$f")"; done
|
||||
pkill -f igneum-prove-host 2>/dev/null; pkill -f sp1-gpu-server 2>/dev/null; pkill -f 'nvidia-smi --query-gpu=timestamp' 2>/dev/null; sleep 2
|
||||
pkill -9 -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT procs_after $(pgrep -a 'sp1-gpu-server|igneum-prove-host' | tr '\n' ';' || echo none)"
|
||||
echo "RESULT live_server sha256=$(sha256sum /root/.sp1/bin/sp1-gpu-server | cut -c1-16) untouched"
|
||||
echo "RESULT restore_end $(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
||||
'@
|
||||
$bash = $bash.Replace('SWEEP_PLACEHOLDER', $sweepW)
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-restore' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
$bashFile = Join-Path $job 'restore.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT prover_on $(Stamp) $(Prove $true)"
|
||||
Start-Sleep -Seconds 20
|
||||
"RESULT gpus_after $(Stamp) $((& nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
"RESULT end $(Stamp)"
|
||||
|
|
@ -44,7 +44,7 @@ run() { # name fixture env...
|
|||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
|
|
|
|||
|
|
@ -44,7 +44,7 @@ run() { # name fixture env...
|
|||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
# Published WITHOUT --stop-miners: the beside-the-miner pair (the 5090 mining at full rate on the same card).
|
||||
# Prover floor (5 October 2026): the peak GPU memory and the time of one compressed shard proof through the PATCHED
|
||||
# sp1-gpu-server (/opt/igneum-floor/home/.sp1/bin, reached by HOME=/opt/igneum-floor/home: the SDK spawns the
|
||||
# server it finds under $HOME/.sp1/bin, sp1-cuda-6.8.1/src/server.rs) on PC 2's RTX 5090, the miners STOPPED by the
|
||||
|
|
@ -19,10 +18,8 @@ Start-Sleep -Seconds 45
|
|||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-measure' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
$kitFile = Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json'
|
||||
if (-not (Test-Path $kitFile)) { Write-Output "RESULT kit missing: $kitFile; republish the fetch after any app update (C32)"; exit 2 }
|
||||
$emptyW = WslPath $kitFile
|
||||
$points = if ($env:FLOOR_POINTS) { $env:FLOOR_POINTS } else { 'run m12 "$EMPTY" SP1_GPU_MEMORY_BUDGET_GB=12;run m12 "$V1" SP1_GPU_MEMORY_BUDGET_GB=12;run me26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864;run me26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864' }
|
||||
$emptyW = WslPath (Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json')
|
||||
$points = if ($env:FLOOR_POINTS) { $env:FLOOR_POINTS } else { 'run x12 "$EMPTY" SP1_GPU_MEMORY_BUDGET_GB=12;run x12 "$V1" SP1_GPU_MEMORY_BUDGET_GB=12;run x12 "$ONE" SP1_GPU_MEMORY_BUDGET_GB=12;run x12 "$FULL" SP1_GPU_MEMORY_BUDGET_GB=12;run xe26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864;run xe26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864;run x16 "$V1" SP1_GPU_MEMORY_BUDGET_GB=16;run x32 "$V1" SP1_GPU_MEMORY_BUDGET_GB=32;run x12r "$V1" SP1_GPU_MEMORY_BUDGET_GB=12 SP1_GPU_RECURSION_TRACE_ALLOCATION=100663296' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
|
|
@ -45,7 +42,7 @@ run() { # name fixture env...
|
|||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
69
tools/prover-floor/pc2-floor-sweep4-miner.ps1
Normal file
69
tools/prover-floor/pc2-floor-sweep4-miner.ps1
Normal file
|
|
@ -0,0 +1,69 @@
|
|||
# Published WITHOUT --stop-miners: the beside-the-miner pair (the 5090 mining at full rate on the same card). Sweep 4.
|
||||
# Prover floor (5 October 2026): the peak GPU memory and the time of one compressed shard proof through the PATCHED
|
||||
# sp1-gpu-server (/opt/igneum-floor/home/.sp1/bin, reached by HOME=/opt/igneum-floor/home: the SDK spawns the
|
||||
# server it finds under $HOME/.sp1/bin, sp1-cuda-6.8.1/src/server.rs) on PC 2's RTX 5090, the miners STOPPED by the
|
||||
# job (--stop-miners) and the live prover switched off for the run (its server would otherwise own the socket).
|
||||
# Every point: every server killed and its socket unlinked, a 1-s nvidia-smi sampler, one `--mode compressed
|
||||
# --shard 0` of the pv1 host (/opt/igneum-pv1, the UNPATCHED SDK and verifier: its VERIFIED is the unpatched
|
||||
# verifier's word on the patched server's proof), the peak, the time, the FLOOR lines the server prints.
|
||||
# The point list comes from the FLOOR_POINTS environment the job carries, else the default sweep below.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
"RESULT start $(Stamp) prover off for the run: $(Prove $false)"
|
||||
Start-Sleep -Seconds 45
|
||||
"RESULT gpus $(Stamp) $((& nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu,power.draw --format=csv,noheader,nounits 2>$null) -join ' | ')"
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-floor-measure' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$jobW = WslPath $job
|
||||
# The empty live shard is a kit left by the chain job (the wiped-jobs-folder class: an app update may clear it), so
|
||||
# its presence is tested first; when it is gone the EMPTY points report measure_failed rows instead of killing the run.
|
||||
$emptyPath = Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json'
|
||||
if (Test-Path $emptyPath) { $emptyW = WslPath $emptyPath; "RESULT empty_fixture present $emptyPath" } else { $emptyW = '/nonexistent/block-83616.json'; "RESULT empty_fixture MISSING (kit wiped): the EMPTY points will report measure_failed" }
|
||||
$points = if ($env:FLOOR_POINTS) { $env:FLOOR_POINTS } else { 'run me26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864;run me26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864;run me25 "$V1" SP1_GPU_ELEMENT_THRESHOLD=33554432;run m12 "$V1" SP1_GPU_MEMORY_BUDGET_GB=12' }
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
CUDA_DIR="$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "$CUDA_DIR" ] && export PATH="$CUDA_DIR/bin:$PATH" && export LD_LIBRARY_PATH="$CUDA_DIR/lib64:/usr/lib/wsl/lib:${LD_LIBRARY_PATH:-}"
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
JOB='JOBW_PLACEHOLDER'; H=/opt/igneum-pv1/igneum-prove-host; FX="/root/igneum-prove-pv1/proving/fixtures"
|
||||
FLOORHOME=/opt/igneum-floor/home; SRV=$FLOORHOME/.sp1/bin/sp1-gpu-server
|
||||
EMPTY='EMPTY_PLACEHOLDER'; V1="$FX/fees-v1-shards2.json"; FULL="$FX/block-338-shard1.json"; ONE="$FX/block-56-transfers.json"
|
||||
[ -x "$SRV" ] || { echo "RESULT measure_failed no patched server at $SRV"; exit 2; }
|
||||
[ -x "$H" ] || { echo "RESULT measure_failed no pv1 host at $H"; exit 2; }
|
||||
echo "RESULT patched_server sha256=$(sha256sum $SRV | cut -c1-64) version=$($SRV --version 2>/dev/null) host=$(sha256sum $H | cut -c1-16)"
|
||||
echo "RESULT live_server sha256=$(sha256sum /root/.sp1/bin/sp1-gpu-server | cut -c1-16) untouched"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT idle_mib $(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits | head -1)"
|
||||
run() { # name fixture env...
|
||||
local name="$1" fx="$2"; shift 2
|
||||
local tag="$name-$(basename $fx .json)"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock
|
||||
local csv="$JOB/smi-$tag.csv" log="$JOB/log-$tag.txt"
|
||||
nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "$csv" 2>/dev/null &
|
||||
local SMI=$!
|
||||
local t0=$(date +%s)
|
||||
timeout -k 5 "${POINT_BUDGET_S:-180}" env HOME=$FLOORHOME SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 "$@" $H "$fx" --mode compressed --shard 0 --out "$JOB/res-$tag.json" > "$log" 2>&1
|
||||
local rc=$?
|
||||
local wall=$(( $(date +%s) - t0 ))
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
local peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$csv")
|
||||
local n=$(wc -l < "$csv")
|
||||
local line=$(grep -E "^RESULT compressed shard" "$log" | tail -1 | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/prove_s=\1 bytes=\2 verify_s=\3 \4/')
|
||||
local cyc=$(grep -E "^RESULT execute shard" "$log" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/')
|
||||
local err=$(grep -iE "error|panick|out of memory|OOM|unsupported" "$log" | grep -v "^FLOOR" | head -1 | cut -c1-200)
|
||||
echo "RESULT floor cfg=$name fixture=$(basename $fx .json) peak_mib=$peak samples=$n wall_s=$wall cycles=${cyc:-na} ${line:-no_result} exit=$rc env='$*' ${err:+err=$err}"
|
||||
grep -E "^FLOOR" "$log" | sed "s/^/RESULT floorline cfg=$name fixture=$(basename $fx .json) /" | head -40
|
||||
}
|
||||
POINTS_PLACEHOLDER_BASH
|
||||
pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
echo "RESULT measure_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('JOBW_PLACEHOLDER', $jobW).Replace('EMPTY_PLACEHOLDER', $emptyW).Replace('POINTS_PLACEHOLDER_BASH', ($points -replace ';', "`n"))
|
||||
$bashFile = Join-Path $job 'measure.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT end $(Stamp) prover back on: $(Prove $true)"
|
||||
93
tools/prover-floor/pc2-packaged-verify.ps1
Normal file
93
tools/prover-floor/pc2-packaged-verify.ps1
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
# Prover floor, the packaging row's verification run (6 October 2026): the runtime half of the shipped-server path on
|
||||
# PC 2, with what exists there today (the v4 server /opt/igneum-floor/bin/sp1-gpu-server 68b2f512 built on 6 October,
|
||||
# the pv1 host). The signed manifest (prover-server.json + .sig, signed on the Mac with the OTA key) comes from the
|
||||
# fetch job floor-pc2-manifest (jobs\floor-pc2-manifest\). Steps, each a RESULT line: (1) the manifest's sha256 and size
|
||||
# against the binary (the app's check_binary, done here in bash); (2) the app's install script, byte for byte
|
||||
# (app/igneum-app/src/proverserver.rs install_script), run as the app's WSL user: the server lands at
|
||||
# ~/.sp1/bin/sp1-gpu-server where the SDK looks; (3) one v1 shard through the pv1 host with the 12 GB profile
|
||||
# (SP1_GPU_ELEMENT_THRESHOLD=67108864) and the DEFAULT HOME, so the SDK spawns the installed server (the peak and the
|
||||
# time, every proof verified); (4) the app's restore script, so the live prover is as it was (the stock server comes
|
||||
# back by the SDK's download at its next shard). The app's prover is switched off for the run and on at the end; the
|
||||
# miners are stopped by the job (--stop-miners); the live /opt/igneum host is never touched.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' }
|
||||
$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/')
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } }
|
||||
"RESULT start $(Stamp) prover off: $(Prove $false)"
|
||||
Start-Sleep -Seconds 30
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$kit = Join-Path $jobs 'floor-pc2-manifest'
|
||||
$manifest = Join-Path $kit 'prover-server.json'
|
||||
if (-not (Test-Path $manifest)) { "RESULT verify_failed no manifest at $manifest (publish the fetch job floor-pc2-manifest first)"; "RESULT end $(Stamp) prover on: $(Prove $true)"; exit 2 }
|
||||
function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } }
|
||||
$manW = WslPath $manifest
|
||||
$job = $env:IGNEUM_JOB_DIR; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
$jobW = WslPath $job
|
||||
# the WSL user the app runs the host as: the first non-root user of the distribution (the app's prover runs unelevated)
|
||||
$bash = @'
|
||||
set -uo pipefail
|
||||
stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
||||
MAN='MAN_PLACEHOLDER'; JOB='JOB_PLACEHOLDER'
|
||||
BIN=/opt/igneum-floor/bin/sp1-gpu-server; H=/opt/igneum-pv1/igneum-prove-host; FX=/root/igneum-prove-pv1/proving/fixtures/fees-v1-shards2.json
|
||||
[ -x "$BIN" ] && [ -x "$H" ] && [ -f "$FX" ] || { echo "RESULT verify_failed missing: bin=$([ -x $BIN ] && echo ok || echo no) host=$([ -x $H ] && echo ok || echo no) fixture=$([ -f $FX ] && echo ok || echo no)"; exit 2; }
|
||||
# (1) the manifest against the binary, as the app's check_binary does (size first, then the sha256)
|
||||
WANT=$(python3 -c 'import json,sys; f=json.load(open(sys.argv[1]))["files"]["sp1-gpu-server"]; print(f["sha256"], f["bytes"])' "$MAN")
|
||||
WANT_SHA=${WANT% *}; WANT_BYTES=${WANT#* }
|
||||
GOT_SHA=$(sha256sum "$BIN" | cut -c1-64); GOT_BYTES=$(stat -c %s "$BIN")
|
||||
if [ "$GOT_SHA" = "$WANT_SHA" ] && [ "$GOT_BYTES" = "$WANT_BYTES" ]; then echo "RESULT check_binary ok sha256=$GOT_SHA bytes=$GOT_BYTES"; else echo "RESULT check_binary FAILED got $GOT_SHA/$GOT_BYTES want $WANT_SHA/$WANT_BYTES"; exit 2; fi
|
||||
# a tampered copy is refused by the same check (the known-failed case of the gate)
|
||||
cp "$BIN" "$JOB/tampered"; printf 'x' >> "$JOB/tampered"
|
||||
T_SHA=$(sha256sum "$JOB/tampered" | cut -c1-64); T_BYTES=$(stat -c %s "$JOB/tampered")
|
||||
if [ "$T_SHA" = "$WANT_SHA" ] || [ "$T_BYTES" = "$WANT_BYTES" ]; then echo "RESULT tampered_check FAILED: the tampered copy passed"; exit 2; else echo "RESULT tampered_check ok: refused ($T_BYTES bytes is not the manifest's $WANT_BYTES)"; fi
|
||||
rm -f "$JOB/tampered"
|
||||
# (2) the app's install script, as src/proverserver.rs writes it (sha compared first, the server stopped, the copy into ~/.sp1/bin)
|
||||
pkill -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock
|
||||
BEFORE=$(sha256sum "$HOME/.sp1/bin/sp1-gpu-server" 2>/dev/null | cut -c1-64 || true); echo "RESULT before user=$(id -un) home=$HOME installed_sha=${BEFORE:-none}"
|
||||
# the live prover on PC 2 runs as this user, so this IS its server: kept aside and put back at the end (the app's own
|
||||
# restore script deletes it and lets the SDK download the stock release again; on the measured machine the copy is kinder)
|
||||
[ -f "$HOME/.sp1/bin/sp1-gpu-server" ] && cp "$HOME/.sp1/bin/sp1-gpu-server" "$JOB/live-server.bak" && echo "RESULT live_server_backup $BEFORE"
|
||||
cat > "$JOB/install.sh" <<EOS
|
||||
set -u
|
||||
D="\$HOME/.sp1/bin"; T="\$D/sp1-gpu-server"
|
||||
mkdir -p "\$D"
|
||||
have="\$(sha256sum "\$T" 2>/dev/null | cut -c1-64)"
|
||||
if [ "\$have" = $WANT_SHA ]; then echo "RESULT server $WANT_SHA kept"; exit 0; fi
|
||||
pkill -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock
|
||||
cp '$BIN' "\$T.new" && chmod +x "\$T.new" && mv -f "\$T.new" "\$T" || { echo "RESULT server $WANT_SHA install FAILED"; exit 1; }
|
||||
got="\$(sha256sum "\$T" | cut -c1-64)"
|
||||
if [ "\$got" = $WANT_SHA ]; then echo "RESULT server $WANT_SHA installed"; else echo "RESULT server \$got MISMATCH after the copy"; exit 1; fi
|
||||
EOS
|
||||
bash "$JOB/install.sh"; rc=$?; echo "RESULT install_exit $rc"
|
||||
bash "$JOB/install.sh" | sed 's/^RESULT server/RESULT server_again/' # the second run keeps it (the sha matches)
|
||||
echo "RESULT installed_version $($HOME/.sp1/bin/sp1-gpu-server --version 2>/dev/null)"
|
||||
# (3) one v1 shard, the 12 GB profile, the DEFAULT HOME: the SDK spawns ~/.sp1/bin/sp1-gpu-server, which is now the shipped one
|
||||
nvidia-smi --query-gpu=timestamp,memory.used --format=csv,noheader,nounits -l 1 > "$JOB/smi.csv" 2>/dev/null & SMI=$!
|
||||
t0=$(date +%s)
|
||||
SP1_PROVER=cuda RUST_LOG=off SP1_GPU_FLOOR_LOG=1 SP1_GPU_ELEMENT_THRESHOLD=67108864 $H "$FX" --mode compressed --shard 0 --out "$JOB/res.json" > "$JOB/prove.log" 2>&1; prc=$?
|
||||
wall=$(( $(date +%s) - t0 )); pkill -f sp1-gpu-server 2>/dev/null; kill $SMI 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock
|
||||
peak=$(awk -F', *' '{ if ($2+0 > m) m=$2+0 } END { print m+0 }' "$JOB/smi.csv")
|
||||
line=$(grep -E "^RESULT compressed shard" "$JOB/prove.log" | tail -1 | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/prove_s=\1 bytes=\2 verify_s=\3 \4/')
|
||||
opts=$(grep -E "^FLOOR opts" "$JOB/prove.log" | head -1 | cut -c1-160)
|
||||
echo "RESULT prove profile=12gb threshold=67108864 peak_mib=$peak wall_s=$wall ${line:-no_result} exit=$prc server_kind=patched server_sha256=$WANT_SHA"
|
||||
echo "RESULT prove_opts ${opts:-none}"
|
||||
grep -iE "error|panick|Could not" "$JOB/prove.log" | grep -v "^FLOOR" | head -2 | sed 's/^/RESULT prove_err /'
|
||||
# (4) the app's restore script: the stock server comes back by the SDK's own download at the live prover's next shard
|
||||
cat > "$JOB/restore.sh" <<'EOS'
|
||||
set -u
|
||||
T="$HOME/.sp1/bin/sp1-gpu-server"
|
||||
pkill -f sp1-gpu-server 2>/dev/null; rm -f /tmp/sp1-cuda-*.sock
|
||||
rm -f "$T"
|
||||
echo "RESULT server stock restored"
|
||||
EOS
|
||||
bash "$JOB/restore.sh"
|
||||
if [ -f "$JOB/live-server.bak" ]; then mkdir -p "$HOME/.sp1/bin" && mv -f "$JOB/live-server.bak" "$HOME/.sp1/bin/sp1-gpu-server" && chmod +x "$HOME/.sp1/bin/sp1-gpu-server" && echo "RESULT live_server_restored $(sha256sum "$HOME/.sp1/bin/sp1-gpu-server" | cut -c1-64)"; fi
|
||||
echo "RESULT after installed_sha=$(sha256sum "$HOME/.sp1/bin/sp1-gpu-server" 2>/dev/null | cut -c1-64 || echo none) (none = the SDK downloads the stock 6.8.1 at the next proof)"
|
||||
echo "RESULT verify_end $(stamp)"
|
||||
'@
|
||||
$bash = $bash.Replace('MAN_PLACEHOLDER', $manW).Replace('JOB_PLACEHOLDER', $jobW)
|
||||
$bashFile = Join-Path $job 'verify.sh'
|
||||
[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false))
|
||||
& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') }
|
||||
"RESULT end $(Stamp) prover on: $(Prove $true)"
|
||||
4
tools/prover-floor/points-core.txt
Normal file
4
tools/prover-floor/points-core.txt
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
runshard ce26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
runshard ce26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
runshard ce27 "$V1" SP1_GPU_ELEMENT_THRESHOLD=134217728
|
||||
runshard ce27 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=134217728
|
||||
5
tools/prover-floor/points-core2.txt
Normal file
5
tools/prover-floor/points-core2.txt
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
runshard ce25 "$V1" SP1_GPU_ELEMENT_THRESHOLD=33554432
|
||||
runshard ce25 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=33554432
|
||||
runshard ce24 "$V1" SP1_GPU_ELEMENT_THRESHOLD=16777216
|
||||
runshard ce26r0 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864 SP1_GPU_MEM_RELEASE_THRESHOLD=0
|
||||
runshard ce25r0 "$V1" SP1_GPU_ELEMENT_THRESHOLD=33554432 SP1_GPU_MEM_RELEASE_THRESHOLD=0
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
run m12 "$EMPTY" SP1_GPU_MEMORY_BUDGET_GB=12
|
||||
run m12 "$V1" SP1_GPU_MEMORY_BUDGET_GB=12
|
||||
run me26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
run me26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
run me25 "$V1" SP1_GPU_ELEMENT_THRESHOLD=33554432
|
||||
run m12 "$V1" SP1_GPU_MEMORY_BUDGET_GB=12
|
||||
|
|
|
|||
9
tools/prover-floor/points-sweep3.txt
Normal file
9
tools/prover-floor/points-sweep3.txt
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
run x12 "$EMPTY" SP1_GPU_MEMORY_BUDGET_GB=12
|
||||
run x12 "$V1" SP1_GPU_MEMORY_BUDGET_GB=12
|
||||
run x12 "$ONE" SP1_GPU_MEMORY_BUDGET_GB=12
|
||||
run x12 "$FULL" SP1_GPU_MEMORY_BUDGET_GB=12
|
||||
run xe26 "$V1" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
run xe26 "$EMPTY" SP1_GPU_ELEMENT_THRESHOLD=67108864
|
||||
run x16 "$V1" SP1_GPU_MEMORY_BUDGET_GB=16
|
||||
run x32 "$V1" SP1_GPU_MEMORY_BUDGET_GB=32
|
||||
run x12r "$V1" SP1_GPU_MEMORY_BUDGET_GB=12 SP1_GPU_RECURSION_TRACE_ALLOCATION=100663296
|
||||
Loading…
Reference in a new issue