From 344cba8e8cb08f1175c69daad5b9613c2c4c254a Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:35:13 +0000 Subject: [PATCH] Proving v1: the memory sweep and the miner-on peaks, the root-socket class fix (cleanup lines, tools/ci/prover-socket-check.sh in CI), the host's --budget re-plan and the S_p curve job, the RAM and aggregation-card gates, N = 8 in the fast-time file and spec 7.4 Co-Authored-By: Claude Fable 5.1 --- .github/workflows/ci.yml | 2 + app/igneum-app/src/detect.rs | 16 +++++ app/igneum-app/src/engine.rs | 2 +- app/igneum-app/src/provedefault.rs | 63 ++++++++++++++----- app/igneum-app/src/prover.rs | 8 +++ docs/bench-log.md | 8 ++- docs/plans/proving-v1.md | 2 + docs/spec/07-execution.md | 6 +- infra/fast-time/override-60x.json | 2 +- proving/igneum-prove/host/src/main.rs | 26 +++++--- tools/ci/prover-socket-check.sh | 19 ++++++ tools/proving-v1/pc2-chain.ps1 | 1 + tools/proving-v1/pc2-memory-miner-on.ps1 | 56 +++++++++++++++++ tools/proving-v1/pc2-memory-sweep.ps1 | 6 +- tools/proving-v1/pc2-sp-curve.ps1 | 78 ++++++++++++++++++++++++ 15 files changed, 264 insertions(+), 31 deletions(-) create mode 100755 tools/ci/prover-socket-check.sh create mode 100644 tools/proving-v1/pc2-memory-miner-on.ps1 create mode 100644 tools/proving-v1/pc2-sp-curve.ps1 diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a5cd62109..7ef1d7a8a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -73,6 +73,8 @@ jobs: run: bash tools/ci/signer-pipe-check.sh - name: pinned guest programs match their manifest and are built only by pin-guests.sh run: bash tools/ci/pinned-guests-check.sh + - name: root prover playbooks kill the GPU server and unlink its socket (the root-socket class, 5 October 2026) + run: bash tools/ci/prover-socket-check.sh - name: no secret file names and no 64-hex secrets in the tree (self-test first, then the tree) run: bash tools/ci/no-secrets-check.sh --self-test && bash tools/ci/no-secrets-check.sh - name: faucet unit tests (validation, the daily limits, the signed transaction; keccak, RLP and secp256k1 vectors) diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index cb7c1f625..87d0b961b 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -983,3 +983,19 @@ mod tests { assert_eq!(big.identities, 8); } } + +/// The machine's RAM in MB (the prover default's RAM gate, src/provedefault.rs): Windows through +/// `Win32_OperatingSystem.TotalVisibleMemorySize` (KB), Linux through `/proc/meminfo`, macOS through `sysctl hw.memsize`; +/// None when unreadable (no gate). +pub fn total_ram_mb() -> Option { + if cfg!(windows) { + let out = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", "(Get-CimInstance Win32_OperatingSystem).TotalVisibleMemorySize"]), None, Duration::from_secs(20))?; + return out.replace('\0', "").trim().parse::().ok().map(|kb| kb / 1024); + } + if cfg!(target_os = "linux") { + let text = std::fs::read_to_string("/proc/meminfo").ok()?; + return text.lines().find(|l| l.starts_with("MemTotal:")).and_then(|l| l.split_whitespace().nth(1)).and_then(|kb| kb.parse::().ok()).map(|kb| kb / 1024); + } + let out = run_timeout(Command::new("sysctl").args(["-n", "hw.memsize"]), None, Duration::from_secs(5))?; + out.trim().parse::().ok().map(|b| b / (1024 * 1024)) +} diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 599902c82..8831769bf 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -267,7 +267,7 @@ impl Shared { } let cards = self.state.lock().unwrap().mining.cards.clone(); let wsl = if cfg!(windows) { Some(crate::wslhost::distro_answers()) } else { None }; - let d = crate::provedefault::decide(&cards, std::env::consts::OS, wsl); + let d = crate::provedefault::decide(&cards, std::env::consts::OS, wsl, crate::detect::total_ram_mb()); let on = d.on || already_on; { let mut s = self.settings.lock().unwrap(); diff --git a/app/igneum-app/src/provedefault.rs b/app/igneum-app/src/provedefault.rs index 78c758fbc..0c12cef5f 100644 --- a/app/igneum-app/src/provedefault.rs +++ b/app/igneum-app/src/provedefault.rs @@ -34,8 +34,19 @@ fn gb(mb: u64) -> u64 { (mb + 512) / 1024 } -/// `os` is `std::env::consts::OS` ("windows", "linux", "macos"); `wsl_answers` is read on Windows only. -pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option) -> Decision { +/// Windows machines under this much RAM stay off until measured (consequences review C4, 5 October 2026): PC 2 at +/// 63 GB had 25.6 GB in use with the WSL2 VM's working set at 7.9 GB while proving; a 16 GB PC would swap. +pub const MIN_RAM_MB_WINDOWS: u64 = 31_000; + +/// The card an aggregation (the chained SP1 recursion, spec 7.8) may run on: the same memory rule as the shard +/// prover (a mining card 20 GB, an idle one 16 GB: 16,751 MiB measured with the miner resident, about 13.4 GB alone). +pub fn aggregation_card(cards: &[CardState]) -> Option<&CardState> { + cards.iter().filter(|c| c.vendor == "nvidia" && c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).max_by_key(|c| c.vram_mb) +} + +/// `os` is `std::env::consts::OS` ("windows", "linux", "macos"); `wsl_answers` is read on Windows only; `ram_mb` is the +/// machine's RAM when the platform reports it (None = unknown, no gate). +pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option, ram_mb: Option) -> Decision { let nvidia: Vec<&CardState> = cards.iter().filter(|c| c.vendor == "nvidia").collect(); // a mining card needs 20 GB (the measured mine-and-prove peak of 16.8 GB), a card that only proves 16 GB let able: Vec<&CardState> = nvidia.iter().copied().filter(|c| c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).collect(); @@ -57,6 +68,13 @@ pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option) -> Decis return off(format!("proving off by default: {why} ({seen})")); }; let card = format!("{} ({} GB{})", best.name, gb(best.vram_mb), if best.enabled { ", mining too" } else { ", proving only" }); + if os == "windows" { + if let Some(ram) = ram_mb { + if ram < MIN_RAM_MB_WINDOWS { + return off(format!("proving off by default: {card} qualifies but this PC has {} GB of RAM; proving needs 32 GB on Windows until a smaller PC is measured (the WSL2 prover held 7.9 GB on a 63 GB PC); Settings switches it on", gb(ram))); + } + } + } match os { "windows" => match wsl_answers { Some(true) => Decision { on: true, line: format!("proving on by default: {card} with WSL2 (Ubuntu-24.04 answers); Settings switches it off") }, @@ -80,49 +98,64 @@ mod tests { #[test] fn a_5090_with_wsl2_on_windows_is_on() { - let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5090", 32_607), card("amd", "AMD Radeon(TM) Graphics", 512)], "windows", Some(true)); + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5090", 32_607), card("amd", "AMD Radeon(TM) Graphics", 512)], "windows", Some(true), Some(63_132)); assert!(d.on); assert!(d.line.starts_with("proving on by default: NVIDIA GeForce RTX 5090 (32 GB, mining too) with WSL2"), "{}", d.line); } #[test] fn windows_without_wsl2_is_off_with_the_setup_hint() { - let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(false)); + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(false), Some(65_000)); assert!(!d.on); assert!(d.line.contains("did not answer") && d.line.contains("Set up"), "{}", d.line); - assert!(!decide(&[card("nvidia", "RTX 4090", 24_564)], "windows", None).on, "an unread probe is not an answer"); + assert!(!decide(&[card("nvidia", "RTX 4090", 24_564)], "windows", None, Some(65_000)).on, "an unread probe is not an answer"); } #[test] fn linux_needs_no_wsl2_and_the_memory_gates_hold() { // a mining 4090 (24 GB) is on; a mining 16 GB card is off with the reason; the same 16 GB card not mining is on - assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None).on); - let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None); + assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, None).on); + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None); assert!(!d.on); assert!(d.line.contains("mining and proving on one card needs 20 GB") && d.line.contains("RTX 5080 16 GB, mining"), "{}", d.line); - let d = decide(&[idle("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None); + let d = decide(&[idle("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None); assert!(d.on); assert!(d.line.contains("(16 GB, proving only)"), "{}", d.line); // a 12 GB card is off either way (the prover alone peaks at 13.8 GB); a 10 GB card too - let d = decide(&[idle("nvidia", "NVIDIA GeForce RTX 3060", 12_288)], "linux", None); + let d = decide(&[idle("nvidia", "NVIDIA GeForce RTX 3060", 12_288)], "linux", None, None); assert!(!d.on); assert!(d.line.contains("no NVIDIA card with 20 GB or more mining, or 16 GB or more free of mining") && d.line.contains("RTX 3060 12 GB"), "{}", d.line); - assert!(!decide(&[card("nvidia", "NVIDIA GeForce RTX 3080", 10_240)], "linux", None).on); - assert!(!decide(&[card("amd", "Radeon RX 9070 XT", 16_384)], "linux", None).on, "no CUDA prover for AMD yet"); - assert!(decide(&[], "linux", None).line.contains("no NVIDIA card")); + assert!(!decide(&[card("nvidia", "NVIDIA GeForce RTX 3080", 10_240)], "linux", None, None).on); + assert!(!decide(&[card("amd", "Radeon RX 9070 XT", 16_384)], "linux", None, None).on, "no CUDA prover for AMD yet"); + assert!(decide(&[], "linux", None, None).line.contains("no NVIDIA card")); } #[test] fn apple_silicon_stays_off() { - let d = decide(&[card("apple", "Apple M5 Max", 65_536)], "macos", None); + let d = decide(&[card("apple", "Apple M5 Max", 65_536)], "macos", None, Some(65_536)); assert!(!d.on); assert!(d.line.contains("Apple silicon")); - assert!(!decide(&[card("nvidia", "RTX 5090", 32_607)], "macos", Some(true)).on, "the OS rule comes first"); + assert!(!decide(&[card("nvidia", "RTX 5090", 32_607)], "macos", Some(true), None).on, "the OS rule comes first"); + } + + #[test] + fn a_windows_pc_under_32_gb_stays_off_and_the_aggregation_card_follows_the_same_gate() { + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(true), Some(16_300)); + assert!(!d.on); + assert!(d.line.contains("16 GB of RAM") && d.line.contains("needs 32 GB on Windows"), "{}", d.line); + assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(true), None).on, "unknown RAM is not a gate"); + assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, Some(16_300)).on, "the RAM gate is Windows only (the WSL2 VM)"); + let cards = [card("nvidia", "RTX 5080", 16_303), idle("nvidia", "RTX 4070 Ti", 12_282)]; + assert!(aggregation_card(&cards).is_none(), "a mining 16 GB card and an idle 12 GB card cannot aggregate"); + let cards = [card("nvidia", "RTX 5080", 16_303), idle("nvidia", "RTX 5080 (2)", 16_303)]; + assert_eq!(aggregation_card(&cards).map(|c| c.name.as_str()), Some("RTX 5080 (2)")); + let cards = [card("nvidia", "RTX 5090", 32_607)]; + assert_eq!(aggregation_card(&cards).map(|c| c.vram_mb), Some(32_607)); } #[test] fn the_biggest_qualifying_card_is_named() { - let d = decide(&[idle("nvidia", "RTX 5080", 16_303), card("nvidia", "RTX 5090", 32_607)], "linux", None); + let d = decide(&[idle("nvidia", "RTX 5080", 16_303), card("nvidia", "RTX 5090", 32_607)], "linux", None, None); assert!(d.line.contains("RTX 5090 (32 GB, mining too)"), "{}", d.line); } } diff --git a/app/igneum-app/src/prover.rs b/app/igneum-app/src/prover.rs index 7b34b9b52..29dac5c8e 100644 --- a/app/igneum-app/src/prover.rs +++ b/app/igneum-app/src/prover.rs @@ -587,6 +587,14 @@ fn aggregate_once(shared: &Shared, t: &Tools, label: &str, payout: &str, attempt if !v1["active"].as_bool().unwrap_or(false) || v1["start"].is_null() { return Ok(None); } + // the card gate (consequences review C2): a chained aggregation peaked at 16,751 MiB with the miner resident, so a + // mining card needs 20 GB and an idle one 16 GB; without such a card this machine proves shards and never aggregates + { + let cards = shared.state.lock().unwrap().mining.cards.clone(); + if crate::provedefault::aggregation_card(&cards).is_none() { + return Ok(Some("no aggregation on this machine: it needs a 24 GB card while mining or a 16 GB card that does not mine (16.8 GB measured with the miner resident); shards still prove".into())); + } + } let tip = hexu(&evm_rpc(shared, "eth_blockNumber", json!([]), Duration::from_secs(10))?); let mut seg = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{tip:#x}")]), Duration::from_secs(10))?; if !seg["executed"].as_bool().unwrap_or(false) { diff --git a/docs/bench-log.md b/docs/bench-log.md index d3de8413d..ad31f15b1 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1611,7 +1611,7 @@ Branches `proving-v1` (main repository, worktree `igneum-wt-proving-v1`; fork `v | Eight consecutive live fixtures | `igneum_exportSegments 0x0..0x13cb4` on node 1's exec RPC (127.0.0.1:26790, read-only, 20:06 BST, tip 81,076): 71,042,616 bytes, 81,077 segments, 28 accounts, 0.5 s; `igneum-prove-export export.json block-.json` for 81046..81053: replayed 81,077 segments from genesis in 1.8 s each, every state root equal to the node's; one shard a block, 0 pgas (no transactions on the devnet tonight), `proving/fixtures/chain/` | | Chain of 2 on the Mac CPU (the known-finished case of `--mode chain` before the GPU; M5 Max under the live nodes, the harness and two builds) | `SP1_PROVER=cpu igneum-prove-host --mode chain --chain block-81046.json,block-81047.json --out results.json` under the run lock, 19:07:48Z to 19:11:28Z: setup 12.2 s; block 81046: shard 0 compressed 55.4 s (1,272,897 bytes, verify 0.036 s), aggregate 52.0 s (1,272,909 bytes, verify 0.031 s), chain_len 1, agg_vk zero; block 81047: shard 41.3 s, aggregate WITH the previous block proof 59.1 s, chain_len 2, agg_vk = the pinned aggregator id; end to end 207.9 s; final proof 1,272,909 bytes, statement 0x232276f4... The recursion over the previous proof cost 7 s more than the first aggregation on this CPU | | `--mode verify-segment` on that proof (the node's path: SP1 light verifier, pinned aggregator key) | VERIFIED in 0.032 s (0.27 s wall, three runs: 0.033, 0.032, 0.032); known-failed: a wrong statement NOT VERIFIED (0.032 s); the shard verifier (`--mode verify`) on the segment proof NOT VERIFIED, "program id 0x474678f3... IS NOT OURS 0x2b1a81cb..." | -| Chain of 8 on the RTX 5090 (N = 2, 4, 8), job `chain-pc2-pv1b` (`tools/proving-v1/pc2-chain.ps1`; the package `igneum-prove-wsl2-pv1.zip` eb6dccf8..., 1.5 MB, fetched by `fetch-prove-pv1` 19:51Z; the first try `chain-pc2-pv1` died in its own export step, fixed) | Ran 19:58:37Z: the export from PC 2's node (72,901,414 bytes, 1.4 s), the host built in WSL2 against the live build's warm target dir in 6 s and installed to `/opt/igneum-pv1` (the live `/opt/igneum` host untouched, sha 29cc4768...), `--mode id` the pinned pair; eight consecutive fixtures 83346..83353 cut, every one MATCHES natively. The chain on the GPU (SP1_PROVER=cuda, the miner mining on the same card at 119 MH/s): setup 12.7 s; block 83346: shard 7.4 s, aggregate 7.6 s (chain_len 1), 15.1 s; block 83347: shard 7.2 s, aggregate WITH the previous proof 9.5 s (chain_len 2, agg_vk the pinned aggregator id), 16.8 s, cumulative 31.8 s over 2 blocks; block 83348: shard 7.0 s, then at 20:01:09Z the app quit for the 0.3.10 update and aborted the job ("aborted (the app is quitting)"). So N = 2 measured: 31.8 s of GPU time for two empty blocks, the chained aggregation 1.9 s dearer than the first; N = 4 and 8 are the re-run `chain-pc2-pv1c` after the restart. An empty shard's compressed proof on the 5090 is 7.0 to 7.4 s (the 200-pgas shard of 4 October took 2.7 s with the card to itself; tonight the miner held it at 92% utilisation) | +| Chain of 8 on the RTX 5090 (N = 2, 4, 8), job `chain-pc2-pv1b` (`tools/proving-v1/pc2-chain.ps1`; the package `igneum-prove-wsl2-pv1.zip` eb6dccf8..., 1.5 MB, fetched by `fetch-prove-pv1` 19:51Z; the first try `chain-pc2-pv1` died in its own export step, fixed) | Ran 19:58:37Z: the export from PC 2's node (72,901,414 bytes, 1.4 s), the host built in WSL2 against the live build's warm target dir in 6 s and installed to `/opt/igneum-pv1` (the live `/opt/igneum` host untouched, sha 29cc4768...), `--mode id` the pinned pair; eight consecutive fixtures 83346..83353 cut, every one MATCHES natively. The chain on the GPU (SP1_PROVER=cuda, the miner mining on the same card at 119 MH/s): setup 12.7 s; block 83346: shard 7.4 s, aggregate 7.6 s (chain_len 1), 15.1 s; block 83347: shard 7.2 s, aggregate WITH the previous proof 9.5 s (chain_len 2, agg_vk the pinned aggregator id), 16.8 s, cumulative 31.8 s over 2 blocks; block 83348: shard 7.0 s, then at 20:01:09Z the app quit and aborted the job ("quit: stopping the miners, then the node", then "job chain-pc2-pv1b: aborted (the app is quitting)"; NOT an update: nothing of 0.3.10 was published; the log gives the quit no source; 20 s earlier the efficiency sweep's administrator prompt had been cancelled at the keyboard, and 13 s earlier the live prover had failed with "CudaClientError: Connect(PermissionDenied)", the root-owned socket my job had left, below). So N = 2 measured: 31.8 s of GPU time for two empty blocks, the chained aggregation 1.9 s dearer than the first; N = 4 and 8 are the re-run `chain-pc2-pv1c` after the restart. An empty shard's compressed proof on the 5090 is 7.0 to 7.4 s (the 200-pgas shard of 4 October took 2.7 s with the card to itself; tonight the miner held it at 92% utilisation) | | The chain of 8, the third run `chain-pc2-pv1c` (20:05:21Z to 20:08:33Z, after the app restart; blocks 83616..83623 from PC 2's node at tip 83646, the same script; results `tools/proving-v1/chain-pc2-2026-10-05.json`) | Build 5 s (warm), eight fixtures cut and MATCHING natively, setup 15.7 s, then on the GPU with the miner mining on the same card: shard proofs 7.3 to 7.7 s each (8 x, 59.5 s), aggregations 7.9 s for the first block and 9.6 to 9.7 s for every chained one (75.5 s), every proof VERIFIED, end to end 135.6 s for 8 blocks (17.0 s a block from the second on). Cumulative: N = 2 at 32.6 s, N = 4 at 66.8 s, N = 8 at 135.6 s. The final proof is 1,272,909 bytes whatever N (chain_len 8, agg_vk the pinned aggregator id), the record 586 bytes; `--mode verify-segment` on it: VERIFIED in 0.039, 0.037, 0.040 s after a 0.26-s light-verifier setup, the same three runs each time. GPU memory over the chain (152 one-second samples): max 16,751 MiB with the miner's 3.4 GB resident, so the chained aggregation holds about 13.4 GB, 1.2 GB over the shard-only peak; WSL used 2,456 MB | Reading the chain numbers. Aggregation is a fixed cost per block (9.7 s here), not per segment: the recursion verifies one more proof whatever `chain_len`, so the record for N blocks costs N aggregations and the verifier one. Against 4 October with the miner stopped (aggregate 2.2 to 2.5 s, a 200-pgas shard 2.7 s), tonight's 9.7 s and 7.3 s say the miner's 92% utilisation slows the prover about 3 to 4x while the prover slows the miner 4%: the card is shared, and the lottery wins the arbitration. A machine that mines and proves at once delivers one empty block's proof and aggregation in 17 s; one that only proves, about 5 s (approximate, from the 4 October stages). @@ -1662,6 +1662,12 @@ Job `memsweep-pc2-pv1` (`tools/proving-v1/pc2-memory-sweep.ps1`), PC 2's RTX 509 Reading. The GPU memory of a compressed shard proof is **13.9 GB for a shard of 280,000 cycles and 28.3 GB for one of 60 M cycles**, and no knob the environment carries moves the floor: the worker counts only slow the proof (11.4 s to 20.8 s), the trace thresholds at 2^26 and 2^27 change nothing, and the smallest trace threshold tried (2^25 elements, 2^20 rows) takes 5.4 GB off the full shard (22.9 GB) at twice the time. The floor sits in the GPU server's own allocation, not in the shard: an empty shard with every knob at its minimum still takes 13.9 GB. So on SP1 6.8.1's `sp1-gpu-server` as shipped, **a 12 GB card cannot prove even an empty shard** (13.9 GB), and the 11.0 GB target of tonight's requirement is out of reach from the environment. The S_p/2 and S_p/4 cuts of block 344 did not run: the package carries no `tools/prove-fixtures/seq.json` (the cut rows need the export; they would sit between the two measured points, and the floor is the binding number anyway). What is left to try, in order: the server's own options (its `--help` and the option names in its strings: the miner-on job prints them), SP1's core-only proof (the node needs the compressed proof, so this changes the protocol), and an SP1 release built for smaller cards (the 6.8.1 release notes are not read here; approximate: the project's documentation names 24 GB as the GPU requirement, `proving/windows-wsl2/setup-wsl.sh` quotes it). +### The same shard with the miner running (the 16 GB requirement), and the GPU server's own options + +Job `memminer-pc2-pv1` (`tools/proving-v1/pc2-memory-miner-on.ps1`), 20:28 to 20:30Z, the miner at full rate on the card, the live prover off for the run, the same 1-s sampler: the full shard at `S_p` (60.4 M cycles) peaked at **30,039 MiB** and took 33.0 s (28,295 MiB and 11.4 s with the card to itself: the miner costs the prover 2.9x in time and 1.7 GB of memory); the empty shard **15,670 MiB** and 7.7 s (13,863 and 2.3 s alone). So a 32 GB card mines and proves the prototype shard with 2.5 GB to spare; a 24 GB card cannot prove it even alone (28.3 GB); a 16 GB card cannot hold even the empty shard beside the miner (15.7 GB, the display and driver on top). `sp1-gpu-server --help` prints only `--version`: it has no options of its own, and its strings carry no memory setting (`CUDA_OUT_OF_MEMORY` is an error name). The shard SIZE is therefore the only lever left on this build, measured next as the S_p curve. + +The root-socket fault (the class, fixed the same evening). The chain and memory jobs ran the host as root inside WSL2; the first `sp1-gpu-server` they started left `/tmp/sp1-cuda-0.sock` owned by root, and the live prover (the app's own WSL user) then failed every shard with `CudaClientError: Connect(Os { code: 13, kind: PermissionDenied })` (PC 2 app log 1791230456, 20:00:56Z) until the socket was gone. Every pv1 playbook now kills the server and unlinks `/tmp/sp1-cuda-*.sock` at its start and end, `tools/ci/prover-socket-check.sh` fails CI on any playbook that runs a prove mode as root without both lines (shown failing on `pc2-prover-cost.ps1` before its `--mode id`-only exemption, passing after), and the plan carries the rule: a prover job on a shared card runs as the app's user or cleans its socket. Playbooks that run the host: `tools/proving-v1/pc2-chain.ps1`, `pc2-memory-sweep.ps1`, `pc2-memory-miner-on.ps1`, `pc2-sp-curve.ps1` (all root, all with the cleanup now; the first two chain and sweep runs had none), `pc2-prover-cost.ps1` (`--mode id` only), `relay/playbooks/shard-test.ps1` and `proving/windows-wsl2/prove-shard.sh`, `prove-block.sh` (the app's user, not root), `tools/proving-v0/run.mjs` (the Mac, no server). + ### Step 4, the rule | What | Measured | diff --git a/docs/plans/proving-v1.md b/docs/plans/proving-v1.md index 63c16e3ec..75ca70434 100644 --- a/docs/plans/proving-v1.md +++ b/docs/plans/proving-v1.md @@ -112,6 +112,8 @@ Reading. Nobody pays an aggregator as a separate role: Aztec's 30% goes to whoev | Apple silicon default | **off** | the gate was "a shard under 60 s with the miner running": the M5 Max CPU took 41.3 and 55.4 s for EMPTY shards under tonight's load and 272 s for a 200-pgas shard on 4 October; a full shard at `S_p` was never under 60 s. Settings switches it on | bench-log "proving v1" CPU chain row; 4 October CPU rows | | The prover profile per card and the 12 GB and 16 GB gates (the project lead: "make sure we can prove on 12gb cards"; "is there any way we can make 12gb cards mine and prove?") | **measured on PC 2, the rows below** | the SP1 6.8.1 GPU server reads `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, `SHARD_SIZE` and the `SP1_WORKER_NUM_*`/`BUFFER_SIZE` knobs from the environment it inherits (`sp1-core-executor-6.8.1/src/opts.rs`, `sp1-prover-6.8.1/src/worker/config.rs`); the app passes a profile per card (`provedefault.rs`) and the host forwards it | the sweep job `memsweep-pc2-pv1` and the miner-on run | +A prover job on a shared card runs as the app's user or cleans its socket (`pkill -f sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock` at the start and the end; `tools/ci/prover-socket-check.sh`): the root-socket fault of 20:00Z, bench-log. + ### The prover profiles (filled from the sweep) (the table of peak against knobs against shard time, with the miner stopped and with the miner running, lands here when `memsweep-pc2-pv1` and the miner-on run report) diff --git a/docs/spec/07-execution.md b/docs/spec/07-execution.md index 5bca4f2e5..a88fa8ccb 100644 --- a/docs/spec/07-execution.md +++ b/docs/spec/07-execution.md @@ -80,9 +80,9 @@ Nothing in consensus changes for any of this: the segment claim already commits | Proving v0 activation | `proving_v0_activation_daa`, default never | Implemented, 7.7 | | Segment record (v1) | per segment of `proving_v1_segment_blocks` chain blocks, 586 bytes (the aggregator guest's 340-byte statement inline), BLS-signed by the aggregator's vote key, in the coinbase extra data before the shard record section (`IGNS`); proof bytes on p2p message 75 (protocol 15) | Implemented, 7.8 (branch `proving-v1`, 5 October 2026) | | Segment records per block | 2 | Implemented, 7.8 (Designed value) | -| Segment length `N` | `proving_v1_segment_blocks`, 4 | Implemented, value Designed (the project lead decides at 0.3.11) | -| Unproven deadline `T` | `proving_v1_unproven_daa`, 600 DAA s after the segment's last chain block | Implemented, value Designed (the project lead decides at 0.3.11) | -| Aggregator share | `proving_v1_aggregator_share_bps`, 1,000 (a tenth of every attested block's pool credit; the shards share the rest) | Implemented, value Designed (the project lead decides at 0.3.11) | +| Segment length `N` | `proving_v1_segment_blocks`, 8 | Implemented, value Decided (5 October 2026, delegated; `docs/plans/proving-v1.md`) | +| Unproven deadline `T` | `proving_v1_unproven_daa`, 600 DAA s after the segment's last chain block | Implemented, value Decided (5 October 2026, delegated) | +| Aggregator share | `proving_v1_aggregator_share_bps`, 1,000 (a tenth of every attested block's pool credit; the shards share the rest) | Implemented, value Decided (5 October 2026, delegated) | | Proving v1 activation | `proving_v1_activation_daa`, default never; the segment grid starts at the first chain block at or above it | Implemented, 7.8 | | Mandatory proofs | the rule of 7.8 item 9, no switch yet, off | Designed | diff --git a/infra/fast-time/override-60x.json b/infra/fast-time/override-60x.json index 1b5a2521a..21966fc01 100644 --- a/infra/fast-time/override-60x.json +++ b/infra/fast-time/override-60x.json @@ -56,7 +56,7 @@ "finality_v3_activation_daa": 18446744073709551615, "fees_v1_activation_daa": 0, "proving_v1_activation_daa": 18446744073709551615, - "proving_v1_segment_blocks": 4, + "proving_v1_segment_blocks": 8, "proving_v1_unproven_daa": 60, "proving_v1_aggregator_share_bps": 1000, "fees": {"pgas": {"version": 1, "cycles_per_pgas": 1000, "intrinsic_pgas_per_tx": 300, "modexp_base": 10, "modexp_per_byte_numer": 1, "modexp_per_byte_denom": 10}, "block_proving_gas_limit": 120000, "shard_proving_gas_budget": 30000, "min_execution_base_fee_wei": 100000000000, "min_proving_base_fee_wei": 10000000000000, "initial_execution_base_fee_wei": 100000000000, "initial_proving_base_fee_wei": 10000000000000, "base_fee_change_denominator": 8} diff --git a/proving/igneum-prove/host/src/main.rs b/proving/igneum-prove/host/src/main.rs index 6891a9658..d7ed375b1 100644 --- a/proving/igneum-prove/host/src/main.rs +++ b/proving/igneum-prove/host/src/main.rs @@ -95,8 +95,11 @@ fn run() -> Result<()> { let fixtures: Vec = list.split(',').map(|s| s.trim().to_string()).filter(|s| !s.is_empty()).collect(); return run_chain(&pinned, &fixtures, prover, out_path.as_deref()); } - let path = args.get(1).filter(|a| !a.starts_with("--")).context("usage: igneum-prove-host [--mode native|execute|shard|compressed|block|all] [--shard N] [--prover 0x..] [--out results.json]; --mode chain --chain [--prover 0x..] [--out results.json]; --mode aggregate --proofs --parent 0x.. [--prev prev.bin] [--out results.json]; --mode verify --proof --statement 0x..; --mode verify-segment --proof --statement 0x..; --mode id")?; + let path = args.get(1).filter(|a| !a.starts_with("--")).context("usage: igneum-prove-host [--mode native|execute|shard|compressed|block|all] [--shard N] [--budget ] [--prover 0x..] [--out results.json]; --mode chain --chain [--prover 0x..] [--out results.json]; --mode aggregate --proofs --parent 0x.. [--prev prev.bin] [--out results.json]; --mode verify --proof --statement 0x..; --mode verify-segment --proof --statement 0x..; --mode id")?; let shard_index: usize = arg("--shard").map(|s| s.parse()).transpose()?.unwrap_or(0); + // `--budget `: re-plan the fixture's block at a TEST budget (the S_p curve of 5 October 2026); the fixture's + // own per-shard plan is then not compared (the chain and the sums still are), and `--out` records the cut + let test_budget: Option = arg("--budget").map(|s| s.parse()).transpose()?; let fixture: Fixture = serde_json::from_str(&std::fs::read_to_string(path).with_context(|| format!("read {path}"))?)?; if fixture.format != igneum_prove_core::fixture::FORMAT { bail!("fixture format {} is not {} (regenerate with igneum-prove-export)", fixture.format, igneum_prove_core::fixture::FORMAT); @@ -107,7 +110,10 @@ fn run() -> Result<()> { // The fee set that meters this block: the fixture's schedule read at the block's DAA score (5 October 2026, // `fees_v1_activation_daa`); the plan's budget must be that set's S_p unless the fixture says test cut. let fee_set = block.fees.at(block.env.daa_score); - if fixture.plan.consensus && fixture.plan.shard_budget != fee_set.shard_proving_gas_budget { + if let Some(b) = test_budget { + println!("TEST CUT: the block is re-planned at a budget of {b} pgas (the fixture's plan at {} is not compared)", fixture.plan.shard_budget); + } + if test_budget.is_none() && fixture.plan.consensus && fixture.plan.shard_budget != fee_set.shard_proving_gas_budget { bail!("the fixture's plan says consensus budget {} but the fee set at DAA score {} ({}) has S_p {}; regenerate with igneum-prove-export", fixture.plan.shard_budget, block.env.daa_score, fee_set.name(), fee_set.shard_proving_gas_budget); } println!( @@ -143,8 +149,11 @@ fn run() -> Result<()> { // 1. Native: the cut, the witnesses, every shard statement, the chain and the sums, against the fixture. stage("native"); let t = Instant::now(); - let (outcome, pre_root, shards) = build_shards(block, fixture.plan.shard_budget, prover); + let budget = test_budget.unwrap_or(fixture.plan.shard_budget); + let (outcome, pre_root, shards) = build_shards(block, budget, prover); let native_s = t.elapsed().as_secs_f64(); + results.insert("budget".into(), budget.into()); + results.insert("test_cut".into(), test_budget.is_some().into()); let e = &fixture.expected; let same = pre_root == e.pre_state_root && outcome.state_root == e.post_state_root && outcome.receipts_root == e.receipts_root && outcome.tx_commitment == e.tx_commitment && outcome.gas_used == e.gas_used && outcome.pgas_used == e.pgas_used; println!( @@ -166,18 +175,20 @@ fn run() -> Result<()> { if e.node_state_root != e.post_state_root { bail!("the fixture's node state root differs from its expected post-state root; the exporter must not have produced this file"); } - if shards.len() != fixture.plan.shards.len() { + if test_budget.is_none() && shards.len() != fixture.plan.shards.len() { bail!("the plan has {} shards here and {} in the fixture", shards.len(), fixture.plan.shards.len()); } let (mut sum_gas, mut sum_pgas) = (0u64, 0u64); let mut prev_root = pre_root; let mut prev_link = igneum_prove_core::Carry::default().link(); - for (s, x) in shards.iter().zip(&fixture.plan.shards) { + for (i, s) in shards.iter().enumerate() { let o = &s.output; let (accounts, slots, leaves, hashes) = s.input.witness.stats(); let bytes = bincode::serialize(&s.input)?.len(); - if o.pre_root != x.pre_root || o.post_root != x.post_root || o.receipts_root != x.receipts_root || o.link_in != x.link_in || o.link_out != x.link_out || o.gas_used != x.gas_used || o.pgas_used != x.pgas_used { - bail!("shard {}: the native statement differs from the fixture's plan", o.shard_index); + if let (None, Some(x)) = (test_budget, fixture.plan.shards.get(i)) { + if o.pre_root != x.pre_root || o.post_root != x.post_root || o.receipts_root != x.receipts_root || o.link_in != x.link_in || o.link_out != x.link_out || o.gas_used != x.gas_used || o.pgas_used != x.pgas_used { + bail!("shard {}: the native statement differs from the fixture's plan", o.shard_index); + } } if o.pre_root != prev_root || o.link_in != prev_link { bail!("shard {}: does not continue the previous shard", o.shard_index); @@ -186,6 +197,7 @@ fn run() -> Result<()> { prev_link = o.link_out; sum_gas += o.gas_used; sum_pgas += o.pgas_used; + results.entry("shard_witness_bytes").or_insert_with(|| serde_json::Value::Array(Vec::new())).as_array_mut().unwrap().push(serde_json::Value::from(bytes as u64)); println!( "RESULT shard {} native: txs {}..{} ({} executed, {} skipped), gas {}, pgas {}{}, witness {accounts} accounts {slots} slots {leaves} leaves {hashes} hashes, input {bytes} bytes, pre {} post {}", o.shard_index, diff --git a/tools/ci/prover-socket-check.sh b/tools/ci/prover-socket-check.sh new file mode 100755 index 000000000..17e4a34ed --- /dev/null +++ b/tools/ci/prover-socket-check.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# The root-socket class (5 October 2026, 20:00Z): a PC 2 job ran igneum-prove-host as root inside WSL2, which started +# an sp1-gpu-server whose socket /tmp/sp1-cuda-0.sock stayed root-owned after the job; the live prover (the app's user) +# then failed every shard with "CudaClientError: Connect(PermissionDenied)" until the socket was gone. Rule: every +# playbook that runs the prover host as root on a shared card kills the server AND unlinks its socket (at the start +# and at the end), or runs as the app's user. This check fails CI when a script runs `igneum-prove-host` under a +# `-u root` WSL session without both lines. +set -euo pipefail +cd "$(dirname "$0")/../.." +fail=0 +while IFS= read -r f; do + # a playbook that only runs `--mode id` or `--mode verify` starts no GPU server; the prove modes do + if grep -qE 'igneum-prove-host' "$f" && grep -qE -- '--mode (compressed|chain|aggregate|block|shard|all)\b' "$f" && grep -qE -- '-u root' "$f"; then + if ! grep -qE 'pkill -f sp1-gpu-server' "$f"; then echo "prover-socket: $f runs the prover host as root without killing sp1-gpu-server"; fail=1; fi + if ! grep -qE 'rm -f /tmp/sp1-cuda-' "$f"; then echo "prover-socket: $f runs the prover host as root without unlinking /tmp/sp1-cuda-*.sock"; fail=1; fi + fi +done < <(git ls-files 'tools/**' 'relay/playbooks/**' 'proving/**' 'packaging/**' | grep -E '\.(sh|ps1|mjs)$') +[ "$fail" = 0 ] && echo "prover-socket: every root prover playbook kills the GPU server and unlinks its socket" +exit $fail diff --git a/tools/proving-v1/pc2-chain.ps1 b/tools/proving-v1/pc2-chain.ps1 index cbb8992e6..18c85634c 100644 --- a/tools/proving-v1/pc2-chain.ps1 +++ b/tools/proving-v1/pc2-chain.ps1 @@ -82,6 +82,7 @@ free -m | awk '/Mem:/ {print "RESULT wsl_ram_now total_mb=" `$2 " used_mb=" `$3} STMT=`$(python3 -c "import json; print(json.load(open('`$JOB/chain-results.json'))['segment_statement'])") PROOF=`$(python3 -c "import json; print(json.load(open('`$JOB/chain-results.json'))['segment_proof_file'])") for i in 1 2 3; do `$H --mode verify-segment --proof "`$PROOF" --statement "`$STMT" 2>&1 | grep -E "^RESULT" | sed "s/^/verify-segment run `$i: /"; done +pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock echo "RESULTS-JSON"; cat "`$JOB/chain-results.json"; echo; echo "END" "@ $bashFile = Join-Path $job 'chain.sh' diff --git a/tools/proving-v1/pc2-memory-miner-on.ps1 b/tools/proving-v1/pc2-memory-miner-on.ps1 new file mode 100644 index 000000000..0376a92f7 --- /dev/null +++ b/tools/proving-v1/pc2-memory-miner-on.ps1 @@ -0,0 +1,56 @@ +# Proving v1, the 12 GB and 16 GB requirements, part 2 (5 October 2026): the GPU memory peak of one shard proof on PC 2's +# RTX 5090 WITH THE MINER RUNNING (the mine-and-prove case), the live prover switched off for the run; the full shard +# at S_p and the empty live shard; then the sp1-gpu-server's own option names (strings, --help) to find any memory knob +# the environment did not reach in part 1. Leaves the prover ON. Never stops the miners. +$ErrorActionPreference = 'Continue' +$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' } +if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' } +$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/') +function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') } +function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } } +"RESULT start $(Stamp) prover off for the run: $(Prove $false)" +Start-Sleep -Seconds 45 +"RESULT gpus $(Stamp) $((& nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu,power.draw --format=csv,noheader,nounits 2>$null) -join ' | ')" +$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-pv1-mem2' }; New-Item -ItemType Directory -Force -Path $job | Out-Null +function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } } +$jobW = WslPath $job +$emptyW = WslPath (Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json') +$bash = @" +set -uo pipefail +export PATH="`$HOME/.cargo/bin:`$HOME/.sp1/bin:`$PATH" +CUDA_DIR="`$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "`$CUDA_DIR" ] && export PATH="`$CUDA_DIR/bin:`$PATH" && export LD_LIBRARY_PATH="`$CUDA_DIR/lib64:/usr/lib/wsl/lib:`${LD_LIBRARY_PATH:-}" +stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +JOB='$jobW'; H=/opt/igneum-pv1/igneum-prove-host; FX="`$HOME/igneum-prove-pv1/proving/fixtures" +EMPTY='$emptyW'; FULL="`$FX/block-338-shard1.json" +pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock +echo "RESULT miner_resident_mib `$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits | head -1) (the miner's working set before the prover starts)" +run() { + local name="`$1" fx="`$2"; shift 2 + pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock + local csv="`$JOB/smi-`$name-`$(basename `$fx .json).csv" + nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "`$csv" 2>/dev/null & + local SMI=`$! + local t0=`$(date +%s) + env SP1_PROVER=cuda RUST_LOG=off "`$@" `$H "`$fx" --mode compressed --shard 0 --out "`$JOB/res-`$name-`$(basename `$fx .json).json" > "`$JOB/log-`$name-`$(basename `$fx .json).txt" 2>&1 + local rc=`$? + local wall=`$(( `$(date +%s) - t0 )) + kill `$SMI 2>/dev/null; sleep 1 + local peak=`$(awk -F', *' '{ if (`$2+0 > m) m=`$2+0 } END { print m+0 }' "`$csv") + local line=`$(grep -E "^RESULT compressed shard" "`$JOB/log-`$name-`$(basename `$fx .json).txt" | tail -1 | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/prove_s=\1 bytes=\2 verify_s=\3 \4/') + local err=`$(grep -iE "error|panick|out of memory|OOM|CUDA" "`$JOB/log-`$name-`$(basename `$fx .json).txt" | head -1 | cut -c1-200) + echo "RESULT mineprove cfg=`$name fixture=`$(basename `$fx .json) peak_mib=`$peak samples=`$(wc -l < "`$csv") wall_s=`$wall `${line:-no_result} exit=`$rc `${err:+err=`$err}" +} +run baseline "`$EMPTY" +run baseline "`$FULL" +run baseline "`$EMPTY" +pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock +S="`$HOME/.sp1/bin/sp1-gpu-server" +echo "RESULT server_help `$(`$S --help 2>&1 | tr '\n' ' ' | cut -c1-900)" +echo "RESULT server_env_names `$(strings `$S | grep -oE '^(SP1|MOONGATE|CUDA|GPU|NVIDIA|MEM|TRACE|SHARD|ELEMENT|HEIGHT|FULL)[A-Z0-9_]{3,}$' | sort -u | tr '\n' ' ' | cut -c1-1200)" +echo "RESULT server_words `$(strings `$S | grep -iE 'gpu memory|vram|out of memory|memory pool|pool size|GiB|24 ?GB|16 ?GB|12 ?GB' | sort -u | head -20 | tr '\n' '|' | cut -c1-1200)" +echo "RESULT end_wsl `$(stamp)" +"@ +$bashFile = Join-Path $job 'mem2.sh' +[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false)) +& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') } +"RESULT end $(Stamp) prover back on: $(Prove $true)" diff --git a/tools/proving-v1/pc2-memory-sweep.ps1 b/tools/proving-v1/pc2-memory-sweep.ps1 index 7b1447759..9f265c179 100644 --- a/tools/proving-v1/pc2-memory-sweep.ps1 +++ b/tools/proving-v1/pc2-memory-sweep.ps1 @@ -25,7 +25,7 @@ CUDA_DIR="`$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true) stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } JOB='$jobW'; H=/opt/igneum-pv1/igneum-prove-host; X=/opt/igneum-pv1/igneum-prove-export; FX="`$HOME/igneum-prove-pv1/proving/fixtures" EMPTY='$emptyW'; FULL="`$FX/block-338-shard1.json" -pkill -f sp1-gpu-server 2>/dev/null; sleep 2 +pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock echo "RESULT servers_before `$(pgrep -a sp1-gpu-server | tr '\n' ' ' || echo none)" # the S_p/2 and S_p/4 cuts of block 344 (27 M pgas: 8 and 16 shards), from the package's own export SEQ="`$HOME/igneum-prove-pv1/tools/prove-fixtures/seq.json" @@ -35,7 +35,7 @@ if [ -f "`$SEQ" ]; then fi run() { # name fixture env... local name="`$1" fx="`$2"; shift 2 - pkill -f sp1-gpu-server 2>/dev/null; sleep 2 + pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock local csv="`$JOB/smi-`$name-`$(basename `$fx .json).csv" nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "`$csv" 2>/dev/null & local SMI=`$! @@ -67,7 +67,7 @@ run w1elem26 "`$EMPTY" ELEMENT_THRESHOLD=67108864 HEIGHT_THRESHOLD=2097152 `$W1 [ -f "`$JOB/block-344-half.json" ] && run w1elem26 "`$JOB/block-344-half.json" ELEMENT_THRESHOLD=67108864 HEIGHT_THRESHOLD=2097152 `$W1 [ -f "`$JOB/block-344-quarter.json" ] && run w1elem26 "`$JOB/block-344-quarter.json" ELEMENT_THRESHOLD=67108864 HEIGHT_THRESHOLD=2097152 `$W1 [ -f "`$JOB/block-344-quarter.json" ] && run baseline "`$JOB/block-344-quarter.json" -pkill -f sp1-gpu-server 2>/dev/null +pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock echo "RESULT sweep_end `$(stamp)" "@ $bashFile = Join-Path $job 'sweep.sh' diff --git a/tools/proving-v1/pc2-sp-curve.ps1 b/tools/proving-v1/pc2-sp-curve.ps1 new file mode 100644 index 000000000..66f0b9399 --- /dev/null +++ b/tools/proving-v1/pc2-sp-curve.ps1 @@ -0,0 +1,78 @@ +# Proving v1, the S_p curve (5 October 2026, the coordinator: peak GPU memory against shard size against time, full +# shards): one compressed shard proof per point on PC 2's RTX 5090, the pv1b host (--budget re-plans a fixture at a +# test budget), the live prover off for the run, the GPU server killed and its socket unlinked around every point. +# Points: the empty live shard (280 k cycles); block-56 (200 pgas transfer); the v1-budget shard (fees-v1-shards2, +# 30,000 pgas, about 4.7 M cycles); block 344 (27 M pgas, modexp) cut at one transaction (2.25 M pgas, about 20 M +# cycles) and two (about 40 M); the full shard at S_p (block-338-shard1, 60 M). Set MINERS=stopped when published with +# --stop-miners (the title says which); the script only reports what nvidia-smi sees before it starts. +$ErrorActionPreference = 'Continue' +$urlFile = if ($env:IGNEUM_APP_DIR) { Join-Path $env:IGNEUM_APP_DIR 'app.url' } else { Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' } +if (-not (Test-Path $urlFile)) { $urlFile = Join-Path $env:LOCALAPPDATA 'igneum\app\app.url' } +$base = (Get-Content $urlFile -Raw).Trim().TrimEnd('/') +function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') } +function Prove($on) { try { (Invoke-RestMethod -Method Post -Uri "$base/api/prove" -ContentType 'application/json' -Body (@{on=$on} | ConvertTo-Json -Compress) -TimeoutSec 10) | ConvertTo-Json -Compress } catch { "error: $_" } } +"RESULT start $(Stamp) prover off for the run: $(Prove $false)" +Start-Sleep -Seconds 45 +"RESULT gpus $(Stamp) $((& nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu,power.draw --format=csv,noheader,nounits 2>$null) -join ' | ')" +$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-pv1-sp' }; New-Item -ItemType Directory -Force -Path $job | Out-Null +function WslPath($p) { $w = (& wsl.exe -d Ubuntu-24.04 -u root -- wslpath -a ($p -replace '\\', '/') 2>$null); if ($w) { ($w -replace "`0", '').Trim() } else { '/mnt/c' + ($p.Substring(2) -replace '\\', '/') } } +$jobW = WslPath $job +$emptyW = WslPath (Join-Path $env:LOCALAPPDATA 'igneum\app\jobs\chain-pc2-pv1c\block-83616.json') +$pkg = Join-Path $env:LOCALAPPDATA 'igneum\prove\igneum-prove-wsl2-pv1b\igneum-prove-wsl2' +if (-not (Test-Path $pkg)) { $pkg = Join-Path $env:LOCALAPPDATA 'igneum\prove\igneum-prove-wsl2-pv1b' } +$pkgW = WslPath $pkg +$bash = @" +set -uo pipefail +export PATH="`$HOME/.cargo/bin:`$HOME/.sp1/bin:`$PATH" +CUDA_DIR="`$(ls -d /usr/local/cuda-12.* 2>/dev/null | sort -V | tail -1 || true)"; [ -n "`$CUDA_DIR" ] && export PATH="`$CUDA_DIR/bin:`$PATH" && export LD_LIBRARY_PATH="`$CUDA_DIR/lib64:/usr/lib/wsl/lib:`${LD_LIBRARY_PATH:-}" +stamp() { date -u +%Y-%m-%dT%H:%M:%SZ; } +JOB='$jobW'; PKG='$pkgW'; DEST="`$HOME/igneum-prove-pv1"; LIVE_TARGET="`$HOME/igneum-prove/proving/igneum-prove/target" +pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock +# the pv1b host (--budget) from the fetched package, built against the warm target dir, into /opt/igneum-pv1 +rsync -a --delete --exclude target "`$PKG/package/" "`$DEST/" +find "`$DEST" -name target -prune -o -type f -exec touch {} + 2>/dev/null +cd "`$DEST/proving/igneum-prove" +t0=`$(date +%s) +if ! CARGO_TARGET_DIR="`$LIVE_TARGET" cargo build --release -p igneum-prove-host --features igneum-prove-host/cuda 2>&1 | tail -2; then echo "RESULT build FAILED"; exit 1; fi +cp "`$LIVE_TARGET/release/igneum-prove-host" /opt/igneum-pv1/ +echo "RESULT build `$(stamp) exit 0 in `$(( `$(date +%s) - t0 )) s; host `$(sha256sum /opt/igneum-pv1/igneum-prove-host | cut -c1-16); live /opt/igneum untouched `$(sha256sum /opt/igneum/igneum-prove-host | cut -c1-16)" +H=/opt/igneum-pv1/igneum-prove-host; FX="`$DEST/proving/fixtures" +echo "RESULT miner_resident_mib `$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits | head -1)" +run() { # name fixture budget env... + local name="`$1" fx="`$2" budget="`$3"; shift 3 + pkill -f sp1-gpu-server 2>/dev/null; sleep 2; rm -f /tmp/sp1-cuda-*.sock + local tag="`$name-`$(basename `$fx .json)" + local csv="`$JOB/smi-`$tag.csv" + nvidia-smi --query-gpu=timestamp,memory.used,utilization.gpu --format=csv,noheader,nounits -l 1 > "`$csv" 2>/dev/null & + local SMI=`$! + local t0=`$(date +%s) + local barg=""; [ "`$budget" != "0" ] && barg="--budget `$budget" + env SP1_PROVER=cuda RUST_LOG=off "`$@" `$H "`$fx" --mode compressed --shard 0 `$barg --out "`$JOB/res-`$tag.json" > "`$JOB/log-`$tag.txt" 2>&1 + local rc=`$? + local wall=`$(( `$(date +%s) - t0 )) + kill `$SMI 2>/dev/null; sleep 1 + local peak=`$(awk -F', *' '{ if (`$2+0 > m) m=`$2+0 } END { print m+0 }' "`$csv") + local line=`$(grep -E "^RESULT compressed shard" "`$JOB/log-`$tag.txt" | tail -1 | sed -E 's/.*prove ([0-9.]+) s, proof ([0-9]+) bytes, verify ([0-9.]+) s, ([A-Z ]+);.*/prove_s=\1 verify_s=\3 \4/') + local cyc=`$(grep -E "^RESULT execute shard" "`$JOB/log-`$tag.txt" | tail -1 | sed -E 's/.*: ([0-9]+) cycles.*/\1/') + local plan=`$(grep -E "^RESULT plan:" "`$JOB/log-`$tag.txt" | sed -E 's/RESULT plan: ([0-9]+) shard.*/shards_per_block=\1/') + local sh=`$(grep -E "^RESULT shard 0 native" "`$JOB/log-`$tag.txt" | sed -E 's/.*pgas ([0-9]+).*input ([0-9]+) bytes.*/pgas=\1 witness_bytes=\2/') + local err=`$(grep -iE "^Error|panick|out of memory|OOM" "`$JOB/log-`$tag.txt" | head -1 | cut -c1-160) + echo "RESULT curve cfg=`$name fixture=`$(basename `$fx .json) budget=`$budget `$plan `$sh cycles=`${cyc:-na} peak_mib=`$peak samples=`$(wc -l < "`$csv") wall_s=`$wall `${line:-no_result} exit=`$rc env='`$*' `${err:+err=`$err}" +} +E25="ELEMENT_THRESHOLD=33554432 HEIGHT_THRESHOLD=1048576" +run base '$emptyW' 0 +run base "`$FX/block-56-transfers.json" 0 +run base "`$FX/fees-v1-shards2.json" 0 +run base "`$FX/block-344-shards4.json" 2249264 +run base "`$FX/block-344-shards4.json" 4500000 +run base "`$FX/block-338-shard1.json" 0 +run e25 "`$FX/fees-v1-shards2.json" 0 `$E25 +run e25 "`$FX/block-344-shards4.json" 2249264 `$E25 +run e25 "`$FX/block-338-shard1.json" 0 `$E25 +pkill -f sp1-gpu-server 2>/dev/null; sleep 1; rm -f /tmp/sp1-cuda-*.sock +echo "RESULT curve_end `$(stamp)" +"@ +$bashFile = Join-Path $job 'curve.sh' +[IO.File]::WriteAllText($bashFile, ($bash -replace "`r`n", "`n"), (New-Object System.Text.UTF8Encoding $false)) +& wsl.exe -d Ubuntu-24.04 -u root -- bash (WslPath $bashFile) 2>&1 | ForEach-Object { ($_ -replace "`0", '') } +"RESULT end $(Stamp) prover back on: $(Prove $true)"