diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 32aeab56d..8fed26f91 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -71,8 +71,14 @@ jobs: run: bash tools/ci/copied-sources-check.sh - name: the signer is never piped into head run: bash tools/ci/signer-pipe-check.sh + - name: bash bodies in PowerShell job scripts pass bash -n, the lost-quote class (self-test first, then the tree) + run: bash tools/ci/bash-body-check.sh --self-test && bash tools/ci/bash-body-check.sh + - name: run jobs test their fetched kit before use, the wiped-jobs-folder class (self-test first, then the tree) + run: bash tools/ci/kit-path-check.sh --self-test && bash tools/ci/kit-path-check.sh - name: pinned guest programs match their manifest and are built only by pin-guests.sh run: bash tools/ci/pinned-guests-check.sh + - name: root prover playbooks kill the GPU server and unlink its socket (the root-socket class, 5 October 2026) + run: bash tools/ci/prover-socket-check.sh - name: no secret file names and no 64-hex secrets in the tree (self-test first, then the tree) run: bash tools/ci/no-secrets-check.sh --self-test && bash tools/ci/no-secrets-check.sh - name: faucet unit tests (validation, the daily limits, the signed transaction; keccak, RLP and secp256k1 vectors) diff --git a/app/igneum-app/Cargo.lock b/app/igneum-app/Cargo.lock index f3a5ca195..f2265f7cf 100644 --- a/app/igneum-app/Cargo.lock +++ b/app/igneum-app/Cargo.lock @@ -219,7 +219,7 @@ dependencies = [ [[package]] name = "igneum-app" -version = "0.3.10" +version = "0.3.11" dependencies = [ "ed25519-dalek", "getrandom", diff --git a/app/igneum-app/Cargo.toml b/app/igneum-app/Cargo.toml index bf0f51632..55f3afc05 100644 --- a/app/igneum-app/Cargo.toml +++ b/app/igneum-app/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "igneum-app" -version = "0.3.10" +version = "0.3.11" edition = "2021" description = "Igneum Miner engine: supervises the node, the miner and the GPU workers, and serves the dashboard on 127.0.0.1" license = "MIT" diff --git a/app/igneum-app/resources/igneum-app.rc b/app/igneum-app/resources/igneum-app.rc index 4a7536eba..a0ab6bc0e 100644 --- a/app/igneum-app/resources/igneum-app.rc +++ b/app/igneum-app/resources/igneum-app.rc @@ -6,8 +6,8 @@ 1 ICON "igneum.ico" 1 VERSIONINFO -FILEVERSION 0,3,10,0 -PRODUCTVERSION 0,3,10,0 +FILEVERSION 0,3,11,0 +PRODUCTVERSION 0,3,11,0 FILEFLAGSMASK 0x3fL FILEFLAGS 0x0L FILEOS VOS_NT_WINDOWS32 @@ -20,12 +20,12 @@ BEGIN BEGIN VALUE "CompanyName", "Igneum" VALUE "FileDescription", "Igneum Miner engine" - VALUE "FileVersion", "0.3.10" + VALUE "FileVersion", "0.3.11" VALUE "InternalName", "igneum-app" VALUE "LegalCopyright", "Igneum contributors" VALUE "OriginalFilename", "igneum-app.exe" VALUE "ProductName", "Igneum Miner" - VALUE "ProductVersion", "0.3.10" + VALUE "ProductVersion", "0.3.11" END END BLOCK "VarFileInfo" diff --git a/app/igneum-app/src/config.rs b/app/igneum-app/src/config.rs index f59fee072..23d633ed7 100644 --- a/app/igneum-app/src/config.rs +++ b/app/igneum-app/src/config.rs @@ -87,6 +87,10 @@ pub struct Settings { /// so it includes proof records it never verified (src/verifier.rs). Default off; a found verifier always wins. #[serde(default)] pub proof_verify_trust: bool, + /// Proving v1 step 1 (5 October 2026): the install-time default for `prove` has been applied once (src/provedefault.rs: + /// on when the machine can prove, never switching an explicit on back off). Older installs apply it at their next start. + #[serde(default)] + pub prove_default_applied: bool, } fn one() -> u32 { @@ -98,7 +102,7 @@ fn yes() -> bool { impl Default for Settings { fn default() -> Settings { - Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false } + Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false, prove_default_applied: false } } } diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index cb7c1f625..87d0b961b 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -983,3 +983,19 @@ mod tests { assert_eq!(big.identities, 8); } } + +/// The machine's RAM in MB (the prover default's RAM gate, src/provedefault.rs): Windows through +/// `Win32_OperatingSystem.TotalVisibleMemorySize` (KB), Linux through `/proc/meminfo`, macOS through `sysctl hw.memsize`; +/// None when unreadable (no gate). +pub fn total_ram_mb() -> Option { + if cfg!(windows) { + let out = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", "(Get-CimInstance Win32_OperatingSystem).TotalVisibleMemorySize"]), None, Duration::from_secs(20))?; + return out.replace('\0', "").trim().parse::().ok().map(|kb| kb / 1024); + } + if cfg!(target_os = "linux") { + let text = std::fs::read_to_string("/proc/meminfo").ok()?; + return text.lines().find(|l| l.starts_with("MemTotal:")).and_then(|l| l.split_whitespace().nth(1)).and_then(|kb| kb.parse::().ok()).map(|kb| kb / 1024); + } + let out = run_timeout(Command::new("sysctl").args(["-n", "hw.memsize"]), None, Duration::from_secs(5))?; + out.trim().parse::().ok().map(|b| b / (1024 * 1024)) +} diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 3ad329d00..5278f089b 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -253,6 +253,40 @@ impl Shared { Ok(json!({ "ok": true, "address": w.address, "display": keys::checksum(&w.address), "private_key": w.private_key, "wallet_file": self.wallet_path.display().to_string() })) } + /// Proving v1 step 1 (5 October 2026): once per install, after the cards are known, the prover goes on by itself + /// when this machine can prove (src/provedefault.rs: an NVIDIA card with 12 GB or more, WSL2 on Windows, Linux + /// native, Apple silicon off until measured). An explicit on is never switched off; the line goes to the log and + /// to the Proving tile. Older installs apply it at their first start on this version. + pub fn apply_prove_default(&self) { + let (applied, already_on) = { + let s = self.settings.lock().unwrap(); + (s.prove_default_applied, s.prove) + }; + if applied { + return; + } + let cards = self.state.lock().unwrap().mining.cards.clone(); + let wsl = if cfg!(windows) { Some(crate::wslhost::distro_answers()) } else { None }; + let d = crate::provedefault::decide(&cards, std::env::consts::OS, wsl, crate::detect::total_ram_mb()); + let on = d.on || already_on; + { + let mut s = self.settings.lock().unwrap(); + s.prove = on; + s.prove_default_applied = true; + s.save(&self.settings_path); + } + { + let mut st = self.state.lock().unwrap(); + st.settings.prove = on; + st.proving.enabled = on; + st.proving.default_note = d.line.clone(); + if !on { + st.proving.status = "off".into(); + } + } + self.log(&format!("prover default: {}{}", d.line, if already_on && !d.on { " (left on: it was switched on by hand)" } else { "" })); + } + /// The prover service switch (src/prover.rs); the thread picks it up within 10 s. pub fn set_prove(&self, on: bool) -> Result { { @@ -415,6 +449,42 @@ fn jitter_secs(seed: u64) -> u64 { 5 + (seed.wrapping_mul(2654435761) >> 7) % 56 } +/// The `--prepare-packs` directory the miner gets, relative to the app data folder (its cwd), in the platform's +/// separator: `packs\prepare` on Windows, `packs/prepare` elsewhere. Every worker gets it (program class v3). +pub(crate) fn prepare_packs_arg() -> String { + if cfg!(windows) { "packs\\prepare".to_string() } else { "packs/prepare".to_string() } +} + +/// Seconds after a resume before every enabled card must be mining (a worker takes 10 to 60 s to its first STATUS +/// line with a hash rate; the pack export before it a few seconds more). +pub(crate) const RESUME_CHECK_SECS: u64 = 90; + +/// What the resume rule reads from a miner slot. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(crate) struct ResumeSlot { + /// the watchdog marked the card faulted + pub faulted: bool, + /// a worker process is alive on the slot + pub live: bool, +} + +/// The resume rule: every slot without a live worker is re-armed, faulted or not. (The rule before 5 October 2026 +/// re-armed only faulted slots; `stop_miners("paused")` had cleared every slot's `restart_at`, so a healthy paused +/// card never restarted: PC 2 at 21:25:11Z, the Mac that afternoon.) +pub(crate) fn slots_to_rearm_on_resume(slots: &[ResumeSlot]) -> Vec { + slots.iter().enumerate().filter(|(_, s)| !s.live).map(|(i, _)| i).collect() +} + +/// The check RESUME_CHECK_SECS after a resume: every enabled, present, non-faulted card must report a hash rate +/// above 0 or its name is returned with its state (the caller logs one line per card). +pub(crate) fn resume_check(cards: &[CardState]) -> Vec { + cards + .iter() + .filter(|c| c.enabled && c.present() && c.state != "faulted" && c.hash_now <= 0.0) + .map(|c| format!("{} is not mining {RESUME_CHECK_SECS} s after resume (state {}, pid {}{})", c.name, c.state, c.pid, if c.message.is_empty() { String::new() } else { format!(", {}", c.message) })) + .collect() +} + struct MinerSlot { card: usize, label: String, @@ -443,6 +513,8 @@ pub struct Engine { node: Option, node_external: bool, node_restart_at: Option, + /// resume rule (5 October 2026): when due, every enabled card must be mining or its name goes to the log + resume_check_at: Option, node_started_at: Instant, node_starts: u32, node_restarts: u32, @@ -545,6 +617,7 @@ impl Engine { node: None, node_external: false, node_restart_at: None, + resume_check_at: None, node_started_at: now, node_starts: 0, node_restarts: 0, @@ -711,6 +784,8 @@ impl Engine { self.detect_again = false; self.shared.send(Cmd::Detect); } + // proving v1 step 1: the install-time prover default, once the cards are known + self.shared.apply_prove_default(); } Cmd::ApplyCards(choices) => self.apply_cards(choices), Cmd::Start => self.start(), @@ -729,13 +804,23 @@ impl Engine { self.shared.save_settings(); self.st().mining.paused = false; self.shared.event("ok", "mining resumed"); - // a user action: faulted cards try again - for m in self.miners.iter_mut() { + // the resume rule (5 October 2026, PC 2 at 21:25:11Z and the Mac that afternoon): stop_miners("paused") + // cleared every slot's restart_at and this arm re-armed only FAULTED slots, so a card whose worker had + // simply been stopped stayed "off" at 0 MH/s until the app was relaunched. Now every slot without a + // live worker is re-armed (a faulted one reset first), its pack is exported again before the start + // (prepared = false: the hour may have turned while paused), and a check 90 s later names any + // enabled card that is not mining (`resume_check`). + let views: Vec = self.miners.iter().map(|m| ResumeSlot { faulted: m.watch.faulted().is_some(), live: m.proc.is_some() }).collect(); + let now = Instant::now(); + for i in slots_to_rearm_on_resume(&views) { + let m = &mut self.miners[i]; if m.watch.faulted().is_some() { m.watch.event(0.0, crate::watchdog::Event::Reset); - m.restart_at = Some(Instant::now()); } + m.restart_at = Some(now); + m.prepared = false; } + self.resume_check_at = Some(now + Duration::from_secs(RESUME_CHECK_SECS)); if !self.running { self.start(); } @@ -1385,13 +1470,16 @@ impl Engine { a.push("--network".into()); a.push(r.network.clone()); } + // The miner runs with the app data folder as its cwd: the pack paths stay relative (the miner splits + // --worker-args on spaces, and %LOCALAPPDATA% may carry a space in the user name). Every worker gets + // --prepare-packs (5 October 2026, program class v3: the Metal worker too compiles v3 only from a prepared + // pack, `prepare class=v3 era=`; before this the flag went to the CUDA and OpenCL + // workers only and a Mac would answer `need` lines at the first v3 epoch and stop mining, the 18:23Z class). + a.push("--prepare-packs".into()); + a.push(prepare_packs_arg()); if card.worker != "Metal" { - // The miner runs with the app data folder as its cwd: the pack paths stay relative (the miner splits - // --worker-args on spaces, and %LOCALAPPDATA% may carry a space in the user name). The prebuilt workers - // take the exported pack with --pack and build the next program from --prepare-packs; a worker built - // from source has the program compiled in and exits 42 at the boundary instead. - a.push("--prepare-packs".into()); - a.push("packs\\prepare".into()); + // The prebuilt workers take the exported pack with --pack and build the next program from + // --prepare-packs; a worker built from source has the program compiled in and exits 42 at the boundary. if card.worker == "OpenCL" { a.push("--job-nonces".into()); a.push("2097152".into()); @@ -1426,7 +1514,9 @@ impl Engine { self.miners[i].starts += 1; let seg = if self.miners[i].starts > 1 { format!("-r{}", self.miners[i].starts) } else { String::new() }; let log = self.shared.runtime.log_dir.join(format!("miner-{}-{}{seg}.log", self.miners[i].label, self.stamp)); - let cwd = if card.worker == "Metal" { None } else { Some(self.shared.runtime.app_dir.clone()) }; + // every worker runs with the app data folder as its cwd, so the relative pack paths resolve (the Metal worker + // too, since it takes --prepare-packs now) + let cwd = Some(self.shared.runtime.app_dir.clone()); self.miners[i].prepared = false; // the fleet's per-card kernel tuning (from the signed manifest) reaches the GPU worker through the miner's environment let envs: Vec<(String, String)> = self.ota.tuning_path().map(|p| vec![("IGNEUM_TUNING_FILE".to_string(), p.display().to_string())]).unwrap_or_default(); @@ -2157,6 +2247,18 @@ impl Engine { self.tick_node(now); self.tick_watch(now); self.tick_miners(now); + if let Some(at) = self.resume_check_at { + if now >= at { + self.resume_check_at = None; + let (cards, paused) = { let st = self.st(); (st.mining.cards.clone(), st.mining.paused) }; + if !paused { + for line in resume_check(&cards) { + self.shared.log(&format!("resume: {line}")); + self.shared.event("error", &format!("resume: {line}")); + } + } + } + } self.tick_telemetry(now); self.tick_sweep(now); } @@ -3356,6 +3458,62 @@ pub fn parse_race(body: &str) -> Option { Some(r) } +#[cfg(test)] +mod prepare_packs_tests { + use super::*; + + /// Program class v3 (5 October 2026): the Metal worker needs the prepare directory too, in the platform's + /// separator, relative to the app data folder the miner runs in. + #[test] + fn every_worker_gets_the_prepare_directory_in_the_platform_form() { + let arg = prepare_packs_arg(); + if cfg!(windows) { + assert_eq!(arg, "packs\\prepare"); + } else { + assert_eq!(arg, "packs/prepare", "the Mac's Metal worker gets a forward-slash path"); + } + assert!(!arg.contains(' ') && !arg.starts_with('/'), "relative, no space: the miner splits --worker-args on spaces and the data folder may carry one"); + } +} + +#[cfg(test)] +mod resume_tests { + use super::*; + + fn card(name: &str, enabled: bool, state: &str, hash: f64) -> CardState { + CardState { name: name.into(), vendor: "nvidia".into(), kind: "discrete".into(), enabled, state: state.into(), hash_now: hash, ..Default::default() } + } + + /// The state machine: paused (every slot stopped, restart_at cleared) -> resumed -> every slot without a live + /// worker is re-armed, faulted or not. + #[test] + fn resume_rearms_every_stopped_slot() { + let slots = [ResumeSlot { faulted: false, live: false }, ResumeSlot { faulted: true, live: false }, ResumeSlot { faulted: false, live: true }]; + assert_eq!(slots_to_rearm_on_resume(&slots), vec![0, 1], "the healthy stopped slot and the faulted one restart; the live one is left alone"); + } + + /// The known-failed case (PC 2, 5 October 2026, 21:25:11Z, app 0.3.9: `[ok] mining resumed`, then `0.00 MH/s, + /// waiting` for 20 minutes): one healthy slot, stopped by the pause, not faulted. The old rule re-armed only + /// faulted slots and returned nothing for it; the new rule returns it. + #[test] + fn the_pc2_resume_of_21_25_11z_restarts_under_the_new_rule_and_not_the_old() { + let old_rule = |slots: &[ResumeSlot]| -> Vec { slots.iter().enumerate().filter(|(_, s)| s.faulted).map(|(i, _)| i).collect() }; + let pc2 = [ResumeSlot { faulted: false, live: false }]; + assert!(old_rule(&pc2).is_empty(), "the 0.3.9 rule left the 5090's slot unarmed: this is the defect"); + assert_eq!(slots_to_rearm_on_resume(&pc2), vec![0]); + } + + /// The check 90 s after a resume names every enabled card without a hash rate, and nothing else. + #[test] + fn resume_check_names_the_cards_not_mining() { + let cards = [card("NVIDIA GeForce RTX 5090", true, "off", 0.0), card("AMD Radeon(TM) Graphics", true, "mining", 3.4), card("Intel UHD", false, "off", 0.0), card("RTX 3060", true, "faulted", 0.0)]; + let lines = resume_check(&cards); + assert_eq!(lines.len(), 1, "{lines:?}"); + assert!(lines[0].starts_with("NVIDIA GeForce RTX 5090 is not mining 90 s after resume (state off"), "{}", lines[0]); + assert!(resume_check(&[card("RTX 5090", true, "mining", 118.9)]).is_empty()); + } +} + #[cfg(test)] mod tests { use super::{digest_from_line, parse_race, switches_of, sync_decision, Reading}; diff --git a/app/igneum-app/src/main.rs b/app/igneum-app/src/main.rs index dc37f2bb3..2e7164393 100644 --- a/app/igneum-app/src/main.rs +++ b/app/igneum-app/src/main.rs @@ -29,6 +29,7 @@ mod jobs; mod jobrun; mod jobbuild; mod prover; +mod provedefault; mod verifier; mod wslhost; mod sweep; diff --git a/app/igneum-app/src/provedefault.rs b/app/igneum-app/src/provedefault.rs new file mode 100644 index 000000000..fd9a8ce8d --- /dev/null +++ b/app/igneum-app/src/provedefault.rs @@ -0,0 +1,169 @@ +//! Proving v1 step 1 (5 October 2026, Josh: "open the proving round asap"): the prover is on by default on every +//! mining machine that can prove, decided once per install after the cards are detected (src/engine.rs +//! `apply_prove_default`). The rule, one line each: +//! +//! | Machine | Default | Why (bench-log 5 October 2026, "proving v1", the S_p curve on the RTX 5090, SP1 6.8.1's GPU prover) | +//! |---|---|---| +//! | NVIDIA card with 24 GB or more, mining or not, Windows with WSL2 (Ubuntu-24.04) answering or Linux | on | a full shard at the adopted v1 budget (30,000 pgas, 4.7 M cycles) peaks at 20,434 MiB alone and 22,210 beside the miner (measured on the 5090; approximate for a 24 GB card's own allocation); the prototype shard the devnet proves until its fee switch (6.75 M pgas) peaks at 28,307 MiB alone and 30,039 beside the miner, so until the switch only a 32 GB card proves it and a 24 GB card's prover waits for shards it can hold (the host refuses nothing; a proof that runs out of memory fails and the shard is left) | +//! | NVIDIA card of 16 to 24 GB | off, with the line saying why | the GPU prover's floor is 13,874 MiB for an EMPTY shard, 15,670 beside the miner; a 16 GB card holds no full shard | +//! | NVIDIA card under 16 GB | off | 13,874 MiB does not fit; Josh's 12 GB requirement is open until a prover build with a smaller floor is measured | +//! | Windows under 32 GB of RAM | off, with the line saying why | the WSL2 prover held 7.9 GB on a 63 GB PC; a 16 GB PC would swap | +//! | Windows with a qualifying card but WSL2 silent | off, with the Set up hint | nothing can prove until the distribution exists | +//! | Apple silicon | off | the M5 Max CPU took 41 to 55 s for an EMPTY shard's compressed proof under load and 272 s for a 200-pgas shard; a full shard was never under 60 s (bench-log 4 and 5 October 2026) | +//! | AMD-only (no NVIDIA card) | off, "mines and does not prove" | no zkVM proves on an AMD GPU today (docs/analysis/amd-proving.md); the SP1 CPU prover on PC 1 cost 82 to 87 s core plus 199 to 202 s compressed a shard at a 30 GB RSS whatever the shard size (bench-log, "the SP1 CPU prover on PC 1") | +//! +//! Decided 5 October 2026 (delegated by Josh: "deploy what is absolute best"), docs/plans/proving-v1.md. The default +//! never switches an explicit on back off, and Settings always wins afterwards. + +use crate::state::CardState; + +/// A card that proves, mining or not, needs this much: the adopted v1 shard peaks at 20,434 MiB alone (the S_p curve, +/// 5 October 2026) and 22,210 beside the miner; `nvidia-smi` reports MiB and a 24 GB card reports 24,564, so the +/// test is at 23 GB. (The same value for a mining and an idle card: the floor is the GPU server's, not the miner's.) +pub const MIN_VRAM_MB_MINING: u64 = 23_552; +pub const MIN_VRAM_MB_PROVE_ONLY: u64 = 23_552; +/// What a 32 GB card alone can do that a 24 GB one cannot: the prototype shard (28,307 MiB alone, 30,039 beside the +/// miner), the devnet's shard until its fee switch at DAA 210,000; `nvidia-smi` reports 32,607 for the RTX 5090. +pub const VRAM_MB_PROTOTYPE_SHARD: u64 = 31_000; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Decision { + pub on: bool, + /// One plain sentence for the log and the Proving tile. + pub line: String, +} + +fn gb(mb: u64) -> u64 { + (mb + 512) / 1024 +} + +/// Windows machines under this much RAM stay off until measured (consequences review C4, 5 October 2026): PC 2 at +/// 63 GB had 25.6 GB in use with the WSL2 VM's working set at 7.9 GB while proving; a 16 GB PC would swap. +pub const MIN_RAM_MB_WINDOWS: u64 = 31_000; + +/// The card an aggregation (the chained SP1 recursion, spec 7.8) may run on: 16,751 MiB measured with the miner +/// resident (13.4 GB alone, approximate), so a 24 GB card mining or not; the same gate as the shard prover. +pub fn aggregation_card(cards: &[CardState]) -> Option<&CardState> { + cards.iter().filter(|c| c.vendor == "nvidia" && c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).max_by_key(|c| c.vram_mb) +} + +/// `os` is `std::env::consts::OS` ("windows", "linux", "macos"); `wsl_answers` is read on Windows only; `ram_mb` is the +/// machine's RAM when the platform reports it (None = unknown, no gate). +pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option, ram_mb: Option) -> Decision { + let nvidia: Vec<&CardState> = cards.iter().filter(|c| c.vendor == "nvidia").collect(); + // a mining card needs 20 GB (the measured mine-and-prove peak of 16.8 GB), a card that only proves 16 GB + let able: Vec<&CardState> = nvidia.iter().copied().filter(|c| c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).collect(); + let off = |line: String| Decision { on: false, line }; + if os == "macos" { + return off("proving stays off on Apple silicon: the M5 Max CPU took 41 to 55 s for an empty shard and minutes for a full one; Settings switches it on (CPU, slow)".into()); + } + let Some(best) = able.iter().max_by_key(|c| c.vram_mb) else { + let seen = if nvidia.is_empty() { + "no NVIDIA card".to_string() + } else { + nvidia.iter().map(|c| format!("{} {} GB{}", c.name, gb(c.vram_mb), if c.enabled { ", mining" } else { "" })).collect::>().join(", ") + }; + let why = if nvidia.iter().any(|c| c.vram_mb >= 15_872) { + "a full shard needs a 24 GB card (measured 20.4 GB on the adopted shard size, 13.9 GB for an empty one); this card is under that, so Settings would switch proving on at your own risk" + } else if nvidia.is_empty() { + "this machine mines and does not prove: no zkVM proves on an AMD GPU today, and the CPU prover costs about 5 minutes a shard at a 30 GB RSS (bench-log, the SP1 CPU prover on PC 1); proving needs an NVIDIA card with 24 GB or more" + } else { + "no NVIDIA card with 24 GB or more (the GPU prover's floor is 13.9 GB for an empty shard and 20.4 GB for a full one)" + }; + return off(format!("proving off by default: {why} ({seen})")); + }; + let card = format!("{} ({} GB{})", best.name, gb(best.vram_mb), if best.enabled { ", mining too" } else { ", proving only" }); + if os == "windows" { + if let Some(ram) = ram_mb { + if ram < MIN_RAM_MB_WINDOWS { + return off(format!("proving off by default: {card} qualifies but this PC has {} GB of RAM; proving needs 32 GB on Windows until a smaller PC is measured (the WSL2 prover held 7.9 GB on a 63 GB PC); Settings switches it on", gb(ram))); + } + } + } + let size_note = if best.vram_mb >= VRAM_MB_PROTOTYPE_SHARD { "" } else { "; until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" }; + match os { + "windows" => match wsl_answers { + Some(true) => Decision { on: true, line: format!("proving on by default: {card} with WSL2 (Ubuntu-24.04 answers){size_note}; Settings switches it off") }, + _ => off(format!("proving off: {card} qualifies but WSL2 (Ubuntu-24.04) did not answer; Set up installs it, then Settings switches proving on")), + }, + "linux" => Decision { on: true, line: format!("proving on by default: {card} on Linux (the host runs next to the engine){size_note}; Settings switches it off") }, + other => off(format!("proving off: {card} on {other}, no prover path there; Settings switches it on")), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn card(vendor: &str, name: &str, vram_mb: u64) -> CardState { + CardState { vendor: vendor.into(), name: name.into(), vram_mb, enabled: true, ..Default::default() } + } + fn idle(vendor: &str, name: &str, vram_mb: u64) -> CardState { + CardState { enabled: false, ..card(vendor, name, vram_mb) } + } + + #[test] + fn a_5090_with_wsl2_on_windows_is_on() { + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5090", 32_607), card("amd", "AMD Radeon(TM) Graphics", 512)], "windows", Some(true), Some(63_132)); + assert!(d.on); + assert!(d.line.starts_with("proving on by default: NVIDIA GeForce RTX 5090 (32 GB, mining too) with WSL2"), "{}", d.line); + } + + #[test] + fn windows_without_wsl2_is_off_with_the_setup_hint() { + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(false), Some(65_000)); + assert!(!d.on); + assert!(d.line.contains("did not answer") && d.line.contains("Set up"), "{}", d.line); + assert!(!decide(&[card("nvidia", "RTX 4090", 24_564)], "windows", None, Some(65_000)).on, "an unread probe is not an answer"); + } + + #[test] + fn linux_needs_no_wsl2_and_the_memory_gates_hold() { + // the tiers of the S_p curve: 32 GB on with no note; 24 GB on with the prototype-size note; 16 GB and 12 GB off + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5090", 32_607)], "linux", None, None); + assert!(d.on && !d.line.contains("fee switch"), "{}", d.line); + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, None); + assert!(d.on && d.line.contains("needs 32 GB, so this card proves from the switch on"), "{}", d.line); + assert!(decide(&[idle("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, None).on, "mining or not, 24 GB proves"); + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None); + assert!(!d.on); + assert!(d.line.contains("a full shard needs a 24 GB card") && d.line.contains("RTX 5080 16 GB, mining"), "{}", d.line); + assert!(!decide(&[idle("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None).on, "16 GB holds no full shard even alone"); + let d = decide(&[idle("nvidia", "NVIDIA GeForce RTX 3060", 12_288)], "linux", None, None); + assert!(!d.on); + assert!(d.line.contains("no NVIDIA card with 24 GB or more") && d.line.contains("RTX 3060 12 GB"), "{}", d.line); + assert!(!decide(&[card("nvidia", "NVIDIA GeForce RTX 3080", 10_240)], "linux", None, None).on); + let d = decide(&[card("amd", "Radeon RX 9070 XT", 16_384)], "linux", None, None); + assert!(!d.on && d.line.contains("mines and does not prove"), "{}", d.line); + assert!(decide(&[], "linux", None, None).line.contains("mines and does not prove")); + } + + #[test] + fn apple_silicon_stays_off() { + let d = decide(&[card("apple", "Apple M5 Max", 65_536)], "macos", None, Some(65_536)); + assert!(!d.on); + assert!(d.line.contains("Apple silicon")); + assert!(!decide(&[card("nvidia", "RTX 5090", 32_607)], "macos", Some(true), None).on, "the OS rule comes first"); + } + + #[test] + fn a_windows_pc_under_32_gb_stays_off_and_the_aggregation_card_follows_the_same_gate() { + let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(true), Some(16_300)); + assert!(!d.on); + assert!(d.line.contains("16 GB of RAM") && d.line.contains("needs 32 GB on Windows"), "{}", d.line); + assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(true), None).on, "unknown RAM is not a gate"); + assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, Some(16_300)).on, "the RAM gate is Windows only (the WSL2 VM)"); + let cards = [card("nvidia", "RTX 5080", 16_303), idle("nvidia", "RTX 4070 Ti", 12_282)]; + assert!(aggregation_card(&cards).is_none(), "a mining 16 GB card and an idle 12 GB card cannot aggregate"); + let cards = [card("nvidia", "RTX 5080", 16_303), idle("nvidia", "RTX 4090", 24_564)]; + assert_eq!(aggregation_card(&cards).map(|c| c.name.as_str()), Some("RTX 4090")); + let cards = [card("nvidia", "RTX 5090", 32_607)]; + assert_eq!(aggregation_card(&cards).map(|c| c.vram_mb), Some(32_607)); + } + + #[test] + fn the_biggest_qualifying_card_is_named() { + let d = decide(&[idle("nvidia", "RTX 4090", 24_564), card("nvidia", "RTX 5090", 32_607)], "linux", None, None); + assert!(d.line.contains("RTX 5090 (32 GB, mining too)"), "{}", d.line); + } +} diff --git a/app/igneum-app/src/prover.rs b/app/igneum-app/src/prover.rs index c5cfb1ca0..8f3e548ed 100644 --- a/app/igneum-app/src/prover.rs +++ b/app/igneum-app/src/prover.rs @@ -339,7 +339,9 @@ pub fn start(shared: Arc, bin_dir: PathBuf) { fn loop_forever(shared: Arc, bin_dir: PathBuf) { let mut attempted: HashSet<(String, u32)> = HashSet::new(); + let mut attempted_segments: HashSet = HashSet::new(); let mut tools: Option = None; + let ram_mb = crate::detect::total_ram_mb(); let mut last_probe = Instant::now() - Duration::from_secs(600); let mut submitted: Vec<(u64, String, u32, u128)> = Vec::new(); let mut last_verifier_read = Instant::now() - Duration::from_secs(600); @@ -406,6 +408,20 @@ fn loop_forever(shared: Arc, bin_dir: PathBuf) { } } let Some(t) = tools.as_ref() else { continue }; + // consequences review C22 (5 October 2026): the SP1 CPU prover takes 29.5 to 30.5 GB of RSS and about five + // minutes a shard whatever the shard size (PC 1, bench-log "the SP1 CPU prover on PC 1"); on a machine under + // 32 GB it would swap the node out, so the CPU path is refused here, Settings or not, with the reason + if !t.cuda { + if let Some(ram) = ram_mb { + if ram < crate::provedefault::MIN_RAM_MB_WINDOWS { + set(&shared, |p| { + p.status = "off".into(); + p.message = format!("the CPU prover needs 32 GB of RAM (30 GB measured on PC 1); this machine has {} GB, so proving stays off here", (ram + 512) / 1024); + }); + continue; + } + } + } set(&shared, |p| { p.enabled = true; p.available = true; @@ -466,6 +482,20 @@ fn loop_forever(shared: Arc, bin_dir: PathBuf) { p.assigned = assigned; p.keys = keys.len() as u32; }); + // proving v1 (spec 7.8): the aggregator step, when the node says v1 is active; one attempt a pass + if let Some((label0, _)) = keys.first() { + match aggregate_once(&shared, t, label0, &payout_address(&shared), &mut attempted_segments) { + Ok(Some(msg)) => { + shared.log(&format!("aggregator: {msg}")); + set(&shared, |p| p.segment_note = msg); + } + Ok(None) => {} + Err(e) => { + shared.log(&format!("aggregator: {e}")); + set(&shared, |p| p.segment_note = e); + } + } + } let Some(w) = choose(&work, &attempted) else { set(&shared, |p| { p.status = if submitted.is_empty() { "idle".into() } else { "submitted".into() }; @@ -505,11 +535,15 @@ fn loop_forever(shared: Arc, bin_dir: PathBuf) { if !ok || !fixture.exists() { return Err(format!("exporter: {}", out.lines().rev().find(|l| !l.trim().is_empty()).unwrap_or("failed"))); } - set(&shared, |p| p.message = if t.cuda { "proving on the GPU".into() } else { "proving on the CPU (slow)".into() }); + set(&shared, |p| p.message = if t.cuda { "proving on the GPU".into() } else { "CPU prover: about five minutes a shard, 30 GB of RAM, paid only when no card proves first".into() }); let prover_env = if t.cuda { "cuda" } else { "cpu" }; let (ok, out) = run_tool(&shared, t, &t.host, &[fix_p, "--mode".into(), "compressed".into(), "--shard".into(), w.shard.to_string(), "--prover".into(), payout.clone(), "--out".into(), res_p], &[("SP1_PROVER", prover_env), ("RUST_LOG", "off")], Duration::from_secs(3 * 3600), &dir.join(format!("prove-{}-{}.log", w.number, w.shard))); if !ok || !results.exists() { - return Err(format!("prover: {}", out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed"))); + let last = out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed").to_string(); + // the root-socket class (5 October 2026, PC 2 at 20:00Z and 21:25Z): a job that ran the host as root + // inside WSL2 left /tmp/sp1-cuda-0.sock owned by root, and this user's client cannot open it + let hint = if last.contains("PermissionDenied") { " (a GPU-server socket /tmp/sp1-cuda-*.sock owned by another user, left by a job that ran the prover as root: remove it as that user, or run the socket-fix job)" } else { "" }; + return Err(format!("prover: {last}{hint}")); } let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?; let statement = res["statement"].as_str().ok_or("no statement in the results")?.to_string(); @@ -558,6 +592,124 @@ fn loop_forever(shared: Arc, bin_dir: PathBuf) { } } +/// Proving v1 (spec 7.8): one aggregation attempt. When the node reports v1 active, takes the newest executed +/// segment that is still pending and not yet attempted here, needs one shard proof per shard of every block in +/// this node's pool (`igneum_getProofBytes`, a verified one when there is one) and, when the previous segment is +/// proven, its aggregated proof (`igneum_getSegmentProofBytes`); runs `igneum-prove-host --mode aggregate` over the +/// run of blocks (one process, one key setup), checks the public values against the node's native statement +/// (every field but `provers`), signs the record with the first key's label and submits it. Returns a line for +/// the log and the tile, or None when there is nothing to do. +fn aggregate_once(shared: &Shared, t: &Tools, label: &str, payout: &str, attempted: &mut HashSet) -> Result, String> { + let hexu = |x: &Value| x.as_str().and_then(|s| u64::from_str_radix(s.trim_start_matches("0x"), 16).ok()).unwrap_or(0); + let st = evm_rpc(shared, "igneum_getProvingStatus", json!([]), Duration::from_secs(10))?; + let v1 = &st["v1"]; + if !v1["active"].as_bool().unwrap_or(false) || v1["start"].is_null() { + return Ok(None); + } + // the card gate (consequences review C2): a chained aggregation peaked at 16,751 MiB with the miner resident; the + // prover's own 24 GB gate applies; without such a card this machine proves shards and never aggregates + { + let cards = shared.state.lock().unwrap().mining.cards.clone(); + if crate::provedefault::aggregation_card(&cards).is_none() { + return Ok(Some("no aggregation on this machine: it needs a 24 GB card (16.8 GB measured with the miner resident); shards still prove".into())); + } + } + let tip = hexu(&evm_rpc(shared, "eth_blockNumber", json!([]), Duration::from_secs(10))?); + let mut seg = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{tip:#x}")]), Duration::from_secs(10))?; + if !seg["executed"].as_bool().unwrap_or(false) { + let first = hexu(&seg["first"]); + if first == 0 || first - 1 < hexu(&v1["start"]) { + return Ok(None); + } + seg = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{:#x}", first - 1)]), Duration::from_secs(10))?; + } + let (first, last) = (hexu(&seg["first"]), hexu(&seg["last"])); + if seg["status"]["status"].as_str() != Some("pending") || attempted.contains(&first) || payout.len() != 42 { + return Ok(None); + } + let dir = shared.runtime.app_dir.join("proving").join(format!("seg-{first}")); + let _ = std::fs::create_dir_all(&dir); + let as_host_path = |p: &Path| if t.wsl { wsl_path(p) } else { p.display().to_string() }; + // the shard proofs, one per shard of every block, from this node's pool + let mut groups: Vec = Vec::new(); + let mut missing: Vec = Vec::new(); + for b in seg["blocks"].as_array().cloned().unwrap_or_default() { + let n = hexu(&b["number"]); + let shards = b["shards"].as_u64().unwrap_or(0) as u32; + let have = b["shardProofs"].as_array().cloned().unwrap_or_default(); + let mut files = Vec::new(); + for i in 0..shards { + let pick = have.iter().find(|e| e["shard"].as_u64() == Some(i as u64) && e["verified"] == json!(true)).or_else(|| have.iter().find(|e| e["shard"].as_u64() == Some(i as u64))); + let Some(e) = pick else { + missing.push(format!("{n}/{i}")); + continue; + }; + let got = evm_rpc(shared, "igneum_getProofBytes", json!([format!("{n:#x}"), i, e["keyHash"]]), Duration::from_secs(60))?; + let hex = got["proof"].as_str().ok_or("no proof bytes")?.trim_start_matches("0x").to_string(); + let bytes: Vec = (0..hex.len() / 2).map(|k| u8::from_str_radix(&hex[2 * k..2 * k + 2], 16).unwrap_or(0)).collect(); + let f = dir.join(format!("b{n}-s{i}.bin")); + std::fs::write(&f, bytes).map_err(|e| e.to_string())?; + files.push(as_host_path(&f)); + } + groups.push(files.join(",")); + } + if !missing.is_empty() { + return Ok(Some(format!("segment {first}..{last}: waiting for shard proofs {} in this node's pool", missing.join(" ")))); + } + // the previous segment's aggregated proof, when the chain continues + let prev = &seg["previous"]; + let (prev_file, expected_pv) = if prev.is_null() { + (None, seg["publicValuesFresh"].as_str().unwrap_or("").to_string()) + } else if prev["proofInPool"] == json!(true) { + let got = evm_rpc(shared, "igneum_getSegmentProofBytes", json!([prev["first"], prev["keyHash"]]), Duration::from_secs(60))?; + let hex = got["proof"].as_str().ok_or("no segment proof bytes")?.trim_start_matches("0x").to_string(); + let bytes: Vec = (0..hex.len() / 2).map(|k| u8::from_str_radix(&hex[2 * k..2 * k + 2], 16).unwrap_or(0)).collect(); + let f = dir.join("prev-aggregated.bin"); + std::fs::write(&f, bytes).map_err(|e| e.to_string())?; + (Some(as_host_path(&f)), seg["publicValuesContinuing"].as_str().unwrap_or("").to_string()) + } else { + return Ok(Some(format!("segment {first}..{last}: the previous segment's proof is not in this node's pool; waiting"))); + }; + attempted.insert(first); + let parent = seg["blocks"][0]["parentHash"].as_str().ok_or("no parent hash")?.to_string(); + let last_hash = seg["blocks"].as_array().and_then(|a| a.last()).and_then(|b| b["hash"].as_str()).ok_or("no last hash")?.to_string(); + let results = dir.join("results.json"); + let started = Instant::now(); + set(shared, |p| p.message = format!("aggregating segment {first}..{last} ({})", if t.cuda { "GPU" } else { "CPU, slow" })); + let mut args: Vec = vec!["--mode".into(), "aggregate".into(), "--proofs".into(), groups.join(";"), "--parent".into(), parent, "--out".into(), as_host_path(&results)]; + if let Some(pf) = prev_file { + args.push("--prev".into()); + args.push(pf); + } + let (ok, out) = run_tool(shared, t, &t.host, &args, &[("SP1_PROVER", if t.cuda { "cuda" } else { "cpu" }), ("RUST_LOG", "off")], Duration::from_secs(2 * 3600), &dir.join("aggregate.log")); + if !ok || !results.exists() { + return Err(format!("segment {first}..{last}: aggregator: {}", out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed"))); + } + let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?; + let pv = res["segment_public_values"].as_str().ok_or("no public values in the results")?.to_string(); + let proof_sha = res["segment_proof_sha256"].as_str().ok_or("no proof hash in the results")?.to_string(); + let proof_file = res["segment_proof_file"].as_str().ok_or("no proof file in the results")?.to_string(); + // the node's native statement, every field but provers (bytes 236..268 of the 340) + let strip = |h: &str| { let h = h.trim_start_matches("0x"); if h.len() == 680 { format!("{}{}", &h[..472], &h[536..]) } else { h.to_string() } }; + if strip(&pv) != strip(&expected_pv) { + return Err(format!("segment {first}..{last}: the aggregated statement differs from the node's native statement (it would be vetoed); ours {} node {}", &pv[..66.min(pv.len())], &expected_pv[..66.min(expected_pv.len())])); + } + let proof_path = if t.wsl { PathBuf::from(proof_file.replace("/mnt/c/", "C:/")) } else { PathBuf::from(proof_file) }; + let sg = crate::detect::run_timeout(crate::platform::quiet(&mut Command::new(&t.miner)).args(["sign-segment-record", label, &chain_name(shared), &first.to_string(), &last.to_string(), &last_hash, payout, &pv, &proof_sha]), None, Duration::from_secs(20)).ok_or("sign-segment-record did not run")?; + let signed: Value = serde_json::from_str(sg.lines().last().unwrap_or("")).map_err(|_| format!("sign-segment-record: {}", sg.trim()))?; + let record = signed["record"].as_str().ok_or("sign-segment-record gave no record")?.to_string(); + let proof = std::fs::read(&proof_path).map_err(|e| format!("proof file {}: {e}", proof_path.display()))?; + let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::()); + let r = evm_rpc(shared, "igneum_submitSegmentRecord", json!([{ "record": record, "proof": proof_hex }]), Duration::from_secs(60))?; + if !r["accepted"].as_bool().unwrap_or(false) { + return Err(format!("segment {first}..{last}: record refused: {}", r["reason"].as_str().unwrap_or("?"))); + } + let secs = started.elapsed().as_secs_f64(); + shared.event("proving", &format!("segment {first}..{last} aggregated and submitted in {secs:.0} s (chain_len {})", res["segment_chain_len"])); + set(shared, |p| p.aggregated += 1); + Ok(Some(format!("segment {first}..{last} aggregated in {secs:.0} s and submitted; paid when a block carries it"))) +} + /// Windows: runs the WSL2 setup from the payload (`wsl2/setup-wsl.sh` next to the engine) in a window of its own; /// the user watches it and reboots when it asks. Elsewhere there is nothing to set up. pub fn setup(shared: &Shared) -> Result { diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index 356b0cf8d..5cf32e57a 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -196,6 +196,11 @@ pub struct ProvingState { /// the pinned guests' ids (`igneum-prove-host --mode id`): the shard program and the aggregator; empty until read pub program_id: String, pub aggregator_id: String, + /// proving v1 step 1: the install-time default's one plain line (why proving is on or off on this machine) + pub default_note: String, + /// proving v1: segment records this machine aggregated and submitted, and the aggregator's last line + pub aggregated: u32, + pub segment_note: String, } #[derive(Clone, Serialize, Default)] diff --git a/app/igneum-app/src/wslhost.rs b/app/igneum-app/src/wslhost.rs index b01b928f3..2c3685b76 100644 --- a/app/igneum-app/src/wslhost.rs +++ b/app/igneum-app/src/wslhost.rs @@ -41,6 +41,43 @@ pub fn candidates(bin_dir: &Path) -> Vec { } /// The candidates as one line for a message. +/// Whether the distribution answers at all (`wsl.exe -d Ubuntu-24.04 -- echo ` within 30 s): the install-time +/// prover default (src/provedefault.rs) needs WSL2 on Windows before it switches proving on. Elsewhere: false. +#[allow(dead_code)] // also compiled into src/bin/prove-verify.rs, which does not call it +pub fn distro_answers() -> bool { + if !cfg!(windows) { + return false; + } + // self-contained (this file is also compiled into src/bin/prove-verify.rs, which has no detect or platform module) + let mut cmd = std::process::Command::new("wsl"); + cmd.args(["-d", DISTRO, "--", "echo", "igneum-wsl-answers"]).stdin(std::process::Stdio::null()).stdout(std::process::Stdio::piped()).stderr(std::process::Stdio::null()); + #[cfg(windows)] + { + use std::os::windows::process::CommandExt; + cmd.creation_flags(0x0800_0000); // CREATE_NO_WINDOW + } + let Ok(mut child) = cmd.spawn() else { return false }; + let Some(out) = child.stdout.take() else { return false }; + let reader = std::thread::spawn(move || { + let mut s = String::new(); + let _ = std::io::Read::read_to_string(&mut std::io::BufReader::new(out), &mut s); + s + }); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30); + loop { + match child.try_wait() { + Ok(Some(_)) => break, + Ok(None) if std::time::Instant::now() < deadline => std::thread::sleep(std::time::Duration::from_millis(100)), + _ => { + let _ = child.kill(); + let _ = child.wait(); + break; + } + } + } + reader.join().map(|o| o.replace('\0', "").contains("igneum-wsl-answers")).unwrap_or(false) +} + pub fn candidates_text(bin_dir: &Path) -> String { candidates(bin_dir).join(", ") } diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index 92b396ea2..af1bb188b 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -1005,7 +1005,8 @@ if (typeof document !== 'undefined') (function () { setText('pv-state', w.word + (pv.backend && enabled && pv.available ? ' (' + pv.backend.toUpperCase() + ')' : '')); setText('pv-state-sub', w.sub); var cell = $('pv-state').parentNode; cell.classList.toggle('ok', w.tone === 'on' || w.tone === 'ok'); cell.classList.toggle('bad', w.tone === 'bad'); - setText('pv-note', w.note); + // proving v1: when off by the install-time default, the default's own line says why (src/provedefault.rs) + setText('pv-note', (!enabled && pv.default_note) ? 'Off. ' + pv.default_note + '.' : w.note); setText('pv-assigned', String(pv.assigned || 0)); setText('pv-submitted', String(pv.submitted || 0)); setText('pv-paid', String(pv.paid || 0)); diff --git a/app/igneum-app/ui/index.html b/app/igneum-app/ui/index.html index ec32a9637..f878ae005 100644 --- a/app/igneum-app/ui/index.html +++ b/app/igneum-app/ui/index.html @@ -206,7 +206,7 @@

Prove shards on this machine

-

Every block on Igneum is turned into a short mathematical proof, in pieces called shards. The chain assigns shards to your keys; this machine proves them and is paid for each one.

+

Every block on Igneum is turned into a short mathematical proof, in pieces called shards. The chain assigns shards to your keys; this machine proves them and is paid for each one. On by default on an NVIDIA card with 24 GB or more (a full shard needs 20.4 GB of GPU memory, measured); off on a Mac, whose CPU prover is slow.

diff --git a/app/windows/version.h b/app/windows/version.h index 4c8414249..5de915941 100644 --- a/app/windows/version.h +++ b/app/windows/version.h @@ -3,6 +3,6 @@ // packaging/windows/Igneum-Miner.iss when the app version moves. Include guards, not #pragma once: rc.exe reads it too. #ifndef IGNEUM_HOST_VERSION_H #define IGNEUM_HOST_VERSION_H -#define IGNEUM_HOST_VERSION_STR "0.3.10" -#define IGNEUM_HOST_VERSION_RC 0,3,10,0 +#define IGNEUM_HOST_VERSION_STR "0.3.11" +#define IGNEUM_HOST_VERSION_RC 0,3,11,0 #endif diff --git a/docs/analysis/amd-proving.md b/docs/analysis/amd-proving.md new file mode 100644 index 000000000..cccba4aee --- /dev/null +++ b/docs/analysis/amd-proving.md @@ -0,0 +1,131 @@ +# Proving on AMD and Apple cards: what exists, what the CPU can do, what to tell the public + +5 October 2026, from Josh's two questions that evening: "test proving on the amd card?" and "can we test proving on +mac?". PC 1 holds an RTX 5090 and an RX 9070 XT (gfx1201, 16 GB) in an eGPU; this Mac is an M5 Max. The prover is +SP1 (`proving/igneum-prove`, `docs/plans/proving-v0.md`, `proving-v1.md`), run on the GPU only through SP1's CUDA +server. Every figure below is measured (with its bench-log entry or job id) or cited (with its file or page); the +rest is labelled approximate. Status words follow `docs/spec/00-overview.md` 0.2. + +**The answer in three lines.** No zkVM proves on an AMD GPU on 5 October 2026: not SP1, not RISC Zero, not Jolt, not +OpenVM, and the ICICLE library underneath them has no AMD backend either. Apple silicon has a shipped Metal prover in +RISC Zero and a Metal backend in ICICLE, but SP1, the prover Igneum runs, is CPU-only on a Mac. So an AMD-only or +Apple-only machine mines and does not prove on its card; it can prove on its CPU, at the times measured in section 2. + +## 1. The backends (read 5 October 2026, 20:30 to 20:50 UTC) + +| Prover | Version read | CPU | NVIDIA (CUDA) | AMD (ROCm or HIP) | Apple (Metal) | Vulkan or WebGPU | Where it says so | +|---|---|---|---|---|---|---|---| +| SP1 (ours) | v6.8.1, 24 Sep 2026 (pinned); `dev` head 318dd530, 28 Sep 2026 | yes; AVX2 and AVX-512 on x86 through Plonky3 | yes: `sp1-gpu-server`, "Compute Capability 8.0 or higher", "24GB or more VRAM", "the CUDA 12 runtime and a compatible NVIDIA driver", Linux x86_64 | **no** | **no** | **no** | docs.succinct.xyz, SP1 docs "Hardware acceleration" page; `crates/sdk/src/lib.rs` (`pub mod cpu`, `mock`, `light`, `#[cfg(feature = "cuda")] pub mod cuda`, `#[cfg(feature = "network")] pub mod network`: no other backend module); `sp1-gpu/README.md` (`CUDA_ARCHS` 89, 90, 100, 120; NTT by NVIDIA cuPQC or sppark); release notes v6.2.3 to v6.8.1 (the only backend line: "add optional cuPQC NTT backend", v6.8.0); a code search of the repository on 5 October: "rocm" 0 files, "metal" 0, "vulkan" 0, "webgpu" 0; `cuobjdump` of sp1-gpu-server 6.8.1: sm_80, 86, 89, 90, 100, 120 and compute_120 PTX, nothing else (`docs/bench-log.md`, "proving v1", 5 October 2026) | +| sppark (SP1's NTT fallback, vendored at `sp1-gpu/crates/sys/sppark`) | `main` README, read 5 October 2026 | | yes: "x86_64 with Nvidia's Volta+ GPU hardware platforms on Linux and Windows" | "A limited support for AMD's RDNA and CDNA GPUs is provided" (upstream README). SP1's tree carries no HIP build: the 0 "rocm" files above, and `sp1-gpu/crates/sys/sppark/util/gpu_t.cuh` is CUDA only | no | no | github.com/supranational/sppark README; the SP1 files named | +| RISC Zero | latest release v3.0.6, 17 Jul 2026 (a v5.0.0-rc.1 of 15 Jan 2026 is also on the releases page) | yes, "nearly any modern CPU (x86 or ARM)" | yes, "RISC Zero targets NVIDIA GPUs using the CUDA framework" | **no** ("rocm", "vulkan": 0 files in the repository) | **yes**: `metal = ["prove"]` in `risc0/zkvm/Cargo.toml`; kernels in `risc0/sys/kernels/zkp/metal/*.metal` (zk, fri, mix, sha); docs: "RISC Zero will use the integrated Metal compute cores" on Apple silicon. The Groth16 wrapper "only works on x86 architecture, and so Apple Silicon is currently unsupported (even via Docker)" | no | dev.risczero.com "Local proving"; `risc0/zkvm/Cargo.toml` features `cuda = [... risc0-zkp/cuda ...]`, `metal = ["prove"]` | +| Jolt (a16z) | v0.3.0-alpha, 1 Oct 2025; "Jolt is in alpha and is not suitable for production use" | yes, "state-of-the-art performance on CPU" | no | **no** | a **draft** PR #1733 (opened 3 Aug 2026, not merged): titled "feat: Metal GPU backend (Apple Silicon)" and marked experimental: 92 Metal kernels, "2.12x speedup at 2^20 scale" on an M4 mini and "3.20x vs same-binary CPU" on an M5 Max, "Apple Silicon + macOS only. No CI coverage" | no | github.com/a16z/jolt README and book (jolt.a16zcrypto.com); PR #1733 | +| OpenVM | v2.0.2, 14 Aug 2026 | yes | yes: `cuda-backend` (v1.4.2 notes), "Improves the Halo2 GPU prover" (v2.0.2) | **no** | **no** | no | github.com/openvm-org/openvm releases | +| ICICLE (Ingonyama; the GPU library behind several provers, not SP1) | v4.0.0, 11 Jul 2025 | yes (MIT) | yes, "CUDA (for NVIDIA GPUs)" | **no** backend listed | yes, "Metal (for Apple Silicon GPUs)"; both under a special licence with a free research licence | Vulkan in the build system (PR #735, merged Jan 2025) and a draft "Vulkan NTT" PR #1019 (Jul 2025, "still wip"); nothing installable | dev.ingonyama.com "Install GPU backend"; the releases page; the README ("backends ... are distributed under a special license") | + +Said plainly: **on 5 October 2026 no zkVM proves on an AMD GPU.** The only AMD code in the whole chain is sppark's +limited HIP path, which SP1 does not build. For Apple silicon the answer is split: RISC Zero ships a Metal prover +and ICICLE a Metal backend; SP1, Jolt and OpenVM do not. Nothing read tonight names an AMD plan with a date. + +## 2. The CPU fallback, measured + +SP1's CPU prover is the path an AMD-only or Apple-only machine has today. Three machines, the same pinned guests +(shard program id `0x2b1a81cb...`, aggregator `0x474678f3...`, pinned 2026-10-05T16:20:38Z), `SP1_PROVER=cpu`, +`--mode shard --shard 0` (execute, core proof, compressed proof, each verified). The RTX 5090 rows are the reference. + +| Fixture (SP1 cycles) | Stage | PC 1 CPU, miner running on both cards (job `cpu-prove-pc1-small2`) | Apple M5 Max CPU (4 October, loaded; bench-log) | RTX 5090 (bench-log) | +|---|---|---|---|---| +| block-56-transfers-3shards shard 0, 200 pgas (315 k) | core | 82.5 s, 7,310,257 B, verify 0.210 s | 83.1 s, 7,310,257 B | not run on the 5090; the nearest rows are block-78 below and an empty live shard: 7.0 to 7.7 s compressed with the miner on the card (5 Oct, `chain-pc2-pv1b`, `pv1c`) | +| | compressed | 199.2 s, 1,272,897 B, verify 0.035 s; 312 s wall for setup 22.8 s, execute 0.14 s, core, compressed | 272.3 s, 1,272,897 B | | +| | peak RSS, CPU | 29.5 GB peak RSS; 978% CPU (9.8 of 16 cores), user 2,516 s, system 537 s | not recorded | | +| block-78-increment, 2 transactions (626 k) | core | 87.0 s, 7,317,857 B, verify 0.209 s | 22.0 s, 7.3 MB (3 October, v0 guest) | 1.4 s (4 October, mining paused) | +| | compressed | 202.3 s, 1,272,897 B, verify 0.034 s; 322 s wall (setup 21.8 s) | 55.7 s, 1.27 MB | 2.7 s | +| | peak RSS, CPU | 30.5 GB peak RSS; 979% CPU, user 2,616 s, system 541 s | not recorded | | +| block-338-shard1, one shard at `S_p` (60.8 M) | core | **not run**, by the PC 1 scheduler's decision at 21:05Z (PC 1's time tonight belongs to the Counter ASIC 2.0 gates; the job `cpu-prove-pc1-sp`, script `tools/amd-prove/pc1-cpu-prove-sp.ps1`, is written and unpublished). Extrapolation, approximate: 60.8 M cycles is about 29 SP1 shards of 2^21 cycles where the small fixtures are one, so the core proof alone is about 29 x 80 s, 40 min, and the compressed recursion over 29 shard proofs adds hours; the floor from the 5090's own ratios (6x on core, 4x on compressed between block-78 and `S_p`) is 9 min core and 13 min compressed. Either way far outside every deadline | not run on the CPU (execute alone 6.9 s) | 8.3 s | +| | compressed | not run (see the core cell) | not run | 10.9 s with the card to itself (4 Oct); 33.0 s with the miner running (5 Oct, `memminer-pc2-pv1`); 7.3 to 7.7 s per EMPTY shard with the miner running (`chain-pc2-pv1c`) | +| | peak RSS, CPU | not run; at least the 30 GB of the small rows | | GPU peak 28,295 MiB alone, 30,039 MiB beside the miner | + +PC 1: Windows 11, WSL2 Ubuntu 24.04 as root, 16 cores and 46,994 MB visible to the VM, the Igneum Miner app 0.3.9 mining on the RTX 5090 (89% mean utilisation through both runs, 59 to 70% minimum: the miner, untouched) and on the RX 9070 XT (not visible to nvidia-smi, mining through the app's OpenCL worker). Job `cpu-prove-pc1-small2`, 20:49:00Z to 20:59:49Z, 649 s wall including a 6-s warm build; the host built without the `cuda` feature from the hosted package `igneum-prove-wsl2-pv1b.zip`, `--mode id` the pinned pair. The first job, `cpu-prove-pc1-small` (20:44 to 20:46Z), built the host cold in 126 s and proved nothing: an apostrophe inside a single-quoted awk program ended the quote, bash refused the whole loop and the job reported exit 0. The class fix: `tools/amd-prove/check-job-bash.sh` runs `bash -n` on the bash body of a PowerShell job before it is published, and the job itself runs `bash -n` inside the distro before the run; both were shown to fire on the bad body and pass the fixed one. Host RAM in the VM: 968 MB used before, 2,351 MB after; the prover's own peak 29.5 to 30.5 GB. + +Mac, fresh run tonight: not taken. The Mac measure lock was held from 20:31Z (a read-width `packbench` under `measure`, three build slots, then a 1,500-s proving-v1 network under `run`) and did not free inside the 10-minute window the coordinator set, so the Apple column is the 4 October rows (M5 Max, 18 cores, 64 GB, load 38 to 47, `nice -n 19`): the same host modes on the same fixture, under heavier load than PC 1 tonight. The Mac's RAM peak was not recorded on 4 October; PC 1's 30 GB says a Mac needs more than 32 GB for the CPU prover, which a 64 GB M5 Max has and a 16 or 24 GB Mac does not. + +The deadlines a CPU proof has to fit (all in the spec and the v1 plan): the exclusive window of an assigned shard is +10 s of DAA time (spec 7.2 item 3; after it anyone may prove and be paid first); the litepaper promises the block's +proof "within about a minute"; the launch target is 20 to 60 s behind the tip; from proving v1 a segment nobody has +proven in `T` = 600 DAA s (10 min) pays nothing (`docs/plans/proving-v1.md`, decisions). So a CPU shard proof is +useful only if it lands inside 10 min and competitive only if it lands inside about a minute. + +Reading. On PC 1 the CPU proof of the smallest shard (315 k cycles) and of the two-transaction block (631 k cycles) cost the same: 82.5 and 87.0 s core, 199.2 and 202.3 s compressed. Doubling the cycles added 4.5 s to the core proof and 3.1 s to the compressed one, so about 280 s of every CPU proof is fixed cost (the recursion that turns the core proof into the 1.27 MB compressed proof the chain carries), and no shard size removes it. Against the deadlines: 282 s a shard (core plus compressed, the client already set up, as the app's loop runs it) is 28x the 10-s assignment window, 4.7x the minute the litepaper promises, and inside the 600-s unproven deadline of v1 with 5 min to spare; but an NVIDIA card proves the same shard in 2.7 to 7.7 s, so a CPU prover only ever wins a shard that no card has taken in 10 minutes. The Mac's 83.1 and 272.3 s of 4 October have the same shape. The RAM peak of 29.5 to 30.5 GB is the second finding: the SP1 CPU prover does not fit a 16 GB machine at all, and WSL2 gives a Windows VM half the host's RAM by default, so the CPU path needs a 64 GB Windows PC or a 32 GB Linux or Mac machine. The `S_p` shard on the CPU can only be slower (the 5090 takes 6x longer at `S_p` than on block-78: 8.3 s against 1.4 s core); it was not run tonight (the scheduler kept PC 1 for the Counter ASIC 2.0 gates) and could not change the conclusion. + +## 3. What this means for each tier (the every-number rule, CLAUDE.md 5 October 2026) + +| Tier | Mines | Proves on the card | The 20% proving-pool share (spec 2.5) | What the software does today | +|---|---|---|---|---| +| AMD-only home miner, one card of 8, 12 or 16 GB (an RX 9070 XT is 16 GB), Windows or Linux | yes (OpenCL worker, `proto-opencl`; PC 1's 9070 XT mines on the devnet) | **no**: no prover exists for the card | **lost**, unless CPU proving at a small shard size becomes a tier (section 4a) | the rig installer: `prover_decision` in `packaging/linux/bin/igneum-rig-lib.sh` (branch `rig-install`) skips every non-NVIDIA card (`[[ "$vendor" == nvidia ]] \|\| continue`) and prints "proving off by default: no NVIDIA card (no CUDA prover for AMD or Intel yet)"; the app: `provedefault.rs` (branch `proving-v1`) considers NVIDIA cards only. Both already right; neither offers the CPU path | +| Apple silicon (M-series, unified memory) | yes: the M5 Max at 26.7 MH/s (bench-log 4 October, "first hourly program swap", Metal `prepare 1` row) | **no** with SP1; RISC Zero and ICICLE have Metal, SP1 does not | **lost** today; a Metal prover behind the swappable interface would restore it (section 4b) | `provedefault.rs`: "proving stays off on Apple silicon: the M5 Max CPU took 41 to 55 s for an empty shard and minutes for a full one; Settings switches it on (CPU, slow)". Right | +| Mixed rig (NVIDIA and AMD cards in one box) | every card | the NVIDIA cards prove for the box; the AMD cards mine | kept, earned by the NVIDIA cards | the rig installer picks the biggest NVIDIA card (`prover_decision`, `PROVER_CARD` overrides), pauses its miner under 20 GB, keeps it mining at 20 GB or more; the AMD cards get a miner unit each. **The prover unit must never select an AMD card**: it does not (the vendor filter above), and that filter is now a stated requirement, not an accident | +| NVIDIA home miner, 8 or 12 GB | yes | no on this SP1 build (13.9 GB floor on an empty shard, `memsweep-pc2-pv1`) | lost unless the shard size moves | unchanged from `proving-v1.md` | +| NVIDIA 16 GB | yes | prove-only, miner paused per shard | kept | unchanged | +| NVIDIA 24 or 32 GB | yes | mines and proves (peak 16.8 GB on empty shards, 30.0 GB on a full prototype shard beside the miner) | kept | unchanged | +| Pool user | through the pool | the pool's own NVIDIA cards prove the shards assigned to the pool's keys (approximate: the pool protocol, spec 09, does not yet say who proves) | by the pool's rules | open, spec 09 | + +## 4. The options + +### 4a. CPU proving at a small shard size, as a tier + +What it is: an AMD-only or Apple machine proves shards cut at a smaller budget than `S_p` on its CPU, through the +same host (`SP1_PROVER=cpu`; the host's `--budget` re-plan from branch `proving-v1`, commit c2544be, cuts a fixture at +any budget). The miner keeps the card; the prover takes the CPU. + +What the numbers say: the fixed cost kills it. 282 s a shard on a 16-core PC and 355 s on the loaded M5 Max, with 30 GB of RAM, at the smallest shard there is; the time sits in the compressed-proof recursion, not in the cycles, so cutting shards smaller does not help, and the launch deadline (20 to 60 s behind the tip) is missed by 5x. It fits only the v1 unproven deadline (600 s), which pays a CPU prover only when no card has proven the shard in 10 minutes: on a chain with one NVIDIA prover that never happens. Recommendation: **no CPU tier**. Settings may still switch the CPU prover on (it does on macOS today), and the Proving tile must then say the proof takes about five minutes and is paid only when no card proves first. + +What it costs the chain: a block cut into more, smaller shards costs more aggregation work (the aggregator guest +verifies one deferred proof per shard; 1.66 M cycles for four shards on the executor, bench-log 4 October; the +chained aggregation is 9.6 to 9.7 s per block on a mining 5090, `chain-pc2-pv1c`) and more records; the assignment +rule (8 assignees, 10 s window, spec 7.2) would need a CPU class with a longer window or the CPU provers only ever +win the open phase. None of that is measured. Status: Designed, nothing implemented. + +### 4b. A second prover backend behind the swappable interface + +The seam exists: `proving/igneum-prove/host/src/proof_system.rs` (`ProofSystem` trait, `Sp1ProofSystem`, +`StubProofSystem`), versioned per the design. The candidates: + +| Target | Most likely backend | What exists | What adopting it costs | +|---|---|---|---| +| Apple silicon | RISC Zero's Metal prover (`metal` feature, shipped) | a shipped feature with kernels in the tree; ICICLE's Metal backend as the other library | a second guest program (the shard statement, `core/` is plain Rust and ports; the precompile patches for keccak and secp256k1 differ), a second pinned program id and verifying key in `elf/manifest.json`, the node's verifier for both proof formats (RISC Zero receipt and SP1 compressed proof) in `--mode verify` and the proof pool, and an aggregation problem: SP1's aggregator folds SP1 proofs by deferred verification; it cannot fold a RISC Zero receipt, so a block with shards from both families needs two aggregations or a wrapper. Approximate: weeks of a person's time, no measurement of a Metal shard time exists; RISC Zero's Groth16 wrapper for light clients does not run on Apple silicon at all | +| AMD | nothing | sppark's limited HIP path (not in SP1's tree); ICICLE's and Jolt's Vulkan and Metal work are not AMD | no backend to adopt. The honest statement is that it lands when a zkVM ships one | + +### 4c. The public line + +The site says today (read 5 October 2026 from `site/litepaper.html`, `site/miner.html`, `site/index.html`): "The same +card proves every block", "The card mines and proves", "Ember finds your GPU, makes a wallet for you and runs the +node, the miner and the prover as one app", "Target: shard size will be set so a 12 GB card proves one shard in about +20 seconds". Every one of those is true of an NVIDIA card with enough memory and false of an AMD or Apple card, and the +litepaper's own rule is "If consumer GPUs cannot prove shards fast enough, Igneum says so and does not launch on promises". + +The recommended line, for the litepaper's proving section, the miner page and the app's Proving tile (copy law): + +> Proving needs an NVIDIA card with 16 GB or more today (20 GB to mine and prove on the same card). AMD and Apple +> cards mine. A prover for them lands when a zkVM ships one. A CPU can prove a small shard in about five minutes with 32 GB of RAM free; the chain pays the first proof, which a card delivers in seconds, so CPU proving is for testing, not income. + +Where the numbers come from: 16 GB and 20 GB are the measured gates of `proving-v1.md` (13.8 GB prover-alone peak, +16.8 GB mine-and-prove peak); "when a zkVM ships one" is section 1. The line changes when the memory sweep moves the +gates or a backend ships; it is reviewed with every prover release. + +## 5. What this analysis does about it (the consequences, before anyone asks) + +| Consequence | Action | Owner | +|---|---|---| +| An AMD-only miner loses the proving share | the CPU tier of 4a is measured here (section 2); whether it becomes a tier is a decision for Josh on those numbers | this analysis; Josh | +| The rig's prover unit must select NVIDIA cards only | already true in `prover_decision`; told the rig-installer agent to keep it as a stated rule and to print the CPU-fallback line for AMD-only rigs | rig-installer agent | +| The app's Proving tile on an AMD-only or Apple machine should say why it is off and name the CPU path | the `provedefault.rs` lines already say so for Apple; AMD-only Windows machines get "no NVIDIA card ..." | proving agent (told) | +| The site and litepaper over-promise for AMD and Apple | the line of 4c, to land with the next site pass (copy law; `node site/build.mjs`; link-check) | site-pages owner; not changed here | +| A Metal prover is the only non-NVIDIA path with a shipped backend | 4b names RISC Zero's Metal path and its cost; no work started | proving agent (told) | + +## 6. Commands, jobs and sources + +| What | Where | +|---|---| +| The PC 1 jobs (signed `run` jobs, PowerShell, not elevated, miners untouched, SP1_PROVER=cpu, host built without the `cuda` feature) | `tools/amd-prove/pc1-cpu-prove.ps1` (small fixtures), `pc1-cpu-prove-sp.ps1` (the `S_p` shard); published as `cpu-prove-pc1-small` (built, proved nothing: the quote bug) and `cpu-prove-pc1-small2` (the numbers) by `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --shell powershell --timeout-minutes 60`; `cpu-prove-pc1-sp` written, not published; `check-job-bash.sh` gates the bash body of every job script here; the package `igneum-prove-wsl2-pv1b.zip` (sha256 df50dee5...), the URL and hash filled at publish time, never committed | +| The Mac run | `tools/lock/with-lock.sh measure /usr/bin/time -l igneum-prove-host block-56-transfers-3shards.json --mode shard --shard 0` with the `proving-v1` worktree's host (pinned ids checked with `--mode id`) | +| Results | `node tools/jobs.mjs cpu-prove-pc1-small2 --all`, `docs/bench-log.md` entry "5 October 2026, the CPU prover on PC 1 and the backend survey" | +| Pages read | SP1: docs.succinct.xyz hardware-acceleration page, github.com/succinctlabs/sp1 (releases, `crates/sdk/src/lib.rs`, `sp1-gpu/README.md`, code search); RISC Zero: dev.risczero.com local-proving, `risc0/zkvm/Cargo.toml`; Jolt: README, book, PR #1733; OpenVM: releases; ICICLE: install_gpu_backend page, releases, README, PRs #735 and #1019; sppark README | diff --git a/docs/analysis/asic-resistance-history.md b/docs/analysis/asic-resistance-history.md new file mode 100644 index 000000000..129dc757e --- /dev/null +++ b/docs/analysis/asic-resistance-history.md @@ -0,0 +1,418 @@ +# ASIC resistance, 2011 to 2026: the history, the papers, the lessons, and the audit of Igneum against them + +5 October 2026 (night), branch `asic-history`. Asked by Josh at 20:05 UTC: "do a full on deep dive into the full history of 'asic resistance' and see if we can add or upgrade anything." Baseline for the audit: the Counter ASIC 2.0 final class decided tonight (`docs/plans/counter-asic-2-status.md` on `ca2-coord`, entries 20:16 to 22:25 UTC; `docs/analysis/chip-model-v3.md` on `ca2-mixer` 1ab8b21). Every figure about another chain cites a repo file, a paper or a dated article, or is labelled approximate. Hash-per-joule gains are computed from the cited hashrate and watt figures of the chip and of the best consumer GPU of the same year, and are approximate by construction (GPU figures vary by tuning). Research gathered by four sub-agents between 20:10 and 20:45 UTC; the fetch failures they reported are listed in section 6. + +## 0. One page for Josh + +**What the history says Igneum is doing right.** + +| # | What | The evidence | +|---|---|---| +| 1 | Binding the hash to random reads over a dataset larger than any on-chip cache, with the dataset growing on a schedule, and measuring the latency-bound share per card | Every compute-bound hash fell to a chip at 20x to 1,200x per joule within 16 to 37 months (rows Scrypt, X11, Blake, kHeavyHash, Blake3). The memory-bound hashes capped the chip at 1.1x to 4.8x (Ethash rows) or saw no chip at all (KawPow, Verthash, Autolykos, FishHash). RandomX's own design chose a 2 GiB dataset because a 2 GiB SRAM die was "questionable" at 7 nm in 2019 (RandomX `doc/design.md`) | +| 2 | A random program per epoch from a VDF seed, a weak-program filter, era draws from chain state, instruction families unlocked by height, and no scheduled human fork | Monero forked four times in 20 months and lost 85% of its hashrate at each fork to chips that returned within months (CryptoNight rows). Vertcoin forked three times and was 51%-attacked after two of them. Ravencoin's X16Rv2 fork was followed by FPGA bitstreams within weeks. Grin's six-monthly tweaks worked only because the lane was scheduled to die. The only random-program hashes with no chip after five years are the ones that never needed a fork (KawPow, FiroPoW, ProgPowZ) | +| 3 | Pricing the on-die-cache recompute chip and spending the mixer budget against it (x8: 0.92x with the 3x factor), with the model public | This is the "light-evaluation attack" Least Authority flagged on ProgPoW in 2019 and ProgPoW never fixed; Bob Rao's hardware audit put an on-die-DAG ProgPoW chip at "<< 0.1x" the energy per hash of a GPU. Kik's 2020 exploit was the same attack through a 64-bit seed. Igneum has a number against it; ProgPoW had a suggestion | + +**What the history says Igneum is missing or under-weighting.** + +| # | What | The evidence | +|---|---|---| +| 1 | The partial-store chip with a custom memory system (HBM or many narrow DRAM channels) is not in the chip model. The model prices only the f = 0 endpoint (all SRAM, recompute everything) | The only chip class that ever beat a memory-bound GPU hash did it this way: Ethash chips reached 2.1x (Linzhi, 2020), 2.9x (E9, 2022) and 4.8x (Jasminer X4, 2021) per joule through custom memory controllers and on-package memory, with no on-die dataset at all. The time-memory curve between f = 0 and f = 1 is open item O-1.6 and has never been drawn (`proto-metal/MEMHARD.md` section 3 item 2). Cuckoo Cycle's "tmto-hard" claim fell to a 50x memory cut for 2x time within two months of publication (Andersen, 2014) | +| 2 | The item-derivation mixer has a fixed shape. That fixed shape is exactly what hands the recompute chip its 3x fixed-function factor (bare 0.31x becomes 0.92x) | RandomX made the item derivation itself a random program (SuperscalarHash: about 450 instructions generated per seed, scheduled for a superscalar core, 170-cycle latency to match DRAM), so a chip cannot hard-wire it. Igneum draws the mixer's constants per day and keeps the shape; a chip hard-wires the shape. CryptoNight-R's random math per block raised chip latency only 2.5x; the lever is in the memory path, which for the recompute chip is the mixer | +| 3 | No clock and no detector. Chips have appeared at market caps from $18M (Radiant) and $23M (Grin) upward, and Vorick's 2018 rule was "any coin with over $20M of block reward in a year has a secret ASIC on it". Monero's secret chips held 85% of the hashrate before anyone saw them. Igneum's bounty is unfunded (D11) and the public benchmark is January 2027 | Rows Monero, Radiant, Grin, Handshake, Kadena; section 2.5 (the market-cap table); MoneroCrusher's nonce analysis (February 2019) found the chips by their share pattern, which is the detector Igneum can run from day one on the observer | + +**The one change I would make first.** Add the partial-store HBM chip to the chip model and draw the time-memory curve before genesis (ranked addition 1, section 4.3). It is an analysis, it costs the GPU nothing, and it is the one chip class that history shows beating memory-bound GPU work. If that row comes out under 2x the x8 decision stands as measured; if it does not, the next lever is known before the vectors are frozen. + +Everything else in this document is the evidence behind that page. + +## 1. The history, one row per attempt + +Columns: what the hash relied on; when it went live; the first chip that beat it (vendor, model, date, rated hashrate and watts); the gain in hash per joule against the best consumer GPU of the time (approximate, derived); how long it held (months from live to first chip); what the chain did. "None" in the chip column means no shipped chip was found in any source as of October 2026. + +### 1.1 The rows + +| # | Hash (chain) | Live | Relied on | First chip (vendor, model, date, rate, watts) | Gain per joule vs GPU (approximate) | Held (months) | Response | Sources | +|---|---|---|---|---|---|---|---|---| +| 1 | Scrypt (Tenebrix, Litecoin, Dogecoin) | Sep and Oct 2011 | 128 KB scratchpad, meant to fit a CPU cache and not a 2011 GPU; latency at SRAM scale | Gridseed GC3355, early 2014, 360 kH/s at 7 to 8 W; Innosilicon A2 Terminator, Apr 2014, 28 nm; KnC Titan, 2014, 300 MH/s at 850 W | Gridseed 19x, A2 68x, Titan 140x (vs Radeon 7970 at 700 kH/s, 285 W); Antminer L7 (2021) 1,100x | 27 | Embraced. Dogecoin merge-mined with Litecoin from Sep 2014 | [S1] [S2] [S3] [S4] | +| 2 | X11 (Dash) and the chain family X13, X15, X17, Quark, Nist5, C11 | Jan 2014 | A chain of 11 SHA-3 candidates; compute only. Duffield said it was meant to replay Bitcoin's CPU to GPU to ASIC path, not to prevent it | iBeLink DM384M, Mar 2016, 384 MH/s at 715 W; Baikal Giant A900 (2016); Antminer D3, Sep 2017, 19.3 GH/s at 1,200 W | DM384M 33x, D3 1,000x (vs R9 280X at 4 MH/s, 250 W) | 26 | Embraced. Baikal's multi-algo units covered the whole family by 2017 | [S5] [S6] [S7] | +| 3 | Ethash (Ethereum) | Jul 2015 | DAG of 1 GB growing per epoch, 128-byte random reads; memory bandwidth | Antminer E3, announced Apr 2018, 180 MH/s at 800 W, 4 GB DDR3; Innosilicon A10 Pro (2020) 500 MH/s at 950 W; Linzhi Phoenix (Dec 2020) 2,733 MH/s at about 3,000 W; Jasminer X4 (Oct 2021) 2.5 GH/s at 1,200 W; Antminer E9 (Jun 2022) 2.4 GH/s at 1,920 W | E3 1.1x to 1.6x (a tuned 1080 Ti beat it per watt); A10 Pro 1.2x vs RTX 3080; Phoenix 2.1x; X4 4.8x; E9 2.9x | 32 to the first chip; about 65 to a chip over 2x | ProgPoW (EIP-1057) debated 2018 to 2020 and shelved; PoS at the Merge, 15 Sep 2022. ASIC share of hashrate stayed small (one 2018 estimate: 3%) | [S8] [S9] [S10] [S11] [S12] [S13] | +| 4 | Etchash (Ethereum Classic) | Nov 2020 (Thanos, ECIP-1099) | Ethash with the DAG cut to 2.5 GB to keep 3 to 4 GB cards mining; not an anti-chip change | Post-Merge Ethash chips moved over: Antminer E9 Pro (Feb 2023) 3.68 GH/s at 2,200 W; Jasminer X16-P (2023) 5.8 GH/s at 1,900 W | E9 Pro about 4x, X16-P about 7x vs RTX 3090 (approximate) | n/a | Embraced | [S14] [S15] | +| 5 | Equihash 200,9 (Zcash, Horizen, Pirate) | Oct 2016 | Generalised birthday problem (Wagner); 144 MB in practice; memory size, with a claimed 1,000x compute penalty for halving memory | Antminer Z9 mini, announced 3 May 2018, 10 kSol/s at 300 W; Innosilicon A9 ZMaster (Jun 2018) 50 kSol/s at 620 W; Z11 (2019) 135 kSol/s at 1,418 W; Z15 (2020) 420 kSol/s at 1,510 W | Z9 mini 12x, A9 29x, Z11 34x, Z15 100x (vs GTX 1080 Ti at 700 Sol/s, 250 W) | 18 | Zcash: no fork (Zcon0 vote 45 to 19 against prioritising resistance, Jun 2018; ECC chose Sapling over resistance); PoS plan announced Nov 2021. Horizen: stayed after a 51% attack (Jun 2018). Pirate: stayed | [S16] [S17] [S18] [S19] [S20] | +| 6 | Equihash parameter forks: Zhash 144,5 (Bitcoin Gold), ZelHash 125,4 (Flux), BeamHash I to III 150,5 (Beam), 210,9 (Aion), 192,7 (Zero) | Jul 2018 (BTG), Jun 2019 (Flux), Jan 2019 (Beam) | Larger memory per solver than 200,9 (Beam and Flux also changed the datapath against FPGA bitstreams) | None found for 144,5, 125,4 or 150,5. Vorick (May 2018) wrote that an Equihash chip able to follow any parameter fork had been designed | n/a | BTG 7 years, Flux 7 years, Beam 7 years, with small prizes (section 2.5) | BTG: forked after the May 2018 51% attack, attacked again Jan 2020. Beam: planned "one or two hard forks" then BeamHash III (Jun 2020) as the last. Flux: stayed on ZelHash | [S21] [S22] [S23] [S24] | +| 7 | Scrypt-N, Lyra2RE, Lyra2REv2 (Vertcoin) | Jan 2014, Dec 2014, Aug 2015 | Memory-hard sponge (Lyra2) inside a hash chain; Scrypt-N grew N over time | Dayun Zig Z1, Sep 2018, 6.8 GH/s at 1,200 W (FPGA bitstreams of about 216 MH/s per board preceded it in 2018) | Z1 20x (vs GTX 1080 Ti at 59 MH/s, 207 W) | 37 (Lyra2REv2) | Lyra2REv3, Feb 2019, "to rid the network of the current generation of ASICs and FPGAs"; 51% attacks via rented hash in Oct to Dec 2018 (22 reorgs) and Dec 2019 | [S25] [S26] [S27] [S28] | +| 8 | Verthash (Vertcoin) | Jan 2021 | A 1.2 GB file generated from the chain's own block headers; random reads; bandwidth, like Ethash | None found | n/a | 69 and counting, small prize | No fork since | [S29] [S30] | +| 9 | X16R (Ravencoin) | Jan 2018 | 16 hashes in an order set by the previous block hash; compute, with order randomised | OW Miner OW1, Sep 2019, about 182 MH/s at 1,400 W; SKC Turing R1 claimed. Widely thought FPGA-based | OW1 1.3x (vs GTX 1080 Ti at 18 to 31 MH/s, 190 to 284 W) | 20 | X16Rv2, 1 Oct 2019 (hashrate fell 70%); FPGA bitstreams for X16Rv2 within weeks (BittWare CVP-13 at 240 MH/s) | [S31] [S32] [S33] | +| 10 | KawPow (Ravencoin; Neoxa, Clore, Meowcoin, Neurai) | 6 May 2020 | ProgPoW 0.9.4 variant: random math per block, DAG, 16 KB cache reads; targets the GPU datapath | None found as of 2026 (no vendor lists one; one retailer listing naming an "Antminer X9" for KawPow is an error) | n/a | 77 and counting | Roadmap: "No additional future algorithm forks are envisaged" | [S34] [S35] [S36] | +| 11 | MTP (Zcoin, now Firo) | Dec 2018 | Argon2d memory array with a Merkle tree (Biryukov and Khovratovich, "Egalitarian computing"); 4 GB; memory size | None | n/a | 34, then replaced | Dinur and Nadler broke the 2 GB instance to under 1 MB at a 170x compute penalty before launch (2017); MTP 1.2 patched it. FiroPoW (ProgPoW variant) Oct 2021 for block size and GPU fairness, not for a chip | [S37] [S38] [S39] [S40] | +| 12 | FiroPoW (Firo) | 26 Oct 2021 | ProgPoW 0.9.4 with a per-block program | None found | n/a | 59 and counting | Nov 2025 fork cut the maximum DAG to 6.76 GB to keep 8 GB cards | [S40] [S41] | +| 13 | Blake-256 14r (Decred) | Feb 2016 | Compute; chosen to be ASIC-friendly ("easy and fast implementation of hardware is the main design goal") | Innosilicon D9, about Apr 2018, 2.4 TH/s at 1,000 W; Obelisk DCR1 (Jun 2018); Antminer DR5 (Dec 2018) 35 TH/s at 1,610 W | D9 130x, DR5 1,200x (vs GTX 1080 Ti at 4.6 GH/s, 250 W) | 26 | Embraced by design | [S42] [S43] [S44] | +| 14 | Blake2b (Sia) | Jun 2015 | Compute | Antminer A3, Jan 2018, 815 GH/s at 1,186 W; Obelisk SC1 (Jul 2018) 550 GH/s at 500 W; Innosilicon S11 (2018) 3.83 TH/s at 1,380 W | A3 58x, S11 230x (vs GTX 1080 Ti at 2.96 GH/s, 250 W) | 31 | Fork at block 179,000 (Oct 2018) to brick Bitmain and Innosilicon units and keep Obelisk's; Innosilicon then held about 37% of hashrate | [S45] [S46] [S47] | +| 15 | Blake2s (Kadena) | Nov 2019 | Compute; chosen to be "GPU mineable and not immediately ASIC mineable (but for which an ASIC can be made)" | Goldshell KD2 and KD5, Mar 2021, 18 TH/s at 2,250 W; Antminer KA3 (Sep 2022) 166 TH/s at 3,154 W | KD5 195x, KA3 1,280x (vs RTX 3080 at 8.9 GH/s, 217 W) | 16 | Embraced by design | [S48] [S49] [S50] | +| 16 | CryptoNight (Bytecoin, Monero) | Jul 2012, Apr 2014 | 2 MB scratchpad sized to a per-core L3, AES rounds, random reads; latency at SRAM scale | Secret chips from about late 2017 (85% of the hashrate vanished at the April 2018 fork); Antminer X3, announced Mar 2018, 220 kH/s at 550 W; Baikal Giant-N | X3 40x to 50x (vs Vega 64 at about 2 kH/s, 200 to 250 W, approximate) | 43 to the secret chips, 47 to the announced one | Forks: CryptoNight v7 (6 Apr 2018), v8 (18 Oct 2018), CryptoNight-R (9 Mar 2019, random math per block seeded by height, chip latency up 2.5x), RandomX (30 Nov 2019). MoneroCrusher's nonce analysis (Feb 2019) found chips at over 85% of the hashrate again, four months after v8 | [S51] [S52] [S53] [S54] [S55] | +| 17 | RandomX (Monero; Wownero, ArQmA, Zephyr, Tari) | 30 Nov 2019 | A VM running 8 chained random programs per hash on a superscalar CPU with floating point; 2 GiB dataset derived from a 256 MiB cache by a random superscalar program (SuperscalarHash); 2 MiB scratchpad in L1, L2, L3 tiers | Antminer X5, Sep 2023, 212 kH/s at 1,350 W (RISC-V cores); Antminer X9, Jul 2026 delivery, 1 MH/s at 2,472 W; Pinecone INIBOX R1X, Mar 2026, 1.2 MH/s at 2,055 W | X5 at parity with a Ryzen 9 7950X (157 against about 200 H/J, approximate); X9 and R1X 2x to 3x over the best CPU (approximate). GPUs are 25x worse per joule than CPUs on it | 46 to parity hardware, about 75 to a 2x to 3x chip | No fork as of Oct 2026. Four audits in 2019 (Trail of Bits, X41, Kudelski, QuarksLab) found nothing critical | [S56] [S57] [S58] [S59] [S60] | +| 18 | Cuckoo Cycle (Grin Cuckaroo lane, Aeternity, Cortex) | Jan 2019 (Grin) | Find a 42-cycle in a random graph; lean solver one bit per edge; memory latency, or bandwidth in the mean solver | None on the Cuckaroo lane (Cuckaroo29 tweaked every 6 months: Cuckarood Jul 2019, Cuckaroom Jan 2020, Cuckarooz Jul 2020) | n/a | 24, retired on schedule | The lane was built to die: 90% of reward at launch falling to 0% in Jan 2021 (HF4) | [S61] [S62] [S63] | +| 19 | Cuckatoo31+ (Grin's chip lane) | Jan 2019 | Same, with plain bits in place of ternary counters to simplify chips; "Proof of SRAM" per Tromp | Obelisk GRN1 announced Jan 2019 and cancelled Jul 2019; Innosilicon G32 announced 2019 and never shipped; iPollo G1, Dec 2020, 36 GPS Cuckatoo32 at 2,800 W | G1 about 4x (vs RTX 3090 at about 1 GPS, 300 W, approximate) | 23, by design | Surrender by schedule. Tromp's $10,000 linear TMTO bounty was claimed in Apr 2025 (N/k bits at about k + 1,000 hashes per edge) | [S61] [S64] [S65] [S66] | +| 20 | ProgPoW (Ethereum proposal; Bitcoin Interest, Sero, Zano as ProgPowZ, Quai) | EIP May 2018; Bitcoin Interest 2018; Quai Jan 2025 | Random math per period on a 32-register file, 16 KB cache reads, 256-byte DAG loads, keccak-f800; "saturate the GPU" | None | n/a | 8 years across its adopters | Ethereum: tentatively approved Jan 2019 and Feb 2020, petition 27 Feb 2020, left "approved" and unscheduled on 6 Mar 2020, dead. Audits: Least Authority (Sep 2019) and Bob Rao (Sep 2019). Kik's 64-bit-seed exploit (Mar 2020) patched in 0.9.4 | [S67] [S68] [S69] [S70] [S71] [S72] | +| 21 | Autolykos v1 and v2 (Ergo) | Jul 2019; v2 Feb 2021 | v1: memory-hard with a per-miner secret key, so puzzles could not be outsourced to pools. v2: the secret removed (contract pools bypassed it); a 2 GB table that grows 5% per 51,200 blocks from block 614,400 | None | n/a | 87 and counting, small prize | v2 by EIP-0009 at block 417,792 | [S73] [S74] | +| 22 | Octopus (Conflux) | Oct 2020 | Ethash-style DAG; the "dense matrix step" could not be verified in `conflux-rust` tonight (unverified) | None | n/a | 72 and counting | CIP-102 (Aug 2022) proposed switching to Ethash to attract post-Merge miners; dormant | [S75] [S76] | +| 23 | kHeavyHash (Kaspa; Bugna kept it) | Nov 2021 | cSHAKE256, a 64x64 4-bit matrix multiply from the pre-PoW hash, cSHAKE256; compute, designed for optical and specialised hardware | IceRiver KS0, Jul 2023, 100 GH/s at 65 W; KS1, KS2 (Sep 2023); Antminer KS3, Aug 2023, 8.3 TH/s at 3,188 W; KS5 Pro (Mar 2024) 21 TH/s at 3,150 W | KS0 250x, KS5 Pro 1,100x (vs RTX 3090 at 910 MH/s, 150 W) | 17 to 20 | Embraced (Sompolinsky, May 2023: "an overall positive"). Hashrate went from under 100 PH/s to over 700 PH/s in months; the GPU share was negligible by late 2023 (approximate). Forks that left: Karlsen (FishHashPlus, Sep 2024), Pyrin (PyrinHash v2, Sep 2024), Spectre (CPU AstroBWTv3), Nexellia, Waglayla, Cryptix, Hoosat | [S77] [S78] [S79] [S80] [S81] | +| 24 | NexaPow (Nexa) | 2023 | SHA-256 plus a secp256k1 Schnorr signature per attempt; framed as "useful ASICs" | DragonBall A21, Jan 2025, 3.4 GH/s at 1,800 W | 3x to 4x (vs RTX 3090 at 123 to 137 MH/s, 230 W, approximate) | 24 | None | [S82] [S83] | +| 25 | Blake3 (Alephium; Iron Fish until 2024) | Nov 2021 | Double Blake3; compute; chosen as ASIC-friendly | Goldshell AL-BOX, 2023, 360 GH/s at 180 W; IceRiver AL0; Antminer AL1 (2024) 15.6 TH/s at 3,510 W; AL3 | AL-BOX 156x, AL1 350x (vs RTX 3090 at 2.3 GH/s, 180 W) | 22 to 24 | Alephium embraced. Iron Fish forked to FishHash (Apr 2024, FIP-3: Ethash-derived, fixed 4.6 GB dataset, 512 iterations, 128-byte mix); Karlsen adopted FishHashPlus (Sep 2024) | [S84] [S85] [S86] [S87] | +| 26 | FishHash (Iron Fish, Karlsen) | Apr 2024 | Ethash-derived, 4.6 GB fixed dataset; bandwidth | None found | n/a | 30 and counting, small prize | None | [S87] | +| 27 | Eaglesong (Nervos) | Nov 2019 | Compute (a new sponge) | Toddminer C1 (Feb 2020); Antminer K5, Mar 2020, 1.13 TH/s at 1,580 W; Goldshell CK5 (Mar 2021) 12 TH/s at 2,400 W | K5 70x, CK5 500x (vs RTX 3090 at about 2.1 GH/s, approximate) | 4 | Embraced | [S88] [S89] | +| 28 | Blake2b + SHA3 (Handshake) | Feb 2020 | Compute | Goldshell HS1, Jun 2020; HS3 (Jul 2020) 2 TH/s at 2,000 W; HS5 | Over 100x (approximate) | 5 | Embraced | [S90] [S91] | +| 29 | SHA512/256d (Radiant) | 2022 | Compute | DragonBall A11; IceRiver RX0, Sep 2024, 260 GH/s at 100 W | About 550x (vs RTX 3090 at 1.3 to 1.5 GH/s, approximate) | About 24 | None | [S92] [S93] | +| 30 | ProgPowZ (Zano), DynexSolve (Dynex), Janushash (Warthog), XelisHash v1 and v2 (Xelis), VerusHash 2.2 (Verus) | 2019 to 2024 | ProgPoW variant; GPU "neuromorphic" useful work; a product of VerusHash and SHA256t to balance CPU and GPU; CPU and GPU balanced; AES-based CPU hash | None found for any of them | n/a | Small prizes throughout | Xelis forked to v2 (Jul 2024) for FPGA resistance | [S94] [S95] [S96] [S97] [S98] | +| 31 | Ethash on EthereumPoW (ETHW) after the Merge | Sep 2022 | As Ethash | The Ethash chips above | About 4x (E9 Pro, X16-P vs RTX 3090, approximate) | n/a | Embraced | [S15] | + +### 1.2 What the rows say when sorted + +| Class of hash | Rows | Months to first chip | First-chip gain per joule | Best gain reached | +|---|---|---|---|---| +| Compute only (chains of hashes, Blake family, SHA-3 family, matrix multiply) | 2, 13, 14, 15, 23, 25, 27, 28, 29 | 4 to 31 (median about 24) | 33x to 250x | 500x to 1,280x | +| Memory at SRAM scale (128 KB scrypt, 2 MB CryptoNight) | 1, 16 | 27, 43 | 19x, 40x to 50x | 1,100x (Scrypt, 2021) | +| Memory size without a bandwidth bound (Equihash, Lyra2REv2) | 5, 7 | 18, 37 | 12x, 20x | 100x | +| Memory bandwidth at DRAM scale (Ethash, Verthash, FishHash, Etchash) | 3, 4, 8, 26 | 32 (Ethash); none for the others | 1.1x to 1.6x | 2.9x to 4.8x | +| Random program on a commodity datapath (RandomX, ProgPoW family, X16R's order randomisation) | 9, 10, 12, 17, 20, 30 | X16R 20 (FPGA-class, 1.3x); RandomX 46 to parity; none for ProgPoW's adopters in 8 years | 1.3x (X16R), 1x (RandomX 2023) | 2x to 3x (RandomX 2026, approximate) | +| Graph search (Cuckoo) | 18, 19 | 23 on the chip lane; never on the tweaked lane | 4x | 4x | + +Two caveats on the random-program rows. The prizes were small: Ravencoin, Firo and Zano never reached the market caps at which the 2018 chips appeared (section 2.5), so "no chip" is partly an economic fact. And RandomX's chips arrived once Monero's reward justified them: parity hardware at 46 months, a 2x to 3x chip at about 75 months (approximate), on a hash whose whole purpose was to make the CPU the chip. + +## 2. The academic side + +### 2.1 Memory-hard functions + +| Paper | Result | What it means for Igneum | +|---|---|---| +| Abadi, Burrows, Manasse, Wobber, "Moderately hard, memory-bound functions", NDSS 2003 and ACM TOIT 2005 [P1]; Dwork, Goldberg, Naor, "On memory-bound functions for fighting spam", CRYPTO 2003 [P2] | The origin of the idea: CPU speed varies 100x across machines, memory latency does not, so a cost function bound by cache misses is fairer than one bound by cycles | Igneum's latency-bound rule is this argument from 2003 applied to GPUs and DRAM: the DRAM row cycle is the same physics for a chip and a card (section 2.6) | +| Percival, "Stronger key derivation via sequential memory-hard functions", BSDCan 2009 [P3] | Defines sequential memory-hardness; ROMix is sequential memory-hard in the random-oracle model; cost measured in area-time (dollar-seconds) | The area-time measure is the one the chip model uses (equal silicon); scrypt's 2011 deployment at 128 KB ignored the paper's own scale | +| Alwen and Serbinenko, "High parallel complexity graphs and memory-hard functions", STOC 2015 [P4] | Cumulative memory complexity (CMC) in the parallel random-oracle model; earlier sequential measures fail against parallel, amortising adversaries | A chip is a parallel, amortising adversary; any Igneum claim about the dataset must be made in a parallel model | +| Alwen and Blocki, "Efficiently computing data-independent memory-hard functions", CRYPTO 2016 [P5]; "Towards practical attacks on Argon2i and Balloon hashing", EuroS&P 2017 [P6] | Any data-independent MHF can be computed in less than n^2 cumulative memory; Argon2i at O(n^1.75 log n), Catena and Balloon at O(n^1.67); the attacks are practical at real parameters | Igneum's addresses are data-dependent (register state), which is the right side of this result; the price is cache-timing leakage, which does not matter for a PoW | +| Alwen, Chen, Pietrzak, Reyzin, Tessaro, "Scrypt is maximally memory-hard", EUROCRYPT 2017 [P7] | scrypt's CMC is Omega(n^2 w) in the parallel ROM, optimal, against parallel amortising adversaries | Data-dependent chains of reads are the construction with the proof; Igneum's item derivation (8 dependent cache reads) is a short chain of this kind, with no proof | +| Biryukov, Dinu, Khovratovich, "Argon2", EuroS&P 2016 [P8]; Boneh, Corrigan-Gibbs, Schechter, "Balloon hashing", ASIACRYPT 2016 [P9] | Argon2d: a one-pass adversary can cut memory at most 3x at equal area-time; Argon2i needs over 10 passes to resist the Alwen-Blocki attack. Balloon: provable in the sequential model only; the paper says parallel ASIC attacks are outside its model | A "memory-hard" label without a stated adversary model has been wrong three times in this list (Argon2i, Balloon, Catena) | +| Biryukov and Khovratovich, "Tradeoff cryptanalysis of memory-hard functions", ASIACRYPT 2015 [P10]; Forler, Lucks, Wenzel, "Catena", 2013 [P11]; Simplicio et al., "Lyra2", IEEE TC 2016 [P12] | The ranking trade-off attack on Lyra2, yescrypt and Argon2; Catena's proofs flawed, 25x area-time cut; designers changed their algorithms | Lyra2REv2 (row 7) carried this construction into a PoW and still fell to a chip at 20x; the cryptanalysis found the shortcut before the chip did | + +### 2.2 Bandwidth-hard functions + +| Paper | Result | What it means for Igneum | +|---|---|---| +| Ren and Devadas, "Bandwidth hard functions for ASIC resistance", TCC 2017 [P13] | Memory-hardness (CMC) bounds a chip's area advantage and says nothing about energy; energy spent on off-chip memory traffic is comparable for a chip and a CPU, so bandwidth-hardness is the lever; scrypt, Catena-BRG and Balloon are bandwidth-hard with suitable parameters; the stacked double butterfly is capacity-hard and not bandwidth-hard | The chip model's "equal silicon" row is an area argument. The energy argument is the one the Ethash chips answered: they moved the same bytes at lower energy per byte with custom memory controllers (rows 3 and 4). Igneum's hash moves 128 x 64 B = 8 KB of DRAM lines per hash on AMD and 128 x 32 B on NVIDIA; a chip with 4-byte access granularity moves 512 B for the same work. That is the bandwidth-per-watt gain the plan's last section warns about, stated in Ren-Devadas's units | +| Blocki, Ren, Zhou, "Bandwidth-hard functions: reductions and lower bounds", CCS 2018 [P14] | Bandwidth cost in the parallel ROM equals the red-blue pebbling cost of the graph; high CMC implies high bandwidth cost; Argon2i and DRSample are maximally bandwidth-hard; a tight lower bound on scrypt's energy | The right formal target for a future proof about the item derivation, if one is ever attempted; none exists today | +| Alwen, Blocki, Harsha, "Practical graphs for optimal side-channel resistant MHFs", CCS 2017 [P15] | DRSample: a practical graph with maximal depth-robustness | Not applicable: Igneum does not need side-channel resistance | + +### 2.3 Asymmetric, egalitarian and graph proofs of work + +| Paper | Result | What it means for Igneum | +|---|---|---| +| Biryukov and Khovratovich, "Equihash", NDSS 2016 [P16] | Wagner's generalised birthday with algorithm binding; claimed 1,000x compute for halving memory | The claim did not survive contact with a chip design: 144 MB in practice fitted the Z9's memory system (row 5) | +| Biryukov and Khovratovich, "Egalitarian computing", USENIX Security 2016 [P17]; Dinur and Nadler, "Time-memory tradeoff attacks on the MTP proof-of-work scheme", CRYPTO 2017 [P18] | MTP: Argon2d plus a Merkle tree. Dinur-Nadler: malicious proofs with under 1 MB in place of 2 GB at a 170x compute penalty, by injecting blocks that steer Argon2d's data-dependent addressing | The attacker who controls the memory's contents controls the addresses. In Igneum the day key comes from a VDF of chain state and the cache fill is a chained block function, so no miner chooses the contents. The analogy still holds for the unreviewed mixer: a structural weakness in M_r is the shortcut this paper found in MTP | +| Tromp, "Cuckoo Cycle", BITCOIN 2015 [P19]; Andersen, "A public review of Cuckoo Cycle", 31 Mar 2014, and "Exploiting time-memory tradeoffs in Cuckoo Cycle", 1 Aug 2014 [P20]; the linear TMTO bounty, claimed Apr 2025 [S66] | Edge trimming cut memory about 50x for about 2x time, two months after publication; Tromp adopted it. The 2025 bounty result: an N/k-bit chip must hash each edge about k + 1,000 times | A time-memory claim is a curve, and the curve was wrong by 50x until someone drew it. Igneum's curve between "store everything" and "recompute everything" has not been drawn (O-1.6) | +| Georghiades, Flolid, Vishwanath, "HashCore", 2019 [P21] | "Inverted benchmarking": random widgets modelled on SPEC CPU workloads so the CPU is already the chip | The same idea as RandomX and ProgPoW stated generally: the hash is a benchmark of the target hardware | + +### 2.4 Program-based proofs of work and their audits + +| Document | What it says | What it means for Igneum | +|---|---|---| +| RandomX `doc/design.md` and `doc/specs.md` (tevador) [S56] [S57] | A VM so that the work is "data and code"; 8 chained programs per hash so a miner cannot filter (filtering 25% of programs at a 50% speedup yields 0.44x honest speed); SuperscalarHash of about 450 instructions with 155 multiplies, scheduled for a superscalar core at a 170-cycle latency to match DRAM, so a light-mode chip with the 256 MiB cache on die pays 760 cycles and 1,240 multiplies per item, "energy comparable to loading 64 bytes from DRAM"; a 2,080 MiB dataset because a 2 GiB SRAM die was "questionable" at 7 nm in 2019; cache-to-dataset ratio capped at 8 to keep the area-time product constant; double-precision floating point to force the whole CPU; "DRAM cannot do more than about 25 million random accesses per second per bank group" | Igneum rebuilt the idea for a GPU. The parts that carried over: the dataset above on-chip cache, the dependent item derivation, the cache-to-dataset ratio (4 at genesis, 8 at year 4 under option C). The parts that did not: per-hash programs (a GPU cannot JIT per hash and stay a GPU), floating point (vendor rounding), a random item-derivation program (Igneum's mixer has a fixed shape with drawn constants). Section 4.3 ranks the last of these | +| Trail of Bits audit of RandomX, 2 Jul 2019 [S60]; Kudelski, X41, QuarksLab (2019) | Two low findings and 47 brittle parameters; the design affirmed | Four paid external reviews before launch, for a hash whose whole value was the resistance claim. Igneum has had none (ledger M7) | +| EIP-1057 ProgPoW [S67]; Least Authority audit, 9 Sep 2019 [S69]; Bob Rao hardware audit, Sep 2019 [S70] | Claimed chip gain 1.1x to 1.2x. Least Authority: no issues, five suggestions, one of them the light-evaluation attack (on-the-fly DAG generation with the 16 MB cache in on-die SRAM) "may become possible within a few years" once about 100 MB of fast on-die SRAM is feasible. Rao: energy per hash is the only meaningful metric; shipping Ethash chips show about 1.6x hashrate per watt; conventional compute chips gain little on ProgPoW; integrating the DAG on die cuts data-movement energy by over 10x, so "ProgPOW ASICs with << 0.1X E/H over GPUs can be built"; an advanced-node chip is "$20M+" and "1+ year"; a 16-die split holding a 2.78 GB DAG was about $172 per board in 2019 against about $240 for a GPU board, and a monolithic die "viable around 2025" | The light-evaluation attack is Igneum's M16 recompute chip. ProgPoW left it as a suggestion; Igneum priced it and spent the mixer against it (x8). Rao's "$172 per board for a 16-die split" is the HBM-class partial-store chip in a different form, and it is the row the Igneum model lacks | +| Kik, "ProgPoW exploit", 4 Mar 2020 [S71] | A 64-bit seed lets a chip skip memory access with a cooperating node; patched in 0.9.4 | Igneum's seed is 256 bits and the program is the epoch's; the nearest analogue is header grinding for cache locality, unmeasured (section 4.3, check 4) | + +### 2.5 The economics of a chip + +**What a chip costs to make** (design plus masks, by node; all figures from the cited articles, which disagree with each other by 2x and say so): + +| Node | Mask set | Full design, IBS as quoted by Semiengineering (2018 and 2021) | Full design, other estimates | Sources | +|---|---|---|---|---| +| 65 nm MPW shuttle | n/a | n/a | Europractice 2025: about €51,000 minimum (9 mm^2 at €5,720 per mm^2) | [E1] | +| 28 nm | "beyond $1M" (SemiAnalysis 2022); $1M to $3M (Silicon Analysts 2026) | $51.3M (2018); $40M (2021) | $5M to $30M total NRE for a small chip (Silicon Analysts) | [E2] [E3] [E4] | +| 16/12 nm | n/a | $106M (2018 revision of a 2014 $310M figure) | Europractice MPW 16 nm: about €125,000 minimum | [E1] [E4] | +| 7 nm | "beyond $10M" (SemiAnalysis); $5M to $10M (Silicon Analysts) | $297.8M (2018); $217M "mainstream" (2021); Semiengineering's own 2023 discount: about $160M | Startups shipped 7 nm chips for "$50M to $75M" all-in (SemiAnalysis); a 10 nm-class mining chip "$20M+" (Rao 2019) | [E2] [E3] [E4] [S70] | +| 5 nm | $10M to $20M | $542.2M (2018); $416M (2021); about $280M discounted (2023) | Marvell 2023: $449M (secondary source, approximate) | [E2] [E3] [E5] | +| 3 nm | "$40M range" | $500M to $1.5B (2018); $590M (2021) | Marvell 2023: $581M (secondary, approximate) | [E2] [E3] [E5] | + +Miner-makers' own numbers: Bitmain's 2018 filing shows R&D of $73M in 2017 and $86M in the first half of 2018 and three failed chips at a reported combined cost of about $500M [E6]; Canaan's 2019 prospectus shows R&D of $26.5M in 2018 and "seven tape-outs" at a 100% success rate [E7]; Vorick wrote that Bitmain brought the Sia A3 to market for "less than $10 million" and took over $20M of orders within eight minutes [S47]; Obelisk's DCR1 was a 28 nm part [S43]; Taylor's 2013 survey gives $150,000 for a 130 nm and $500,000 for a 65 nm Bitcoin chip NRE in 2012 [E8]. + +**Where the 256 MiB cache lands a chip.** The Counter ASIC 2.0 analysis priced the 256 MiB SRAM mirror at 128 mm^2 and $46 per good die at N5 on the shipped-product density (`sram-mirror.md` revision 2, from AMD V-Cache 64 MB on 41 mm^2 at N7 [E9], TSMC N5 HD macro 31.8 Mib/mm^2 [E10]). The history adds the node question: a cheap chip is a 28 nm chip ($1M to $3M of masks, a $5M to $30M project), and a 28 nm bit cell is about 6x an N7 cell (approximate, from memory: TSMC 28 nm HD about 0.127 um^2 against N7's 0.027 [E10]), so 256 MiB at 28 nm is about 1,000 mm^2 of SRAM on the V-Cache density: more than a reticle. The cache forces the recompute chip onto a 7 nm or better node, which moves its project from the $5M class to the $50M class (SemiAnalysis's 7 nm startup figure). That is a stronger statement than the $46 per die, and it is the reason the cache size matters more than its per-die cost. Option C (the cache doubles with the dataset) keeps it true as nodes shrink: at the 6% per year density trend the status file cites, a 512 MiB mirror in year 4 costs more mm^2 than 256 MiB today. + +**When chips appeared** (CoinMarketCap historical snapshots pulled by the research agent; daily issuance is arithmetic from each chain's schedule; all approximate): + +| Chain | First public chip | Market cap then | Daily issuance then (USD) | +|---|---|---|---| +| Litecoin | Gridseed, Dec 2013 | $817M | $1.0M | +| Dash | PinIdea DR-100, Aug 2017 (iBeLink 2016 widely cited, date unverified) | $2.2B | $0.6M | +| Siacoin | Obelisk SC1 announced Jun 2017; Antminer A3 Jan 2018 | $430M; $1.5B | $0.43M; $1.1M | +| Decred | Obelisk DCR1 Jun 2017; Innosilicon D9 Apr 2018 | $216M; $353M | $0.18M | +| Monero | Antminer X3, Mar 2018 (secret chips from early 2017 per Vorick, unverified) | $3.3B | $0.75M | +| Ethereum | Antminer E3, Apr 2018 | $37.4B | $7.6M | +| Zcash | Z9 mini, May 2018 | $1.1B | $2.1M | +| Bitcoin Gold | the same chips, May 2018 | $1.3B | $0.14M | +| Grin | GRN1 announced Jan 2019 (cancelled); G32 Apr 2019 (never shipped); iPollo G1 Dec 2020 | $23M (Apr 2019); $23M (Dec 2020) | $0.24M; $33K | +| Nervos | Toddminer C1, Feb 2020 | $75M | $65K to $85K | +| Handshake | Goldshell HS1, Jun 2020 | $30M | $31K | +| Kadena | Goldshell KD5, Mar 2021 | $42M | $21K | +| Kaspa | IceRiver KS0, Jul 2023 | $480M | $0.42M | +| Alephium | Goldshell AL-BOX, May 2024 | $179M | $0.1M | +| Radiant | IceRiver RX0, Sep 2024 | $18M | $22K | + +Sources: [E11] (the research agent's CoinMarketCap pulls, dates in the table) and the chip rows above. Reading: a compute-bound hash gets a chip at $20K to $30K of daily issuance (Radiant, Kadena, Handshake); the 2018 cluster sat at $0.15M to $2M a day. Vorick's rule from May 2018: any coin with over $20M of block reward in a year (about $55K a day) should assume a secret chip [S47]. For Igneum the clock is the day its issuance in dollars crosses about $50K; a memory-bound hash buys time against that clock (Ethash: 32 months at the largest prize in the table), and the random program buys more (section 1.2), but nothing in the table says it buys forever. + +### 2.6 Latency as the resource + +| Source | What it says | What it means for Igneum | +|---|---|---| +| Li, Reddy, Jacob, "A performance and power comparison of modern high-speed DRAM architectures", MEMSYS 2018 [L1] | Row timings from datasheets: DDR4 tRCD 14, tRAS 33, tRP 14 ns; GDDR5 tRCD 14, tRAS 28, tRP 12; HBM and HBM2 tRCD 14, tRAS 34, tRP 14. Row cycle tRC about 40 ns (GDDR5) to 48 ns (DDR4, HBM2). "The memory-latency problem does still remain" | The row cycle is the floor under every dependent random read whatever the controller; HBM does not shorten it. What HBM and a custom controller change is the number of rows that can be opened per second per watt (channels, banks, pseudo-channels), which is the throughput of random reads in flight, which is what the 9070 XT probe measured as the card's ceiling (2.4 G reads/s against the 5090's 17.5 G) | +| Chang, CMU thesis, Dec 2017 [L2] | Over two decades DRAM capacity improved 128x, bandwidth 20x, latency 1.3x | The latency-bound rule has a long half-life; the bandwidth-per-watt lever (the Ethash chips) does not stand still | +| NVIDIA profiling guide and the Ampere tuning deck [L3] | L1 and L2 lines are 128 bytes in four 32-byte sectors; DRAM-to-L2 transactions default to 64 bytes since Volta, configurable 32, 64 or 128 on A100 | The 5090's 32-byte sector per 4-byte read is where a custom controller gains bandwidth efficiency (8x fewer bytes), the Ren-Devadas energy lever; the 9070 XT's 64-byte line is 16x. Neither changes the row cycle | +| RandomX `doc/design.md` [S56] | About 25 million random accesses per second per DRAM bank group; "all Dataset accesses read one CPU cache line (64 bytes) and are fully prefetched"; one program iteration tuned to "typical DRAM access latency (50-100 ns)" | The same arithmetic Igneum uses (128 dependent reads per hash against the card's random-read ceiling), from the design that held longest | +| Condrey, "PoSME", arXiv Apr 2026 (single author, not peer reviewed) [L4] | Latency-bound pointer chasing with hash compute under 3.5% of the step cost; GPUs 14x to 19x slower than a consumer CPU | The only dedicated latency-bound PoW paper found; its GPU-vs-CPU gap is the cost RandomX pays, and the cost Igneum avoids by keeping thousands of loads in flight per card | + +The paper that does not exist: nothing found treats cache timing as a feature; Catena and Argon2i treat it as a leak. No peer-reviewed survey of ASIC resistance as such surfaced; the nearest are Cho's 2018 multi-hash evaluation [P22] (X11-style resistance "is not strong enough"), Feng and Luo's 2020 three-processor study [P23] (GPUs dominate CryptoNight, Ethash and Cuckoo on CPU, GPU and Xeon Phi) and Yaish and Zohar's 2023 pricing of mining hardware as a bundle of options [P24]. + +## 3. The lessons + +Each lesson is stated once, with the rows it comes from. + +1. **Compute-bound work loses by 30x to 1,000x within two years, whatever its shape.** Chains of eleven hashes (row 2), sixteen hashes in a random order (row 9), a 64x64 matrix multiply (row 23), a new sponge (row 27), a signature per attempt (row 24): every one got a chip, the random-order chain at 1.3x by an FPGA within 20 months and the rest at 33x to 1,100x. Multiplying the number of fixed functions multiplies the chip's die, not its difficulty. For Igneum: nothing in the program's ALU work is a defence and the design already says so (ledger M1); the defence is the memory path. + +2. **Memory at SRAM scale is compute-bound with extra steps.** Scrypt's 128 KB (row 1) and CryptoNight's 2 MB (row 16) were sized to a 2011 and a 2014 CPU cache; a chip put the same memory on die and won 19x and 40x. Igneum's answer is the 256 MiB cache growing with the dataset (option C) and the 1 GiB to 2 GiB dataset; section 2.5 shows the cache size also sets the chip's node and therefore its project cost. The hot table (layer 5) was a step back toward SRAM scale, and the measurement agreed (the honest card paid 7% to 16%, the chip paid $0.23 per MB). + +3. **Bandwidth-bound work gets a memory chip at 2x to 5x.** Ethash held 32 months and then got chips whose whole design was the memory system: DDR3 (E3, no gain), GDDR6 (A10 Pro, 1.2x), custom controllers (Linzhi, 2.1x), on-package memory (Jasminer X4, 4.8x) (rows 3, 4). Rao's audit explains why in energy terms: the chip moves the same bytes at lower energy per byte, and a split-die design holding the DAG was already cheaper than a GPU board in 2019. Ren and Devadas give the bound: a chip's energy advantage on a bandwidth-hard function is the ratio of its memory energy per bit to the GPU's. Igneum's rule "avoid leaning on bandwidth" is right; its model has no row for this chip (section 4.3, addition 1). + +4. **Latency-bound and random-program work held longest, and the prize was usually small.** CryptoNight's latency bound at SRAM scale held 43 months, then fell to secret chips (row 16). RandomX's at DRAM scale held 46 months to parity hardware and about 75 to a 2x to 3x chip (row 17, approximate), on the largest prize any resistant hash has carried. ProgPoW's adopters have had no chip in 8 years on small prizes (rows 10, 12, 20). Verthash, Autolykos, Octopus and FishHash have none on small prizes (rows 8, 21, 22, 26). The honest reading: the random program on a DRAM-latency bound is the strongest construction the history has, and nobody has tested it at Ethereum's prize. + +5. **Periodic human forks fail as a defence.** Monero: four forks in 20 months; chips were back at 85% of the hashrate within four months of the v8 fork (row 16), and Vorick wrote that a chip able to survive forks at under a 5x hit had been designed. Vertcoin: three forks, two followed by rented-hash 51% attacks within weeks, because each fork reset the hashrate to a rentable size (row 7). Ravencoin: FPGA bitstreams for the new order within weeks (row 9). Sia: the fork bricked competitors' chips and left one vendor at 37% (row 14). Grin: the tweaks worked because the lane was scheduled to die (row 18). The chip's design cycle is 5 months for Bitmain (Vorick) and 13 for a startup; a fork every 6 months is a race the chip wins on the second lap, and each fork is a governance event. Igneum's draws are automatic and scheduled at genesis; that is the right side of this lesson, and section 4.3 asks whether the epoch can also be shorter than a bitstream (addition 5). + +6. **RandomX got the target right and the derivation right, and it costs GPUs 25x.** Right: the work is "code and data" so a fixed circuit cannot serve it; chained programs defeat filtering (0.44x); the dataset is above any SRAM die; the item derivation is a random superscalar program tuned to DRAM latency so the light-mode chip pays as much energy per item as a DRAM read (section 2.4). The cost: a GPU runs the VM at 25x worse per joule than a CPU (row 17), which is the cost Igneum refuses, and the reason the program is per hour and compiled. What Igneum did not take: the random item derivation (addition 2) and four external audits before launch (addition 3). + +7. **ProgPoW got the datapath right and lost on governance and one unpriced attack.** Right: target the commodity hardware's whole datapath (random math, register file, cache reads, DAG loads) so a chip has to be a GPU; the hardware audit agreed for compute-only chips (1.1x to 1.2x, Rao). Unpriced: the DAG on die (Least Authority suggestion 2, Rao's "<< 0.1x"), the same attack Igneum calls M16. Not adopted: two tentative approvals, a petition, bugs found late (Kik), authorship disputes and a PoS roadmap; the change needed a contentious fork on a live chain (row 20). Igneum's lesson is the one it already follows: every layer goes in before the public testnet as a genesis rule or a reserve, so no adoption vote is ever needed. + +8. **What a "GPU-friendly" chain lost when its GPU miner fell behind: the miners, then the chain's shape.** Kaspa's hashrate rose 7x in months and its GPU share went to nothing; seven forks left to re-resist (row 23). Alephium, Nervos, Handshake, Kadena and Radiant went the same way without the forks (rows 25, 27, 28, 15, 29). Iron Fish forked away from its own Blake3 within a year of the first box (row 25). The chains kept their security budget and lost the fleet that had launched them; the fleet's hardware went to the next GPU chain. For Igneum the metric is the share of hashrate on consumer cards by model, which is what the January 2027 benchmark should report and what the observer can estimate earlier (addition 4). + +9. **An unreviewed memory-hard construction has a shortcut until someone looks.** MTP fell from 2 GB to under 1 MB before launch (row 11); Catena's proofs were flawed; Argon2i's parameters were attackable at the IRTF's "paranoid" setting; Cuckoo's memory claim was off by 50x within two months (section 2.3). Igneum's M_r and chained cache have had no cryptanalysis (`MEMHARD.md` section 3, ledger M7); the acceptance rule is a statistical filter, not a proof. The x8 decision multiplies the mixer's weight in the chip model, which multiplies the cost of a structural weakness in it (addition 3). + +10. **The secret chip is found by its share, and it is on the chain before the announcement.** Monero's chips held 85% before anyone saw them; a nonce-pattern analysis found them (row 16). Zcash's Z9 was "5x to 10x below" what Obelisk's own study said the hash allowed, which is Vorick's evidence that better secret chips existed (section 1.1 row 5). A detector costs an observer query; a bounty costs escrow (addition 4). + +## 4. The audit of Igneum against the history + +### 4.1 The first-generation layers (live in class v2 and carried into v3) + +| Layer | Answers which failure | Does not answer | Evidence | +|---|---|---|---| +| Random program per epoch from a VDF seed, 12 integer families, nonce-dependent select | Fixed-function chips (lesson 1); program filtering and seed grinding (RandomX's 0.44x, spec 04's 130-to-1) | A "GPU without graphics": a programmable sequencer over 12 ops and 8 registers (ledger M1); an FPGA overlay or bitstream compiled within the hour (rows 7, 9: FPGAs were the first adversary of Lyra2REv2 and X16R); the 7.5x AMD gap is a one-vendor fleet | Rows 9, 10, 16, 17, 20 | +| Weak-program acceptance, exact 16 loads, fresh-source rule | Per-program hash-rate spread (1.10x residual) that a chip could pick | Nothing it claims to; a chip's advantage cannot come from the program (status 20:16) | Census [I1] | +| 1 GiB to 2 GiB dataset of 4-byte random reads, latency-bound, cache 256 MiB | SRAM-scale memory (lesson 2); the bandwidth lever at the honest card (lesson 3: 128 x 4 B keeps the 5090 at 9% of its stream bandwidth) | The partial-store chip with a custom memory system (lesson 3); the time-memory curve (O-1.6) | Rows 1, 3, 16; [P13] | +| 8 dependent cache reads per item, fixed-shape mixer with drawn constants | The on-die recompute chip (Least Authority's light-evaluation attack), priced at 2.45x bare under v2 | The fixed shape gives the chip its 3x factor (lesson 6); no cryptanalysis (lesson 9) | [S69] [S70]; M16 | +| Era draws from chain state (op weights, fold rotations), reserve families by height, dataset growth, no human release | Fork fatigue and fork-reset attacks (lesson 5); the chip that "survives forks at under 5x" (Vorick) is the chip that the draws are meant to outlast | The draws touch the program, not the item derivation, so they cost the recompute chip nothing (`chip-model-v3.md` section 2) | Rows 7, 16, 18 | +| Warp-unit CPU verification without the dataset (2.1 ms per warp under x8) | Keeps the verifier light, the Equihash and Cuckoo goal | Caps every lever: the mixer budget stops at the 10 ms gate | [P16] [P19] | + +### 4.2 The Counter ASIC 2.0 layers as decided tonight + +| Layer | Decision (status file) | Answers | Does not answer | History's verdict | +|---|---|---|---|---| +| 1 Load width 4, 16, 64 B | Keep 4 B (w16 closes nothing) | Keeps the 5090 latency-bound (9% of stream) | The AMD 7.5x gap (2.4 G reads/s at every width) | Right by lesson 3; the vendor gap is a 3.0 question and a soft form of lesson 8 | +| 2 Per-program width mix | Out (spread over 5% on every card) | n/a | n/a | Right: a per-program spread is what a chip picks (lesson 1's X16R: randomised order gave 1.3x, the shape still fixed) | +| 3 Per-warp scratch with RMW | Out (does not move the recompute chip; 2.4x at every share; costs GPUs 12% to 48%) | n/a | n/a | Right: SRAM-tier work favours the chip (lesson 2; Rao: SRAM is the chip's weapon) | +| 4 + 8 Era layout (stride, interleave) and per-site windows | In (era inside the class) | A hard-wired layout tuned to one era | A programmable address decoder (era-layout.md section 8 says so); costs the recompute chip nothing | Small by itself; its value is in lesson 5 (automatic change without a fork) | +| 5 Hot table sized to GPU cache | Measured, not adopted (honest card pays g = 0.84 to 0.93; chip pays SRAM) | n/a | n/a | Right by lesson 2 | +| 6 Cache growth | Option C: doubles with the dataset (256 MiB, 512 MiB year 4, 1 GiB year 12) | Keeps the mirror on a leading node (section 2.5) | n/a | Right; RandomX's cache-to-dataset ratio of 8 is reached at year 4 | +| 7 INT8 matrix family | Reserve R1 = mm8, W_new 4, unlock era 4 or 90% signal | A family that a 12-op chip lacks | Matrix hardware is the most abundant custom silicon on earth; Least Authority's suggestion 5 was "watch ML hardware"; Apple's emulation costs 1.6x to 4.7x per op | Keep in reserve, order it last (addition 6) | +| 9 Epoch length as an era parameter (10 min to 2 h) | Reserve only, design on `ca2-epoch` | The bitstream-per-epoch FPGA (rows 7, 9) | The FPGA overlay (a soft GPU) and the HBM FPGA | Rank it up (addition 5) | +| Mixer x8 (M16's lever) | In: 0.31x bare, 0.92x with the 3x factor, verifier 2.1 ms per warp, daily build 23 to 77 ms | The on-die recompute chip (lesson 6, Least Authority's attack) | Its own fixed shape (the 3x factor stays) and its lack of review (lesson 9) | The right lever; additions 2 and 3 are what the history says to do to it next | + +### 4.3 Ranked additions and upgrades + +Ranked by how much the history says each would change the outcome, with the cost to GPUs and the risk. "Genesis" means a rule fixed before the public testnet; "reserve" means a named family or parameter in the genesis reserve, unlockable by height or 90% signal; "nowhere" means do not add. + +| Rank | Addition | What it does | Evidence | Cost to GPUs | Risk | Where | +|---|---|---|---|---|---|---| +| 1 | **Price the partial-store chip and draw the time-memory curve.** A chip that stores a fraction f of the dataset in HBM or on many narrow DRAM channels, recomputes the rest from a 256 MiB on-die cache under x8, and reads with 4-byte granularity. Rows for f = 0.25, 0.5, 1 at HBM3 and at GDDR7 random-read rates, priced in energy per hash (Rao's metric) and in reads in flight per watt | The only chip class that beat a memory-bound GPU hash: Ethash's 2.1x to 4.8x came from the memory system with no on-die dataset (rows 3, 4); Rao priced a 16-die DAG holder under a GPU board in 2019; Cuckoo's curve was wrong by 50x until drawn [P20]; O-1.6 is open and `MEMHARD.md` section 3 item 2 says the curve was never drawn | None (analysis) | The row may come out over 2x, which would qualify the public claim before anyone else does | Genesis (before the vectors freeze) | +| 2 | **A random item-derivation program per day** in place of the fixed-shape mixer: a SuperscalarHash-style generator, integer only, drawn from the day key, with its own acceptance test, compiled once a day by miners and verifiers | RandomX's reason for SuperscalarHash: a fixed derivation is hard-wired by a chip; a random one makes the light-mode chip a CPU (section 2.4). In Igneum's model the fixed shape is the 3x factor that turns 0.31x into 0.92x; removing the factor is worth more than x8 to x16 would be (x16: 0.46x with the factor by M16's table) | None per hash (the daily build is 23 to 77 ms at x8 and would roughly double); the verifier needs a per-day compiled derivation (a JIT, or a round schedule drawn from a fixed set of reviewed rounds), measured against the 10 ms gate | Cryptanalysis of random ARX programs; weak draws; a JIT in the verifier is new attack surface; the vendors must agree bit-exactly on a program they compile | Reserve (named family, unlock by height or signal) now; genesis if the verifier cost is measured under the gate before the freeze | +| 3 | **External cryptanalysis of M_r, the chained cache and the acceptance rule before genesis**, with the x8 shape as the target | Lesson 9 (MTP, Catena, Argon2i, Cuckoo); RandomX bought four audits for $141,000 before launch [S60]; the x8 decision multiplies the mixer's weight in the chip model, so a shortcut inside the mixer is now worth 8x more to a chip | None | Finding something late moves the vectors; not finding it in time moves nothing | Genesis gate (ledger M7, raised in priority) | +| 4 | **The clock and the detector.** (a) A share-pattern detector on the observer: per-program hash-rate spread, nonce-group patterns and per-card-model rate bands, with an alert when a population behaves like one fixed design (MoneroCrusher's method); (b) a stated trigger: the bounty escrowed and the benchmark live before daily issuance crosses about $50K (Vorick's rule), not on a calendar date | Lesson 10 (85% secret share); section 2.5's table (chips at $20K to $30K a day on compute-bound hashes); D11 (the bounty is unfunded) | None | A detector with false positives; a trigger Josh has to fund | Not a layer; genesis-independent; do it before the public testnet | +| 5 | **Rank layer 9 (the epoch length) up, and measure the FPGA lane**: the compile-ahead cost per card at a 10-minute epoch (the `ca2-epoch` work), plus an estimate of a soft-overlay FPGA miner with HBM (reads in flight per watt against the 5090's 17.5 G/s) | FPGAs were the first adversary of Lyra2REv2 and X16R and came back within weeks of X16Rv2 (rows 7, 9); Xelis forked for FPGA resistance (row 30); a per-hour program is a bitstream target in a way a per-hash program is not | At 10-minute epochs: 6x the compile work per card (measured on `ca2-epoch`); the VDF lead shrinks | A short epoch moves the difficulty window (spec 1.12) and the seed path | Reserve (as decided), with the measurement before the public testnet | +| 6 | **Order the reserve by chip-unfriendliness**: families that force a full 32-bit datapath per lane first (byte permute, bit-field extract, variable shifts, popcount, select, the second shuffle form), mm8 last | Least Authority's "watch ML hardware"; int8 matrix blocks are licensable IP at every node; Apple pays 1.6x to 4.7x per emulated dot4 (status 20:38) | None at launch | None | Reserve ordering, genesis | +| 7 | **A vendor-share metric and a 3.0 target for the AMD gap**: the share of hashrate by vendor published with the benchmark, and the line-width question kept open as the plan says | Lesson 8: a one-vendor fleet is a softer version of chip capture; Equihash's NVIDIA tilt and Ethash's balance were part of each chain's miner politics (rows 3, 5) | n/a | A width that closes the gap makes the 5090 bandwidth-bound (status 20:27) | Counter ASIC 3.0 | + +Checks the history suggests that are not layers: + +| Check | Why | Source | +|---|---|---| +| 1. Header grinding for cache locality: can a miner search the pre-PoW header hash H for 32-lane groups whose 128 loads cluster into fewer DRAM rows or cache lines, at a search cost below the gain? | Kik's ProgPoW exploit and Dinur-Nadler's MTP attack were both "the attacker steers the addresses" | [S71] [P18] | +| 2. The chip detector's baseline: the per-program spread per card model, from the first week of the public testnet | Needed before addition 4(a) can alert | [S54] | +| 3. The 28 nm SRAM density figure in section 2.5 (approximate, from memory) and the node-cost consequence, cited properly | It is the argument that the cache size sets the chip's project cost | [E10] | + +Evaluated and placed nowhere, with the reason: + +| Candidate | Verdict | Reason | +|---|---|---| +| Program entropy per hash (RandomX) instead of per hour | Nowhere | Per-hash programs need an interpreter or JIT on the GPU, which is the 25x GPU penalty RandomX pays (row 17) and the reason Igneum compiles per epoch. The filtering attack per-hash chaining prevents is already closed by the VDF seed and the acceptance rule. Per-hour's residual exposure is the FPGA lane, which addition 5 addresses with a shorter epoch, not with per-hash programs | +| Superscalar dependency-chain design for the program itself | Nowhere, beyond what exists | The program's ALU work is not the defence (lesson 1); the dependency chain that matters is the 8 dependent cache reads per item and the 128 dependent loads per hash, both in place. The superscalar idea belongs in the item derivation (addition 2) | +| Verthash's table from the blockchain; a dataset derived from chain history | Nowhere | Against the recompute chip and the partial-store chip it changes nothing: both build the table from the same public inputs the GPU does. The day key already comes from a VDF of chain state, which gives the unpredictability without a history dependency; a history dependency costs the verifier the history (Verthash needs the headers) and ties the hash to pruning (spec 10) | +| Grin's dual PoW with a shifting split | Nowhere | It is a scheduled surrender (row 19). Igneum's automatic schedules (dataset growth, reserve unlocks, cache doubling) are the shifting split applied to one hash; a second lane would hand a chip a lane | +| Autolykos v1's non-outsourceability | Nowhere | It stops pools, not chips, and Ergo removed it after 19 months because contract pools bypassed it (row 21); Igneum needs pools (spec 09) | +| A per-hash VRF against nonce grinding | Nowhere | A signature per attempt is what NexaPow did and it got a 3x to 4x chip (row 24): EC arithmetic is fixed-function work. The grinding Igneum must guard is the header-locality search (check 1), which a VRF does not touch | +| Ternary or variable-precision integer ops | Nowhere, beyond the reserve | Every family must be bit-exact on three vendors; dot4 is native on NVIDIA and AMD and emulated on Apple at 1.6x to 4.7x (status 20:38), so each precision added is paid by the weakest vendor. The reserve already holds the integer-exact candidates; adding more does not change lesson 1 | +| Cache-timing-bound reads (ProgPoW's 16 KB cache, RandomX's L1 tier) | Nowhere | Measured out tonight at the L2 tier (layer 5) and the per-warp tier (layer 3): the honest card pays and the chip buys SRAM at $0.23 per MB. Rao's audit says the same about ProgPoW's cache reads | +| Divergent data-dependent branches | Nowhere (already excluded) | Branches cost a GPU divergence and a chip nothing; RandomX's single predictable branch targets speculative CPUs, which Igneum does not have | +| Floating point | Nowhere (already excluded) | Vendor rounding splits the chain (spec 1.14); RandomX could afford it because its target is one ISA family with IEEE semantics | + +## 5. Decisions this raises for Josh + +| # | Decision | Recommendation | +|---|---|---| +| 1 | Add the partial-store chip rows to `chip-model-v3.md` and draw the time-memory curve before the public testnet | Yes, before the vectors freeze (addition 1) | +| 2 | Name a random item-derivation program as a reserve family, and fund the verifier measurement that would move it to genesis | Reserve now; genesis if the verifier lands under the gate (addition 2) | +| 3 | Commission the external cryptanalysis of M_r and the chained cache before genesis, with the x8 shape as the target | Yes (addition 3; ledger M7) | +| 4 | Escrow the bounty and set its trigger to daily issuance, not to a date; build the share-pattern detector on the observer | Yes to the detector now; the escrow is Josh's (D11) | +| 5 | Rank the epoch-length reserve above the mm8 reserve, and measure the FPGA lane | Yes (additions 5 and 6) | + +## 6. Sources and limits of this research + +Research was gathered by four sub-agents between 20:10 and 20:45 UTC on 5 October 2026 and checked against the citations below. Fetch failures they reported: eprint.iacr.org PDFs sit behind a challenge page (abstract pages worked), so the Ren-Devadas energy figures, the Alwen-Blocki EuroS&P tables and the Lyra2 exponent come from abstracts; medium.com and bitcointalk.org returned 403 (Vorick's post was read through archive.sia.tech and secondary coverage; the IfDefElse posts through the Veil interview); Bitmain's prospectus PDF was blocked; the Dash iBeLink date and Octopus's "matrix step" are unverified; the 28 nm SRAM bit cell is from memory. Every hash-per-joule gain is derived from the cited rate and watt figures and is approximate. + +Igneum sources: [I1] `docs/analysis/weak-program-census-2026-10-03.md`; `docs/plans/counter-asic-2.md` (be4b295); `docs/plans/counter-asic-2-status.md` and `docs/plans/counter-asic-2-rollout.md` (`ca2-coord`); `docs/analysis/chip-model-v3.md` (`ca2-mixer` 1ab8b21); `docs/analysis/m16-recompute-attacker-2026-10-05.md`; `docs/analysis/sram-mirror.md` revision 2 (`ca2-analysis`); `docs/plans/era-layout.md` (`ca2-era`); `docs/plans/hot-table.md` (`ca2-cache`); `docs/plans/read-width.md` (`readwidth`); `docs/spec/01-lottery-hash.md`, `04-seeds-and-vdf.md`; `docs/bench-log.md` ("the 9070 XT on the eGPU", `opencl-rdna4`); `proto-metal/MEMHARD.md`; `docs/fud-ledger.md` M1, M3, M7, M16, C2, D11. + +History rows: +- [S1] https://medium.com/@Linzhi/what-is-memory-hard-45a363b59dfe (Tenebrix's 2011 claim); https://en.wikipedia.org/wiki/Litecoin +- [S2] https://jamesachambers.com/early-bitcoin-asic-miner-pictures-history/ (Gridseed); https://www.mikewesson.com/2013/04/29/mining-litecoin-on-ati-radeon-7970s/ (7970 at 700 kH/s) +- [S3] https://www.design-reuse.com/news/34403/innosilicon-28nm-litecoin-asic-reference-miner.html (A2, Apr 2014); https://www.coindesk.com/markets/2014/05/14/kncminer-reveals-additional-titan-scrypt-asic-specs (Titan) +- [S4] https://www.asicminervalue.com/miners/bitmain/antminer-l3-504mh ; https://cryptoage.com/en/2550-bitmain-antminer-l7-is-a-new-asic-miner-for-litecoin-and-dogecoin.html ; https://www.coindesk.com/markets/2014/09/11/dogecoin-community-celebrates-as-merge-mining-with-litecoin-begins +- [S5] https://docs.dash.org/en/stable/docs/user/introduction/features.html ; https://www.dash.org/news/happy-birthday-darkcoin/ +- [S6] https://cryptomining-blog.com/7117-the-first-x11-mining-asic-ibelink-dm384m-asic-dash-miner/ ; https://cryptomining-blog.com/7493-power-usage-and-noise-of-the-ibelink-dm384m-x11-asic-miner/ ; https://bitcointalk.org/index.php?topic=854257.320 (R9 280X at 4 MH/s) +- [S7] https://99bitcoins.com/guides-and-tutorials/dash-mining/antminer-d3-review/ ; https://www.cryptocompare.com/mining/asic-miner-market/baikal-giant-x10-x11-10ghs/ +- [S8] https://ethereum.org/developers/docs/consensus-mechanisms/pow/mining/mining-algorithms/dagger-hashimoto/ +- [S9] https://cryptoslate.com/bitmain-e3-asic-ethereum-miner/ (4 Apr 2018: E3 4.44 W/MH against a tuned 1080 Ti and RX 570); https://hothardware.com/news/bitmain-launches-ethereum-asic-miner-hashrate-comparable-8-gtx-1080-gpus +- [S10] https://innosilicon.global/product/innosilicon-a10-pro-6gb-ethereum-miner-500-mh-s/ ; https://www.notebookcheck.net/The-NVIDIA-GeForce-RTX-3080-is-an-Ethereum-mining-monster-overclocked-cards-deliver-nearly-100-MH-s-double-the-Radeon-RX-5700-XT.494246.0.html +- [S11] https://www.coindesk.com/tech/2020/12/21/linzhi-begins-rollout-of-long-awaited-ethereum-miner-phoenix ; https://www.theblock.co/post/88622/questions-new-ethash-asic-ethereum (F2Pool: 2,733 MH/s at about 3,000 W) +- [S12] https://miningnow.com/asic-miner/jasminer-x4-2500mh-s/ ; https://www.asicminervalue.com/miners/bitmain/antminer-e9-2-4gh ; https://2miners.com/blog/asic-miners-for-ethereum-antminer-e3-vs-innosilicon-a10-eth-master-comparison/ (the 3% estimate) +- [S13] https://eips.ethereum.org/EIPS/eip-1057 ; https://www.theblock.co/news/ecosystems/2020-02-26-ethereum-community-members-submit-dissenting-progpow-petition-57061 ; https://ethereum.org/roadmap/merge/ ; https://cointelegraph.com/news/bitmains-antminer-e3-to-continue-mining-ether-with-new-update (the E3's 4 GB limit) +- [S14] https://ethereumclassic.org/blog/2020-11-27-thanos-hard-fork-upgrade/ +- [S15] https://www.asicminervalue.com/miners/bitmain/antminer-e9-pro-3-68gh ; https://pool.kryptex.com/device/asic/jasminer/x16-p ; https://whattomine.com/coins/151-eth-ethash/asics +- [S16] https://eprint.iacr.org/2015/946 (Equihash); https://en.wikipedia.org/wiki/Zcash +- [S17] https://variance.hu/2017/05/08/748-solsec-zcash-equihash-teljesitmeny-egy-gtx-1080-ti-kartyabol/ (1080 Ti at 748 Sol/s) +- [S18] https://www.coindesk.com/markets/2018/05/03/bitmains-latest-crypto-asic-can-mine-zcash ; https://coinguides.org/innosilicon-a9-zmaster-50k-sols-equihash-asic/ ; https://support.bitmain.com/hc/en-us/articles/360012223994-Z9-Specifications ; https://www.asicminervalue.com/miners/bitmain/antminer-z11 ; https://d-central.tech/miners/antminer-z15/ +- [S19] https://github.com/ZcashFoundation/zfnd/blob/master/_posts/blog/2018-05-08-statement-on-asics.md ; https://www.coindesk.com/tech/2018/06/28/zcash-votes-against-asic-resistance-in-boon-for-big-miners ; https://electriccoin.co/blog/ecc-roadmap-calls-for-focus-on-wallet-proof-of-stake-and-interoperability/ +- [S20] https://blog.horizen.io/zencash-statement-on-double-spend-attack/ ; https://blog.horizen.io/horizen-zen-statement-on-mining-algorithm/ ; https://forum.zcashcommunity.com/t/list-of-all-coins-projects-on-equihash-asic-resistant-not-resistant/29085 +- [S21] https://en.wikipedia.org/wiki/Bitcoin_Gold ; https://gist.github.com/metalicjames/71321570a105940529e709651d0a9765 +- [S22] https://fluxofficial.medium.com/zels-custom-pow-algorithm-zelhash-activation-in-mid-june-ad3d14d72135 ; https://uploads-ssl.webflow.com/60c73eaed3399e074029d643/60fd7d883200fcde5ceb7049_ZelHash_v1.0.pdf +- [S23] https://github.com/BeamMW/beam/wiki/BEAM-Mining ; https://docs.beam.mw/BeamHashII.pdf ; https://medium.com/minerstat/beamhashiii-beam-forks-to-a-new-algorithm-at-block-777777-dd2aeacc9e5 +- [S24] https://aion.theoan.com/blog/aion-mainnet-launch-kilimanjaro/ ; https://miningpoolstats.stream/zero +- [S25] https://vertcoin.io/history/ ; https://www.newsbtc.com/2014/12/01/vertcoin-introduces-new-pow-algorithm-promises-asic-free-features/ +- [S26] https://cryptoage.com/en/1231-first-asic-miner-lyra2rev2-dayun-zig-z1.html ; https://www.asicminervalue.com/miners/dayun/zig-z1 ; https://whattomine.com/gpus/36-nvidia-geforce-gtx-1080-ti ; https://arxiv.org/pdf/1905.08792 (FPGA bitstreams) +- [S27] https://cryptobriefing.com/vertcoin-vtc-51-percent-attack/ ; https://en.wikipedia.org/wiki/Vertcoin ; https://www.fxstreet.com/cryptocurrencies/news/vertcoin-cryptocurrency-network-fell-victim-to-attack-51-201912030704 +- [S28] https://github.com/vertcoin-project/vertcoin-core/releases/tag/0.14.0 (Lyra2REv3) +- [S29] https://soundcloud.com/vertcoin-talk/vertcoin-talk-episode-24-verthash-fork-happens-january-30th-2021 ; https://crazy-mining.org/en/software/wallets/vertcoin-vtc-instructions-for-mining-on-verthash/ +- [S30] https://coincub.com/mining/how-to-mine-vertcoin-vtc/ +- [S31] https://ravencoin.org/assets/documents/X16R-Whitepaper.pdf ; https://tronblack.medium.com/ravencoin-asic-thoughts-e6c0079609e6 +- [S32] https://cryptoage.com/en/1782-asics-ow-miner-ow1-and-skc-miner-turing-r1-for-the-x16r-algorithm-exist.html ; https://en.cryptonomist.ch/2019/09/17/mining-ravencoin-hashrate/ +- [S33] https://en.cryptonomist.ch/2019/10/02/ravencoin-rvn-hard-fork/ ; https://cryptomining-blog.com/11320-ravencoin-rvn-getting-fpga-mining-support-for-the-x16rv2-algorithm/ +- [S34] https://medium.com/minerstat/kawpow-ravencoin-forks-to-a-new-algorithm-2e730cd09fb3 ; https://tronblack.medium.com/ravencoin-kawpow-expectations-a6a063df58f2 +- [S35] https://github.com/RavenProject/Ravencoin/blob/master/roadmap/README.md ; https://whattomine.com/coins/234-rvn-kawpow/gpus ; https://miningreturns.com/learn/ravencoin-mining-guide +- [S36] https://www.neoxa.net/whitepaper/ ; https://woolypooly.com/en/blog/ravencoin-algorithm +- [S37] https://arxiv.org/pdf/1606.03588 (Egalitarian computing, MTP) +- [S38] https://eprint.iacr.org/2017/497 (Dinur and Nadler); http://blog.zorinaq.com/attacks-on-mtp/ ; https://firo.org/2017/07/21/mtp-audit-and-implementation-bounty.html +- [S39] https://firo.org/2018/12/05/mtp-faq-all-you-need-to-know.html +- [S40] https://firo.org/2021/10/01/firopow-and-instantsend-release.html +- [S41] https://firo.org/2025/11/19/hardfork-successful-nov-2025.html +- [S42] https://docs.decred.org/research/blake-256-hash-function/ +- [S43] https://www.asicminervalue.com/miners/innosilicon/d9-decredmaster ; https://www.asicminervalue.com/miners/obelisk/dcr1 ; https://crypto.news/hardware-companies-are-launching-dedicated-asic-miners-for-decred/ (DCR1 at 28 nm) +- [S44] https://cryptoage.com/en/1254-bitmain-antminer-dr3-7,8-th-s-on-the-algorithm-blake-14r-decred.html ; https://medium.com/luxor/bitmain-antminer-dr5-decred-setup-guide-1c417f5f61fc +- [S45] https://1stminingrig.com/antminer-a3-review-bitmain-surprises-everyone-with-this-new-siacoin-miner/ ; https://medium.com/obelisk-blog/obelisk-update-may-june-2018-260fce12a825 ; https://www.eastshoremining.com/tutorial-innosilicon-s11-siamaster-3-83th-siacoin-miner/ +- [S46] https://www.coindesk.com/markets/2018/10/19/sia-network-releases-hard-fork-code-to-block-crypto-mining-giants ; https://siasetup.info/learn/forks +- [S47] Vorick, "The state of cryptocurrency mining", 13 May 2018: https://archive.sia.tech/the-state-of-cryptocurrency-mining-538004a37f9b (read through https://davidgerard.co.uk/blockchain/2018/05/14/from-sia-an-incendiary-post-on-the-state-of-cryptocurrency-mining-in-2018/ and https://zycrypto.com/asic-manufacturer-shares-important-information-for-token-creators-and-miners/); Bitmain's reply https://blog.bitmain.com/en/bitmain-sia-state-cryptocurrency-mining/ +- [S48] https://medium.com/kadena-io/kadena-public-blockchain-releases-fully-public-testnet-v3-hashing-algorithm-and-mining-api-e230a51c7b26 ; https://www.coindesk.com/markets/2019/11/04/kadena-goes-live-announces-new-token-sale-aiming-for-20-million +- [S49] https://www.asicminervalue.com/miners/goldshell/kd5 ; https://asicmarketplace.com/product/goldshell-kd2-kadena-miner-6-4-th-s/ ; https://asicmarketplace.com/product/bitmain-antminer-ka3-kadena-miner-166th/ +- [S50] https://minerstat.com/hardware/nvidia-rtx-3080-lhr +- [S51] https://bytecoin.org/old/whitepaper.pdf ; https://docs.getmonero.org/proof-of-work/cryptonight/ ; https://en.wikipedia.org/wiki/CryptoNote +- [S52] https://news.8btc.com/bitmain-to-release-antminer-x3-cryptonight-asic-miner-with-220-khs-hashrate ; https://bitcointalk.org/index.php?topic=3127974.0 ; https://www.asicminervalue.com/miners/baikal/bk-n ; https://cointelegraph.com/news/bitmain-announces-new-monero-mining-antminer-x3-cryptos-devs-say-will-not-work +- [S53] https://github.com/monero-project/monero/pull/3253 (v7); https://coinguides.org/monero-network-upgrade-v8-cnv2-beryllium-bullet/ ; https://github.com/SChernykh/CryptonightR (CN-R: chip latency up 2.5x) +- [S54] https://medium.com/@MoneroCrusher/analysis-more-than-85-of-the-current-monero-hashrate-is-asics-and-each-machine-is-doing-128-kh-s-f39e3dca7d78 ; https://beincrypto.com/hashrate-analysis-reveals-asics-account-for-85-of-monero-mining/ +- [S55] https://github.com/tevador/randomx (30 Nov 2019) +- [S56] https://github.com/tevador/RandomX/blob/master/doc/design.md +- [S57] https://github.com/tevador/RandomX/blob/master/doc/specs.md +- [S58] https://xmrig.com/benchmark/5kFcJv (3950X); https://whattomine.com/coins/101-xmr-randomx/gpus (RTX 3090 at 2.0 kH/s, 290 W) +- [S59] https://bt-miners.com/products/bitmain-antminer-x5-monero-miner-212k-bt-miners/ ; https://bitmain.com.vc/news/bitmain-launches-antminer-x9 ; https://pineconeinibox.shop/product/pinecone-matches-inibox-r1x-xmr-edition/ ; https://github.com/xmrig/xmrig/blob/master/doc/ALGORITHMS.md ; https://rfc.tari.com/RFC-0131_Mining ; https://www.theblock.co/post/353240/tari-privacy-network-merged-monero-mining-launch-mainnet +- [S60] https://github.com/tevador/RandomX/blob/master/README.md (the four audits and their cost); https://blog.trailofbits.com/2019/07/02/state/ +- [S61] https://github.com/mimblewimble/docs/blob/master/docs/about-grin/proof-of-work.md ; https://github.com/tromp/cuckoo/blob/master/README.md ; https://github.com/tromp/cuckoo/blob/master/doc/cuckoo.pdf +- [S62] https://forum.grin.mw/t/mid-july-pow-hardfork-cuckaroo29-cuckarood29/5082 ; https://www.cudominer.com/grin-network-update-hard-fork-16th-january-2020/ ; https://forum.grin.mw/t/grin-v5-0-0-network-upgrade-hard-fork-4-january-2021/7895 +- [S63] https://docs.aeternity.com/aeternity-core-concepts/protocol/consensus-mechanisms/cuckoo-cycle-proof-of-work ; https://medium.com/cortexlabs/miners-can-now-test-mine-on-testnet-dolores-in-preparation-for-the-mainnet-launch-cf851d7b0146 +- [S64] https://forum.grin.mw/t/introducing-the-grn1-a-cuckatoo31-asic-from-obelisk/2519 ; https://medium.com/obelisk-blog/grn1-cancellation-announcement-54782c6e3e83 +- [S65] https://bitcointalk.org/index.php?topic=5219851.0 ; https://forum.grin.mw/t/innosilicons-grin-asics-canceled/6932 ; https://ipollo-miners.com/product/ipollo-g1/ +- [S66] https://forum.grin.mw/t/another-cuckatoo-bounty-succesfully-claimed/11739 (Apr 2025) +- [S67] https://eips.ethereum.org/EIPS/eip-1057 ; https://github.com/ifdefelse/ProgPOW +- [S68] https://github.com/ethereum/pm/blob/master/AllCoreDevs-EL-Meetings/Meeting%2052.md ; https://www.coindesk.com/markets/2019/01/04/ethereum-developers-give-tentative-greenlight-to-asic-blocking-code ; https://souptacular.github.io/2020-03-02-progpow-the-ethereum-community-speaks/ ; https://www.coindesk.com/tech/2020/03/06/ethereums-progpow-call-features-frustration-but-little-progress +- [S69] https://leastauthority.com/static/publications/LeastAuthority-ProgPow-Algorithm-Final-Audit-Report.pdf (9 Sep 2019) +- [S70] https://github.com/ethcatherders/progpow-audit ("Bob Rao - ProgPOW Hardware Audit Report Final.pdf", Sep 2019) +- [S71] https://github.com/kik/progpow-exploit (4 Mar 2020); https://github.com/Souptacular/linzhi (Linzhi's 3x to 8x claim) +- [S72] https://cryptoage.com/en/1238-bitcoin-interest-bci-and-new-mining-algorithm-progpow.html ; https://en.wikipedia.org/wiki/Zano_(blockchain_platform) ; https://github.com/sero-cash/serominer ; https://x.com/QuaiNetwork/status/1880037240149057759 ; https://veil-project.com/blog/2020-OhGodAGirl/ +- [S73] https://docs.ergoplatform.com/mining/autolykos/ ; https://ergoplatform.org/en/blog/2019_07_09_after_launch/ ; https://bytwork.com/en/news/khardfork-ergo-07 +- [S74] https://www.hashrate.no/gpus/3090/ERG +- [S75] https://mining.confluxnetwork.org/ ; https://github.com/Conflux-Chain/conflux-rust +- [S76] https://github.com/Conflux-Chain/CIPs/blob/master/CIPs/cip-102.md +- [S77] https://www.kaspafaq.com/sp_accordion_faqs/what-is-kheavyhash/ ; https://github.com/Dagmbisrat/Kaspa-FPGA-Miner +- [S78] https://whattomine.com/coins/352-kas-kheavyhash/gpus/49-nvidia-geforce-rtx-3090 +- [S79] https://www.cryptominerbros.com/product/iceriver-ks0-100gh-s-kas-miner/ ; https://www.asicminervalue.com/miners/iceriver/ks1 ; https://apextomining.com/product/new-bitmain-antminer-ks3-8-3t-3188w-kas-miner-asic-mining-machine-profitable-comining-soon/ ; https://www.asicminervalue.com/miners/bitmain/antminer-ks5-pro-21th +- [S80] https://hashdag.medium.com/kaspa-where-to-part-iv-last-c68717a8d309 (May 2023); https://miningreturns.com/news/kaspa-asic-mining-era-what-you-need-to-know +- [S81] https://x.com/karlsennetwork/status/1829148683104870534 ; https://www.hashrate.no/c/Algorithm_change_for_Karlsen_and_Pyrin ; https://github.com/spectre-project/rusty-spectre ; https://cryptix-network.org/whitepaper ; https://network.hoosat.fi/public/htn-whitepaper-2.pdf ; https://bugna.org/ +- [S82] https://spec.nexa.org/mining/NexaPOW/ +- [S83] https://www.cryptominerbros.com/product/dragonball-miner-a21-nexa-miner/ ; https://whattomine.com/coins/357-nexa-nexapow +- [S84] https://docs.alephium.org/frequently-asked-questions/ ; https://medium.com/@alephium/one-year-of-mainnet-b7ed5d3024ee +- [S85] https://hashrate.no/gpus/3090/ALPH +- [S86] https://www.asicminervalue.com/miners/goldshell/al-box ; https://mineshop.eu/bitmain-antminer-al1 ; https://www.zeusbtc.com/Asic-Miner/Asic-Miner-Details.asp?ID=3719 +- [S87] https://fips.ironfish.network/fips/fip-3-memory-hard-mining-algorithm ; https://fips.ironfish.network/fips/fip-10-hardfork-1 ; https://github.com/iron-fish/fish-hash ; https://github.com/karlsen-network/fish-hash-plus +- [S88] https://medium.com/nervosnetwork/a-decentralized-mainnet-launch-for-nervos-ckb-9cb119d15540 +- [S89] https://www.asicminervalue.com/miners/bitmain/antminer-k5-1130gh ; https://www.asicminervalue.com/miners/goldshell/ck5 ; https://2miners.com/blog/nervos-ckb-network-hashrate-increased-asics-are-the-cause/ +- [S90] https://www.coindesk.com/markets/2020/02/04/handshakes-uncensorable-web-domains-go-live-on-mainnet +- [S91] https://www.goldshell.com/news/goldshell-announces-best-handshakehns-miner-hs1-coming-soon/ ; https://www.asicminervalue.com/miners/goldshell/hs3 +- [S92] https://radiantblockchain.org/ ; https://d-central.tech/miners/rxd-rx0/ +- [S93] https://cryptoage.com/en/2929-video-card-hashrate-based-on-the-sha512-256d-algorithm-cryptocurrency-mining-radiant-rxd.html +- [S94] https://cryptomining-blog.com/11865-mining-zano-using-the-progpowz-proof-of-work-algorithm/ +- [S95] https://github.com/dynexcoin/DynexSolve ; https://minerstat.com/coin/DNX/faq +- [S96] https://docs.warthog.network/janushash/ ; https://github.com/CoinFuMasterShifu/Janushash +- [S97] https://docs.xelis.io/network-upgrades +- [S98] https://docs.verus.io/overview/verus-proof-of-power.html + +Papers: +- [P1] https://www.microsoft.com/en-us/research/publication/moderately-hard-memory-bound-functions/ +- [P2] https://www.wisdom.weizmann.ac.il/~naor/PAPERS/mem.pdf +- [P3] https://www.tarsnap.com/scrypt/scrypt.pdf +- [P4] https://eprint.iacr.org/2014/238 +- [P5] https://eprint.iacr.org/2016/115 +- [P6] https://eprint.iacr.org/2016/759 +- [P7] https://eprint.iacr.org/2016/989 +- [P8] https://www.cryptolux.org/images/d/d0/Argon2ESP.pdf +- [P9] https://eprint.iacr.org/2016/027 +- [P10] https://eprint.iacr.org/2015/227 +- [P11] https://eprint.iacr.org/2013/525 +- [P12] https://eprint.iacr.org/2015/136 +- [P13] https://eprint.iacr.org/2017/225 +- [P14] https://eprint.iacr.org/2018/221 +- [P15] https://eprint.iacr.org/2017/443 +- [P16] https://eprint.iacr.org/2015/946 +- [P17] https://arxiv.org/abs/1606.03588 +- [P18] https://eprint.iacr.org/2017/497 +- [P19] https://eprint.iacr.org/2014/059 +- [P20] https://da-data.blogspot.com/2014/03/a-public-review-of-cuckoo-cycle.html ; http://www.cs.cmu.edu/~dga/crypto/cuckoo/analysis.pdf +- [P21] https://arxiv.org/abs/1902.00112 +- [P22] https://ieeexplore.ieee.org/document/8516911/ +- [P23] http://www.vldb.org/pvldb/vol13/p898-feng.pdf +- [P24] https://arxiv.org/abs/2002.11064 + +Economics and silicon: +- [E1] https://europractice-ic.com/schedules-prices-2025/ +- [E2] https://newsletter.semianalysis.com/p/the-dark-side-of-the-semiconductor (24 Jul 2022) +- [E3] https://semiengineering.com/big-trouble-at-3nm/ (21 Jun 2018); https://semiengineering.com/the-increasingly-uneven-race-to-3nm-2nm/ (24 May 2021); https://semiengineering.com/what-will-that-chip-cost/ (30 Oct 2023) +- [E4] https://siliconanalysts.com/analysis/fabless-startup-tapeout-cost-guide (1 Mar 2026, secondary) +- [E5] https://patentpc.com/blog/chip-manufacturing-costs-in-2025-2030-how-much-does-it-cost-to-make-a-3nm-chip (secondary, approximate) +- [E6] https://techcrunch.com/2018/09/26/bitmain-hong-kong-ipo/ ; https://bitcoinmagazine.com/markets/bitmain-ipo-prospectus-reveals-offering-may-be-gamble-investors ; https://www.chaincatcher.com/en/article/2057998 +- [E7] https://www.sec.gov/Archives/edgar/data/1780652/000119312519297270/d773846d424b4.htm +- [E8] https://michaeltaylor.org/papers/bitcoin_taylor_cases_2013.pdf ; https://michaeltaylor.org/papers/Taylor_Bitcoin_IEEE_Computer_2017.pdf +- [E9] https://www.tomshardware.com/news/amd-unveils-more-ryzen-3d-packaging-and-v-cache-details-at-hot-chips (Aug 2021); https://www.graphcore.ai/posts/introducing-second-generation-ipu-systems-for-ai-at-scale ; https://www.theregister.com/software/2020/09/29/groq-is-hard-to-grok-but-reckons-its-ai-chips-roq-ex-googlers-unorthodox-design-now-shipping-to-customers/1170931 +- [E10] https://newsletter.semianalysis.com/p/tsmcs-3nm-conundrum-does-it-even (21 Dec 2022); https://fuse.wikichip.org/news/7343/iedm-2022-did-we-just-witness-the-death-of-sram/ ; https://www.tomshardware.com/news/no-sram-scaling-implies-on-more-expensive-cpus-and-gpus ; https://images.nvidia.com/aem-dam/Solutions/geforce/blackwell/nvidia-rtx-blackwell-gpu-architecture.pdf (GB202: 128 MB L2 on the full die, 96 MB on the RTX 5090, 750 mm^2) +- [E11] CoinMarketCap historical snapshots, https://coinmarketcap.com/historical/YYYYMMDD/ for the dates in the section 2.5 table (pulled 5 Oct 2026); HBM pricing https://www.trendforce.com/presscenter/news/20240506-12125.html and https://www.nextplatform.com/2024/02/27/he-who-can-pay-top-dollar-for-hbm-memory-controls-ai-training/ ; the E3's DDR3 and the A10's presumed GDDR6 from the ProgPoW FAQ https://medium.com/@ifdefelse/progpow-faq-6d2dce8b5c8b (Jan 2019, read through secondary coverage) and https://coingeek.com/memory-limitations-prompt-bitmain-antminer-e3-to-halt-etc-support/ + +Latency: +- [L1] https://terpconnect.umd.edu/~blj/papers/memsys2018-dramsim.pdf +- [L2] https://arxiv.org/abs/1712.08304 +- [L3] https://docs.nvidia.com/nsight-compute/ProfilingGuide/index.html ; https://developer.download.nvidia.com/video/gputechconf/gtc/2020/presentations/s21819-optimizing-applications-for-nvidia-ampere-gpu-architecture.pdf +- [L4] https://arxiv.org/abs/2604.15751 diff --git a/docs/analysis/card-lifetime-2026-10-05.md b/docs/analysis/card-lifetime-2026-10-05.md new file mode 100644 index 000000000..28c7d3b72 --- /dev/null +++ b/docs/analysis/card-lifetime-2026-10-05.md @@ -0,0 +1,85 @@ +# Card lifetime per tier: how many years a card keeps mining + +5 October 2026. Consequences review, sub-agent of the consequences reviewer. Desk arithmetic only; nothing was run. + +## 1. Inputs + +| Input | Source | Value used | +|---|---|---| +| Dataset schedule | `docs/spec/01-lottery-hash.md` 432 to 437 | 2 GiB at genesis plus 0.5 GiB a year (2,048 + 512 x years MiB) | +| Index mapping (a) | same file, line 442 | multiply-shift: the dataset grows every day, continuous | +| Index mapping (b) | same file, line 442 | power-of-two steps 2, 4, 8 GiB on the schedule's average: 4 GiB at year 4, 8 GiB at year 12; my extrapolation: 16 GiB at year 28, 32 GiB at year 60 | +| Scratch per resident warp | `igneum-wt-ca2-cache/docs/plans/hot-table.md` 66 to 73 | 32 or 128 KiB per warp; 5090 = 170 SMs x 48 warps = 8,160 (approximate, from memory) | +| Hot table, buffers | same file, 70 | hot table 32, 64 or 96 MiB (96 used here); buffers 128 MiB | +| Cache | hot-table.md 70 (resident, 256 MiB in every total) against `igneum-wt-ca2-era/docs/plans/era-layout.md` 93 ("resident only while the day's dataset is built, then free") | both readings carried: resident = worst case, freed = best case. The two plans disagree and gate 1 should say which | +| Cache growth | `igneum-wt-ca2-coord/docs/plans/counter-asic-2-status.md` 79 (layer 6 option C) | 256 MiB at genesis, 512 MiB at year 4, 1 GiB at year 12; by the same rule 2 GiB at year 28, 4 GiB at year 60 | +| Budget rule | same file, 17: the whole working set stays under 6 GB on an 8 GB card | my reading: 75% of card memory at every tier. Apple: 50% of unified memory, because macOS, the display and the node share it; that share is my assumption | +| Public claims | `site/index.html` 443, 461; `site/litepaper.html` 560; `docs/evidence.md` | quoted in Table 3. evidence.md has no row on card lifetime | + +Card memory is binary (8 GB = 8,192 MiB). The hot-table row "An 8 GB card at 5090 occupancy" (line 73) counts 8,160 warps; a real 8 GB card has 20 to 24 SMs, so its scratch is about a tenth of that row. Resident warps below are SMs x 48 (NVIDIA Ampere and later), SM counts from memory, approximate; Apple uses the 2,048 warps the Metal harness launches (hot-table.md 66). + +## 2. Table 1: non-dataset working set per tier (MiB) + +Worst = scratch 128 KiB, cache resident. Columns g / y4 / y12 = genesis, year 4, year 12 (the cache doublings). Freed = era-layout's reading, constant over the years. + +| Tier | Card assumed (SMs, approximate) | Warps | Scratch 128 KiB | Scratch 32 KiB | Cache resident, 128 KiB: g / y4 / y12 | Cache resident, 32 KiB: g / y4 / y12 | Cache freed: 128 / 32 KiB | +|---|---|---|---|---|---|---|---| +| 4 GB | GTX 1650 (14 SMs x 32 warps, Turing) | 448 | 56 | 14 | 536 / 792 / 1,304 | 494 / 750 / 1,262 | 280 / 238 | +| 8 GB | RTX 3050 (20) | 960 | 120 | 30 | 600 / 856 / 1,368 | 510 / 766 / 1,278 | 344 / 254 | +| 12 GB | RTX 3060 (28) | 1,344 | 168 | 42 | 648 / 904 / 1,416 | 522 / 778 / 1,290 | 392 / 266 | +| 16 GB | RTX 5060 Ti (36) | 1,728 | 216 | 54 | 696 / 952 / 1,464 | 534 / 790 / 1,302 | 440 / 278 | +| 24 GB | RTX 4090 (128) | 6,144 | 768 | 192 | 1,248 / 1,504 / 2,016 | 672 / 928 / 1,440 | 992 / 416 | +| 32 GB | RTX 5090 (170) | 8,160 | 1,020 | 255 | 1,500 / 1,756 / 2,268 | 735 / 991 / 1,503 | 1,244 / 479 | +| Apple 8 to 64 GB | M-series, harness launch count | 2,048 | 256 | 64 | 736 / 992 / 1,504 | 544 / 800 / 1,312 | 480 / 288 | + +Every row = scratch + 96 (hot table) + 128 (buffers) + cache (256 / 512 / 1,024 when resident). The freed reading still peaks at dataset + cache during the daily build, but that peak is smaller than the resident total whenever hashing pauses for the build, so the freed column is the steady-state set. + +## 3. Table 2: dataset room and the year the dataset outgrows it + +Room = usable memory (75%, Apple 50%) minus Table 1. Worst = 128 KiB scratch, cache resident (room shrinks at years 4, 12, 28, 60). Best = 32 KiB scratch, cache freed. Option (a): the year 2,048 + 512 x y exceeds the room. Option (b): the first step the room cannot hold; the card mines up to that day. + +| Tier | Usable MiB (share) | Room at genesis, worst / best | (a) ends, years, worst / best | (b) ends, year, worst / best | +|---|---|---|---|---| +| 4 GB | 3,072 (75%) | 2,536 / 2,834 | 1.0 / 1.5 | 4 / 4 | +| 8 GB | 6,144 (75%) | 5,544 / 5,890 | 6.3 / 7.5 | 12 / 12 | +| 12 GB | 9,216 (75%) | 8,568 / 8,950 | 12.0 / 13.5 | 12 / 28 | +| 16 GB | 12,288 (75%) | 11,592 / 12,010 | 17.1 / 19.5 | 28 / 28 | +| 24 GB | 18,432 (75%) | 17,184 / 18,016 | 28.0 / 31.2 | 28 / 60 | +| 32 GB | 24,576 (75%) | 23,076 / 24,097 | 37.6 / 43.1 | 60 / 60 | +| Apple 8 GB | 4,096 (50%) | 3,360 / 3,808 | 2.6 / 3.4 | 4 / 4 | +| Apple 16 GB | 8,192 (50%) | 7,456 / 7,904 | 10.1 / 11.4 | 12 / 12 | +| Apple 32 GB | 16,384 (50%) | 15,648 / 16,096 | 25.1 / 27.4 | 28 / 28 | +| Apple 64 GB | 32,768 (50%) | 32,032 / 32,480 | 55.1 / 59.4 | 60 / 60 | + +What the table says per tier: + +| Tier | Reading | +|---|---| +| 4 GB | Mines at genesis with 488 to 786 MiB spare. Under (a) it is out within 1 to 1.5 years. Under (b) it lasts to the year-4 step, as the spec's own remark says (line 442) | +| 8 GB | 6 to 7.5 years under (a). 12 years under (b): "more than a decade" is true only under (b), and only just | +| 12 GB | The year-12 cache doubling (1 GiB resident) is what ends it, under both options, if the cache stays resident. With the cache freed it reaches year 28 under (b). This tier's lifetime is decided by the cache residency question, not by the dataset | +| 16 GB | 17 to 19.5 years under (a), year 28 under (b) | +| 24 GB | Under the resident reading the year-28 cache doubling (2 GiB) ends it the same day under both options. Freed: 31 years or year 60 | +| 32 GB | 38 to 43 years under (a), year 60 under (b). Not a constraint for any plan | +| Apple 8 GB | 2.6 to 3.4 years under (a), year 4 under (b). The base 8 GB Apple laptop is a short-lived miner | +| Apple 16 GB | 10 to 11.4 years under (a), year 12 under (b): the same shape as an 8 GB card | +| Apple 32 / 64 GB | 25 years and 55 years or more. No constraint | + +Proving is a separate budget (the 15.6 GB peak the 12 GB mine-and-prove question came from); this file covers mining only. + +## 4. Table 3: the public sentences against the numbers + +| Where | Sentence now | What the tables give | Proposed sentence (Josh decides the wording) | +|---|---|---|---| +| `site/index.html` 443 | Memory: "2 GB, fixed" (RandomX) / "2 GB, growing" (Igneum) | 2 GiB at genesis, plus 0.5 GiB a year on average under either option | "2 GB, growing 0.5 GB a year". The row is right; the rate is the useful addition | +| `site/index.html` 461 | "Any 4 GB card, approximate." | True at genesis (2,584 to 2,834 MiB of a 3,072 MiB budget). Ends at 1 to 1.5 years under (a), year 4 under (b) | "Any 4 GB card at launch, 8 GB for the long run, approximate." | +| `site/litepaper.html` 560 | "a 4 GB card mines for about four years and an 8 GB card for more than a decade, approximate." | 4 GB: 1 to 1.5 years (a) or 4 years (b). 8 GB: 6.3 to 7.5 years (a) or 12 years (b). Both numbers hold only under option (b) | If gate 1 picks (b): "a 4 GB card mines until the first dataset step at year 4, an 8 GB card until the second at year 12 and a 16 GB card until year 28, approximate." If (a): "a 4 GB card mines for about a year, an 8 GB card for about seven and a 16 GB card for about seventeen, approximate." | +| `site/litepaper.html` 560 | "12 GB or more proves full shards." | Not a lifetime claim; left as is. For mining, 12 GB lasts 12 years with the cache resident, year 28 with it freed under (b) | No change from this file | +| `docs/evidence.md` | No row on card lifetime | The litepaper sentence is a public claim with no row | Add a row, label "designed", sources: spec 1.13.3 and this file; status moves to "tested" once a 4 GB and an 8 GB card run the genesis working set under the cap | + +## 5. Reading + +- The two index-mapping options end on the same day where a cache doubling takes the last of the room. With the cache resident that is the 12 GB tier at year 12 and the 24 GB tier at year 28 (Table 2, worst column). Under option (b) every tier ends on a step day by construction, so a tier ends on the same day under both options exactly when option (a) also ends it on a doubling day. +- Everywhere else option (b) is kinder: 4 GB gains about 2.5 years, 8 GB about 5, 16 GB about 10. The site and litepaper numbers are option (b) numbers. If gate 1 picks (a), both public sentences are wrong today by 2.5 to 5 years. +- The cache residency disagreement (hot-table.md 70 against era-layout.md 93) decides the 12 GB tier's lifetime (12 against 28 years) and nothing else. It should be settled at gate 1 beside the mapping choice. +- The 75% rule is my generalisation of "under 6 GB on an 8 GB card"; at 4 GB it leaves 1 GB for the driver and the display, which a headless rig would not need. A 4 GB card on a bare Linux rig might hold out to year 2 under (a). Not measured. diff --git a/docs/analysis/chip-model-v3.md b/docs/analysis/chip-model-v3.md new file mode 100644 index 000000000..1d47b2f72 --- /dev/null +++ b/docs/analysis/chip-model-v3.md @@ -0,0 +1,100 @@ +# The on-die-cache recompute chip against the RTX 5090, class v2 and class v3, everything combined + +5 October 2026 (night), Counter ASIC 2.0, worker ca2-mixer. The model is M16's +(`docs/analysis/m16-recompute-attacker-2026-10-05.md`): the strongest chip the plan has priced holds the whole +cache in SRAM and derives every dataset item instead of reading it, so its cost per hash is item derivations, +and its rate at a 50 T op/s integer budget (an RTX 5090's, approximate) is `50 T / (ops per hash)`. Nothing here +is a measurement of a chip; every GPU figure says where it was measured. "Approximate" marks a figure from memory. + +## 1. Inputs + +| Input | Value | Source | +|---|---|---| +| Items per hash | 128 (one item per load, 128 loads per hash, median 128.00 distinct) | spec 01 sections 1.4.2 and 1.8.5; the 20,000-program census | +| Integer operations per mixer application | about 130 | spec 01 section 1.8.4 | +| Mixer applications per item | 9 under v2; 36 under v3 (`m = 4`, `docs/plans/mixer-x4.md`) | `memhard::Shape::mixers_per_item` | +| Integer operations per item | 1,170 (v2); 4,680 (v3) | 9 x 130; 36 x 130 | +| Integer operations per hash | 149,760 (v2, "150,000"); 599,040 (v3, "600,000") | 128 x the above | +| Chip integer budget | 50 T op/s (approximate: 21,760 ALUs at about 2.4 GHz, one 32-bit operation each per clock) | M16 section 3 | +| Fixed-function factor | 3x (approximate, from memory: 2x to 5x is the usual credit for a pipeline with no scheduling or divergence) | M16 section 3 | +| RTX 5090, version 2 programs, measured | 136.1 MH/s (readwidth, tonight, `docs/plans/read-width.md`, pack w4 on PC 2); 139.7 MH/s (M11, 4 October, `docs/bench-log.md`) | this analysis uses tonight's 136.1 as the denominator and quotes both | +| RTX 5090 at w16 (16-byte loads), measured | 139.8 MH/s | readwidth table, tonight (the width stays 4 B: w16 closes nothing) | +| Cache mirror, 256 MiB, N5 headline density | 128 mm^2, $46 per good die (64 mm^2, $21 at the bit-cell lower bound) | `docs/analysis/sram-mirror.md` revision 2, sections 4 and 5 (`ca2-analysis` e6085c6) | +| Cache mirror plus a 96 MB hot table, N5 headline | 175 mm^2, $68 | same, so a hot table costs 0.49 mm^2 and $0.23 per MB (linear, approximate) | +| 512 MiB and 1 GiB mirrors, N5 headline | 255 mm^2 and 510 mm^2; $111 to $306 | same, section 4 (the growth rule's cache at years 4 and 12, priced at today's node) | +| GPU-class die | 750 mm^2 (the equal-silicon comparison) | M16 section 3 | +| CPU verifier, one M5 Max core (loaded, load average 5.6; ratios are the measurement) | v2 1.31 to 1.36 ms per unit, x4 1.92 to 1.96 (1.45x), x8 2.79 (2.1x); worst cold 1.58 / 2.04 / 2.94 ms | `docs/plans/mixer-x4.md` section 6.4, 5 October 2026 21:40 UTC | + +## 2. The rows + +Chip rate = 50 T op/s / ops per hash. "Bare" = chip rate / 136.1 MH/s. "With the factor" = bare x 3. "Equal +silicon" = bare x (750 - SRAM) / 750 x 3: the SRAM takes die area the logic does not get, the M16 convention +("minus the area the SRAM takes"). SRAM in mm^2 and dollars at the N5 headline density. + +| Row | Mixer | Ops per hash | Chip rate at 50 T op/s | SRAM the chip holds | mm^2 / $ (N5 headline) | Bare gain against 136.1 MH/s | With the 3x factor | Equal silicon, SRAM deducted, with the factor | +|---|---|---|---|---|---|---|---|---| +| v2 as shipped (the M16 and scratch-soundness row) | x1 | 149,760 | 334 MH/s | 256 MiB | 128 / $46 | 2.45x (2.39x against 139.7) | 7.4x | 6.1x | +| v2 at w16 (not adopted; the chip's cost is items, not bytes: unchanged) | x1 | 149,760 | 334 | 256 MiB | 128 / $46 | 2.39x against 139.8 | 7.2x | 5.9x | +| x4 (the candidate measured beside v3; not v3) | x4 | 599,040 | 83.5 MH/s | 256 MiB | 128 / $46 | 0.61x | 1.84x | 1.53x | +| MEASURED, NOT ADOPTED (layer 5 decided out of v3 on the PC rows, coordinator 21:40 UTC): v3 plus a 32 MiB hot table, added form (16 dataset loads and k hot loads): the honest card pays the hot loads, this chip pays SRAM only | x4 | 599,040 (a hot load is one SRAM read, no item) | 83.5 | 288 MiB | 144 / $53 | 0.66x at the Mac's g = 0.93 (126.6 MH/s); 0.70x at the 5090's g = 0.87 (118.4); the 9070 XT's g 0.84 | 1.98x (Mac g), 2.11x (5090 g) | 1.60x, 1.71x | +| MEASURED, NOT ADOPTED: v3 plus a 64 MiB hot table, added form | x4 | 599,040 | 83.5 | 320 MiB | 160 / $61 | 0.71x at the Mac's g = 0.87 (118.4 MH/s); 0.73x at the 5090's g = 0.84 (114.3); the 9070 XT's g 0.80 | 2.12x (Mac g), 2.19x (5090 g) | 1.67x, 1.73x | +| v3 at year 4 (cache 512 MiB under option C, dataset 4 GiB), no hot table | x4 | 599,040 | 83.5 | 512 MiB | 255 / $111 | 0.61x | 1.84x | 1.21x | +| v3 at year 12 (cache 1 GiB, dataset 8 GiB) | x4 | 599,040 | 83.5 | 1 GiB | 510 / $306 | 0.61x | 1.84x | 0.59x | +| **v3: mixer x8** (decided 22:05 UTC under the delegated rule: verify 2.1 ms per unit on one Mac core against the 10 ms gate, the daily 1 GiB build 23 to 77 ms on the 5090 and the 9070 XT) | x8 | 1,198,080 | 41.7 | 256 MiB | 128 / $46 | 0.31x | 0.92x | 0.76x | +| x8 at year 4 | x8 | 1,198,080 | 41.7 | 512 MiB | 255 / $111 | 0.31x | 0.92x | 0.61x | + +The era draws of spec 1.13.1 cost the chip nothing in this model: the mixer round count is not drawn, the op +weights and fold rotations change the program, not the item derivation, so the chip's ops per hash stand. The +width rule (4-byte loads kept) changes nothing either: w16 would have moved the honest denominator by 2.7% and the +chip's cost not at all. + +Arithmetic, row v3: 36 x 130 = 4,680 ops per item; x 128 = 599,040 per hash; 50 x 10^12 / 599,040 = 83.5 x 10^6 +hashes per second; 83.5 / 136.1 = 0.613; x 3 = 1.84; equal silicon (750 - 128) / 750 = 0.829, x 1.84 = 1.53. +Hot table rows: 32 MiB x 0.49 mm^2 per MB = 16 mm^2, 64 MiB = 32 mm^2 (the 96 MB column of `sram-mirror.md` +scaled linearly); (750 - 144) / 750 = 0.808 and (750 - 160) / 750 = 0.787. The honest denominator in the added +form is the v2 rate times `g`, the card's measured ratio with the hot loads added: on the M5 Max tonight +`g = 0.93 / 0.87 / 0.83` at 32 / 64 / 96 MiB (the cache agent, relayed by the coordinator at 21:23 UTC; +`docs/plans/hot-table.md` carries the runs); the 5090's and the 9070 XT's `g` are the PC rows, owed, and until they +land the row carries the Mac's `g` against the 5090's rate, which is a mixed figure and is marked so. Year 4 and 12 rows: the mirror of +`sram-mirror.md` section 4 at N5 for 512 MiB and 1 GiB plus the 64 MiB table, at today's density (the node of +those years is denser by about 1.8x at year 10 on the trend the same file cites; the row is a floor on the area, +not a forecast). + +## 3. The margin, plainly + +The combined headline row is the mixer row alone (layer 5 is out: the added form costs the 5090 13 to 16 percent +and the 9070 XT 16 to 20 percent against the 0.97 bar, coordinator 21:40 UTC; the width stays 4 bytes; the era +draws and the cache growth cost this chip nothing at year 0), and class v3 is x8 (decided 22:05 UTC). The headline: +**the on-die-cache recompute chip at 50 T op/s reaches 41.7 MH/s against the 5090's 136.1, 0.31x bare, 0.92x with +the 3x fixed-function factor, 0.76x with the mirror's area deducted: under 1x with the factor, 0.92x, a margin of 8 +percent on the factor (a 3.3x factor reads 1.0x) and of 9 percent on the budget (55 T op/s reads 1.0x).** The x4 +candidate, measured beside it, read 1.84x and 1.53x. The hot-table rows above are kept as measured, not adopted: +against THIS chip an added hot table is a cost to the honest card and none to the chip, so it would have moved the +row the wrong way by the card's own `g`. The margin, plainly: + +- the 3x fixed-function factor is approximate and from memory; at 3.3x the equal-budget row reads 2.0x; +- the denominator is one card's measured rate on one night (136.1 against 139.7 the night before: 2.6% apart); +- the 50 T op/s budget is approximate; a chip at 55 T op/s reads 2.0x; +- the hot table in the added form lowers the honest denominator by whatever the hot loads cost the GPU (owed from + the PC rows), which raises the chip's gain by the same share, 1.84x or more if the hot loads are free, higher if + not; the hot table's only cost to this chip is 16 to 32 mm^2 of die. + +What keeps it under 1x is the mixer, and nothing else in Counter ASIC 2.0 moves this chip (the scratch at any share +gave 2.4x, `docs/analysis/scratch-soundness.md` section 3.4; the hot table taxes the DRAM-only chip, not this one; +the cache growth taxes it only in die area, which is cheap at year 0 and real at year 12). The next levers, in +order: + +1. Mixer x16 (the next step of the same lever): 0.16x bare and 0.46x with the factor against 136.1; the verifier + by the measured increments (+0.63 ms at x4, +1.46 at x8 on the M5 Max core: about +3.1 ms at x16, 3.7 ms per + unit, 9 ms on a 2.5x slower laptop core, approximate) is at the edge of the 10 ms gate, so a 2019-class laptop + core measurement (O-1.14) decides it, not this model. +2. The hot table: adopted or not on the PC rows (`docs/plans/hot-table.md`); in the added form it costs the GPU + 7 to 17 percent on the Mac and the chip die area only, so against this chip it is a lever in the wrong + direction and against a DRAM-only chip the first lever; if it is adopted, the mixer must carry the extra `1/g` + (x8 at g = 0.87 reads 1.06x at the equal budget, 0.84x with the SRAM deducted). + +## 4. What this does not settle + +The items of M16 section 5 stand: the inline kernel on NVIDIA with a 64 MiB cache inside L2 (a measured point +under the "50 T op/s" row) is a PC job not yet run; the time-memory curve (O-1.6) is not drawn; the mixer has had +no cryptanalysis, and a shortcut inside it cuts the 4,680 directly; no chip has been priced beyond its SRAM. diff --git a/docs/analysis/int8-matrix-family.md b/docs/analysis/int8-matrix-family.md new file mode 100644 index 000000000..0876503ac --- /dev/null +++ b/docs/analysis/int8-matrix-family.md @@ -0,0 +1,175 @@ +# Layer 7: the integer matrix family (INT8 x INT8 into INT32) as a reserved instruction family, design + +5 October 2026 (night), Counter ASIC 2.0 (`docs/plans/counter-asic-2.md`, layer 7), branch `ca2-analysis`. Design +only: nothing here touches the generator, a vector or a node. Every figure is cited (vendor document, URL, section) or +measured (machine, date, command) or labelled approximate. + +## 1. The primitive per vendor, from the vendor documents + +| Vendor, hardware | Per-lane dot4 (4 bytes x 4 bytes into a 32-bit integer) | Warp or wave matrix (int8 tiles, int32 accumulate) | Source | +|---|---|---|---| +| NVIDIA, sm_61 and later (Pascal on) | PTX `dp4a.atype.btype d, a, b, c` with `.atype = .btype = {.u32, .s32}`: "Four-way byte dot product which is accumulated in 32-bit result"; semantics `d = c; for i in 0..3: d += Va[i] * Vb[i]` with the bytes sign- or zero-extended by type; introduced in PTX ISA 5.0, "Requires sm_61 or higher". CUDA: `__device__ int __dp4a(int srcA, int srcB, int c)` ("Four-way signed int8 dot product with int32 accumulate") and the unsigned form, plus `char4`/`uchar4` overloads | `mma.sync` with `.u8`/`.s8` A and B and `.s32` C and D: shape `.m8n8k16` "requires sm_75 or higher" (Turing on, PTX 6.5); shapes `.m16n8k16` and `.m16n8k32` require sm_80 (Ampere on, PTX 7.0); sparse `.m16n8k32` and `.m16n8k64` with `.u8`/`.s8` also exist | PTX ISA 9.4, section 9.7.1.24 (dp4a) and 9.7.16.5 (mma), https://docs.nvidia.com/cuda/parallel-thread-execution/index.html ; CUDA Math API, integer intrinsics, https://docs.nvidia.com/cuda/cuda-math-api/cuda_math_api/group__CUDA__MATH__INTRINSIC__INT.html ; read 5 October 2026 | +| AMD RDNA 3 (gfx11) | `v_dot4_i32_iu8` (VOP3P; each operand signed or unsigned by a per-operand bit, optional clamp) reached from clang/HIP/OpenCL C as `__builtin_amdgcn_sudot4(bool a_signed, int a, bool b_signed, int b, int acc, bool clamp)` (LLVM feature `dot8-insts`: "Has v_dot4_i32_iu8, v_dot8_i32_iu4 instructions"); `v_dot4_u32_u8` as `__builtin_amdgcn_udot4` (`dot7-insts`: "Has v_dot4_u32_u8, v_dot8_u32_u4"); `v_dot4_i32_i8` as `__builtin_amdgcn_sdot4` (`dot1-insts`: "Has v_dot4_i32_i8 and v_dot8_i32_i4"). gfx11's common feature set carries dot7, dot8, dot9, dot10 and dot12 | `V_WMMA_I32_16X16X16_IU8`: `__builtin_amdgcn_wmma_i32_16x16x16_iu8_w32` and `_w64` (feature `wmma-256b-insts`), a 16x16x16 tile per wave | LLVM `clang/include/clang/Basic/BuiltinsAMDGPU.td` (main, read 5 October 2026), lines defining `__builtin_amdgcn_sdot4`, `udot4`, `sudot4`, `wmma_i32_16x16x16_iu8_w32`; AMD GPUOpen, "How to accelerate AI applications on RDNA 3 using WMMA", https://gpuopen.com/learn/wmma_on_rdna3/ ; the RDNA 3 ISA PDF itself did not download tonight (AMD's CDN refused curl and the fetcher timed out), so the instruction names are from the compiler and GPUOpen, not quoted from the ISA guide | +| AMD RDNA 4 (gfx12, the 9070 XT) | the same `sudot4` and `udot4` builtins: LLVM's `FeatureISAVersion12_Generic` carries `FeatureDot7Insts` and `FeatureDot8Insts` and not `FeatureDot1Insts`, so `__builtin_amdgcn_sdot4` is NOT exposed on gfx12 and `sudot4` with both operands signed is the signed form to use | `__builtin_amdgcn_wmma_i32_16x16x16_iu8_w32_gfx12` and `_w64_gfx12` (feature `wmma-128b-insts`): the int8 WMMA exists on RDNA 4 with a narrower per-lane operand (2 ints per lane for A and B against 4 on RDNA 3); AMD's RDNA 4 WMMA guide names the same builtin | LLVM `llvm/lib/Target/AMDGPU/AMDGPU.td` (`FeatureISAVersion12_Generic`) and `BuiltinsAMDGPU.td` (main, 5 October 2026); AMD GPUOpen, "WMMA guide for AMD RDNA 4 architecture GPUs, part 2", https://gpuopen.com/learn/wmma-guide-amd-rdna-4-gpus-part-2/ ; the RDNA 4 ISA guide (AMD document 70651, April 2025) was not readable tonight (docs.amd.com returned 401 to a direct fetch) | +| AMD CDNA 3 (MI300) | the same VOP3P dot instructions (approximate: not checked in the CDNA 3 guide tonight) | `V_MFMA_I32_16X16X32_I8` and `V_MFMA_I32_32X32X16_I8` (opcodes 87 and 86 in the VOP3P-MFMA table) | AMD Instinct MI300 CDNA 3 ISA Reference Guide, 5 August 2025, https://www.amd.com/content/dam/amd/en/documents/instinct-tech-docs/instruction-set-architectures/amd-instinct-mi300-cdna3-instruction-set-architecture.pdf (downloaded and grepped 5 October 2026) | +| AMD, OpenCL on Adrenalin (Windows) | `cl_khr_integer_dot_product` is NOT in the 24 extensions Adrenalin lists for gfx1201 (the list: fp64, the int32 and int64 atomics, 3d image writes, byte addressable store, fp16, gl sharing, amd device attribute query, amd media ops and media ops2, d3d10, d3d11 and dx9 sharing, image2d from buffer, subgroups, gl event, depth images, mipmap image and writes, amd copy buffer p2p); the platform is OpenCL 2.1 so the OpenCL C 3.0 feature macro `__opencl_c_integer_dot_product_input_4x8bit` is not expected. What IS reachable: the Adrenalin OpenCL compiler is clang (driver string `PAL,LC`), and `__builtin_amdgcn_sudot4` from OpenCL C has been shown to emit `V_DOT4_I32_IU8` on a Radeon 780M (gfx1103, RDNA 3, driver 32.0.31041, Windows 11) at 2.7x the scalar fallback (1.27 to 3.42 TMAC/s) | not from OpenCL C | Adrenalin 26.9.2 extension list, https://geeks3d.com/20260904/amd-radeon-adrenalin-26-9-x-graphics-driver/ ; the OpenCL C route: https://github.com/1640675651/CPPminer/pull/1 (third party, one machine; the 9070 XT run of this document's probe is the check) ; `cl_khr_integer_dot_product` itself: OpenCL C 3.0 specification section 6.2.2.16, `int dot(char4, char4)` and `int dot_acc_sat(char4, char4, int)`, https://registry.khronos.org/OpenCL/specs/3.0-unified/html/OpenCL_Ext.html | +| Apple, Metal (MSL 4.1, 4 June 2026) | none. MSL has no dp4a or packed byte dot product: the built-in `dot(T x, T y)` is a geometric function on floating-point vectors (section 6.9); the integer functions of section 6.4 have no dot form. A per-lane dot4 is scalar emulation (section 4 below measures it) | `simdgroup_matrix` exists for T = half, bfloat (Metal 3.1 and later) and float only (section 2.4: "T is half, bfloat ... or float"); no integer SIMD-group matrix. BUT Metal 4's tensor operation `mpp::tensor_ops::matmul2d` (section 7.2.1, table 7.3, "MatMul2D data type supported") lists A `char` x B `char` into C `int` (Metal 4) and `uchar` x `uchar` into `int` (Metal 4 and OS 26.4), plus `char` x `int4b_format` into `int`. So Apple has an exact int8 x int8 into int32 matrix path, on tensors (device or threadgroup memory, or a `cooperative_tensor` per SIMD-group or threadgroup), not on registers, and only through Metal 4's tensor API. Which GPU families run it in hardware (the M5's neural accelerators) against emulation is in the Metal Feature Set Tables, which the spec defers to and which were not read tonight | Metal Shading Language Specification version 4.1, https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf , sections 2.4, 6.9, 7.2.1 table 7.3 (PDF downloaded and text-extracted 5 October 2026) | + +The Apple finding, stated plainly: the brief's expectation ("Apple has no int8 matrix or dot path") is half right. There +is no per-lane dot4 and no integer `simdgroup_matrix`. There is an exact `char x char -> int` matmul2d in Metal 4 +(table 7.3). Two things about it are unverified tonight and matter for conformance: whether the int accumulate wraps or +saturates (the spec text I extracted says nothing either way; a vector at the int32 edge on the M5 settles it), and the +feature-set table (which Apple GPUs run it natively). What is settled: a generator op that is a per-lane dot4 has no +Apple intrinsic and costs scalar emulation; a generator op that is a whole-unit 8x8x16 or 16x16x16 int8 tile has a +native path on all three vendors (mma.sync on sm_75+, WMMA on RDNA 3 and 4, matmul2d on Metal 4), with Apple's path +living in a different API shape (tensors, not register fragments). + +## 2. The family's semantics as a generator op (integer only, bit-exact) + +Two forms are proposed; the reserve can hold both as separate families or one. + +### 2.1 `dot4`: per-lane + +``` +dot4 dst = dst + dot4_u8(src, src2) + where dot4_u8(a, b) = sum over i in 0..3 of byte_i(a) * byte_i(b), bytes zero-extended, sum modulo 2^32 +``` + +- Bytes are UNSIGNED. Reason, measured below: on Apple the unsigned emulation costs 1.6 ALU-chain steps per dot4 + and the signed one 4.7 (section 4), while NVIDIA (`dp4a.u32.u32`) and AMD (`V_DOT4_U32_U8`, `udot4`, `dot7-insts`) + carry the unsigned form natively as they carry the signed one. Signed bytes buy nothing for the hash (the input is a + pseudo-random register) and cost the vendor without the intrinsic 3x more. +- Accumulation wraps modulo 2^32 like every other op in section 1 (spec 1.14 item 5). The maximum dot of four unsigned + bytes is 4 x 255 x 255 = 260,100, so no single dot4 overflows; the wrap is in the running sum, which is why the + AMD `clamp` bit and the OpenCL `dot_acc_sat` form are NOT the primitive (saturation would change results). +- Operands: `dst`, `src`, `src2` with `src != dst` as for `mad`; `src2` may equal either. +- Verifier: one closed-form integer expression per lane; the register-major interpreter of 1.11 adds four byte + multiplies and adds per lane. The CPU reference in the probes (`dot4_ref` in `proto-opencl/dot4-probe.c`) is this + expression. + +### 2.2 `mm8`: the 32-lane unit as one int8 tile + +The 32 lanes of a unit (spec 1.9) hold, in `src`, a 4-byte row fragment of an 8 x 16 int8 matrix A and, in `src2`, +a 4-byte column fragment of a 16 x 8 int8 matrix B, in exactly the layout of PTX `mma.m8n8k16` with `.u8` operands +(PTX ISA 9.4 section 9.7.16.5, "Matrix Fragments for mma.m8n8k16", the integer-type layout): + +``` +lane l (0..31): A[row = l >> 2][k = 4 * (l & 3) .. 4 * (l & 3) + 3] = the 4 bytes of src (byte 0 = lowest k) + B[k = 4 * (l & 3) .. +3][col = l >> 2] = the 4 bytes of src2 +result C[r][c] = sum over k in 0..15 of A[r][k] * B[k][c] (uint8 x uint8, 16 products, exact, at most 1,040,400) +mm8 dst = dst + C[l >> 2][2 * (l & 3) + bit] bit = an immediate 0 or 1 drawn by the generator +``` + +Every lane receives one of the two C elements its lane position owns in the PTX fragment (`c0` for bit 0, `c1` for +bit 1), added into `dst` modulo 2^32. The whole op is a function of the unit's `src` and `src2` across all 32 lanes, +like `shfl`, so it needs the unit to be exactly 32 logical lanes (the wave64 rule of 1.9 applies: a wave64 device +holds two units and the local-memory path is used). + +How each vendor runs it: + +| Vendor | Native form | Cost per `mm8` (approximate until measured) | +|---|---|---| +| NVIDIA sm_75+ | one `mma.sync.aligned.m8n8k16.row.col.s32.u8.u8.s32` per warp, A and B fragments straight from `src` and `src2`, C = 0 in, `c0`/`c1` out, one add | one tensor instruction plus one add | +| AMD RDNA 3 and 4 | one `V_WMMA_I32_16X16X16_IU8` per wave32 with the 8x16 and 16x8 tiles zero-padded into 16x16 (the WMMA fragment layout differs from PTX's: a fixed permutation of bytes between lanes, which is a few `ds_bpermute` or `v_perm` operations, bit-exact) | one WMMA plus the permutation and the pad | +| AMD CDNA | `V_MFMA_I32_16X16X32_I8` with padding | as above | +| Apple, Metal 4 | `matmul2d` on `uchar` A and B into an `int` cooperative tensor (table 7.3 row "uchar, uchar, int", OS 26.4), the fragments written from registers into a threadgroup tensor first (32 lanes x 8 bytes = 256 bytes), the C element read back per lane | one tensor op plus two threadgroup round trips; on Apple GPUs without the neural accelerators the runtime's emulation, unmeasured | +| Any vendor, fallback | 16 scalar byte products per lane after gathering the 16 bytes of B's column from the 4 lanes that hold them (4 shuffles or one 64-byte threadgroup exchange) | 4 shuffles plus 4 `dot4` emulations: on Apple about 4 x 1.6 = 6.4 ALU steps plus the shuffles (approximate, from the probe) | + +Verifier: the unit evaluates C as 8 x 8 x 16 = 1,024 unsigned byte products once per `mm8` instruction and hands +each lane its element. That is 1,024 multiply-adds per instruction per unit, against 64 x 8 = 512 instructions per +hash: at W_new = 4 (section 3) a program carries about 2.6 `mm8` per iteration, 21 per hash, 21,500 multiply-adds per +unit per hash, under 10 microseconds on one core (approximate), far inside the 0.63 ms the verifier already spends per +unit (spec 1.11). The simulation stays exact because every product and sum is an integer with a defined wrap. + +### 2.3 Which form to reserve + +`mm8` is the one that takes matrix hardware at GPU scale from a chip (the plan's layer 7 row): a chip without tensor +units pays 1,024 products per unit per instruction where a GPU pays one tensor instruction. `dot4` is a per-lane ALU op +that a chip matches with four 8-bit multipliers, which is cheap silicon; it adds little chip resistance and costs Apple +emulation. Recommendation: reserve `mm8`; keep `dot4` out, or in only as `dot4_u8` behind `mm8`. + +## 3. The genesis reserve entry (spec text for 1.13.2) + +Proposed wording, to go under 1.13.2 as the first named reserve family once the conformance runs of section 5 pass: + +> Reserve family R1, `mm8` (integer matrix). Semantics: section 2.2 of `docs/analysis/int8-matrix-family.md`, +> uint8 operands from `src` and `src2` in the m8n8k16 fragment layout, one int32 element of C per lane selected by +> the immediate `bit`, added into `dst` modulo 2^32. Weight at unlock `W_new = 4` points, taken proportionally from the +> ten live non-load families (the load weight and count are untouched, 1.13.1). Edge vectors, each a hand-built unit +> run on every vendor: all bytes 0xFF in A and B (C = 16 x 65,025 = 1,040,400 everywhere); all bytes 0x80 (C = 16 x +> 16,384 = 262,144); A all zero (C = 0); `dst` = 0xFFFFFFFF with a nonzero C (the wrap); alternating 0x00 and 0xFF by +> lane (the fragment mapping: C[r][c] nonzero only where the row and column bytes meet); `bit` = 0 and 1 on the same +> fragments. Unlock: at the start of era n = 4 (two years after genesis, DAA 62,208,000), or earlier by the 90% +> signalling path of section 5.7; never by a release. + +The era-4 choice is deliberate: two years is long enough for the three vendors' tensor paths (and Apple's Metal 4 +feature-set coverage) to be in every miner's driver, and short enough to land before any chip built against the +launch instruction set has paid back (approximate; a chip programme is 12 to 24 months, approximate, from memory). + +Reserve rule for a vendor that can only emulate. Spec 1.13.2 as written requires conformance on every vendor; it says +nothing about cost. Proposed addition: + +> A family enters the reserve when it is bit-exact on every vendor of 1.15. A vendor that reaches the result only by +> emulation (no instruction or library path) does not block entry if the measured penalty of the emulation on that +> vendor, on the family's own probe (a dependent chain of the op, G ops/s against the same vendor's integer ALU chain), +> is at most 8x per op, AND the family's weight at unlock keeps the emulating vendor's hash-rate loss under 5% on the +> memory-hard hash (the hash is latency-bound, so a per-op penalty on 4% of the instructions is a small fraction of a +> hash whose time is 128 dependent DRAM reads; the 5% is checked on the vendor's card with the family live, not +> computed). A family whose emulation exceeds either bound stays out of the reserve until the vendor ships a path. + +With tonight's numbers: on Apple the unsigned `dot4` emulation is 1.6x per op (inside the bound); the signed one 4.7x +(inside, but why pay it); `mm8` through Metal 4's matmul2d is a path, not an emulation, and its cost is owed. + +## 4. dp4a-class throughput, measured so far + +Probe: a dependent chain of one dot4 per step per lane (`acc = dot4(x, y, acc); x = x * K + acc; y = rotl(y, 7) ^ +(acc + s)`), 1,048,576 lanes x 4,096 steps, best of 3, device time, bit-exact against a CPU reference on two lanes +per run, beside the ALU chain of the 9070 XT bench-log entry (`x = x * K + rotl(y, 7); y = (y ^ x) + s`, 5 ops per +step counted). Sources: `proto-metal/dot4-probe.swift` (Metal), `proto-opencl/dot4-probe.c` (OpenCL: scalar, the +`cl_khr_integer_dot_product` `dot`, AMD `__builtin_amdgcn_sudot4`, NVIDIA inline PTX `dp4a.s32.s32`), +`proto-cuda/dot4-probe.cu` (CUDA `__dp4a` and the scalar emulation, for a PC with nvcc). Each OpenCL variant is built on +its own and a variant the platform cannot compile prints a "build failed" row. + +| Card, API | Date, command | ALU chain, G steps/s | dot4 signed emulation, G dot4/s | dot4 unsigned emulation, G dot4/s | dot4 intrinsic, G dot4/s | Penalty of the emulation per op (ALU steps per dot4) | ok (bit-exact) | +|---|---|---|---|---|---|---|---| +| Apple M5 Max, Metal | 5 October 2026 20:0x UTC, `with-lock.sh measure ./dot4-probe` (swiftc -O), GPU start-to-end time | 879.8 (4.882 ms) | 188.2 (22.82 ms) | 548.2 (7.834 ms) | none exists | signed 4.7x, unsigned 1.6x | yes, all three kernels | +| Apple M5 Max, Apple OpenCL 1.2 | same, `with-lock.sh measure ./dot4-probe-cl --device 0`, event time | 871.5 (4.928 ms) | 188.4 (22.80 ms) | not in this probe | `cl_khr_integer_dot_product` not listed; the kernel using `dot(char4, char4)` compiled anyway and ran at 846 G/s but MISMATCHED the CPU reference on every lane checked (Apple's `dot` on char4 is not an integer dot; the extension macro must gate it) | signed 4.6x | alu and dot4e yes; dot4_khr NO | +| RTX 5090 (PC 1, ae432dc7), NVIDIA OpenCL 3.0 CUDA, driver 617.14 | 5 October 2026 20:29 UTC, job `run-dot4-20261005` (`relay/playbooks/dot4-probe.ps1`, both mining cards switched off in the app first, restored after; `node tools/jobs.mjs run-dot4-20261005`), event time | 8,753.5 (0.491 ms) | 1,239.1 (3.466 ms) | not in the OpenCL probe | 7,453.6 (0.576 ms) via inline PTX `dp4a.s32.s32` | emulation 7.1x; the intrinsic 1.17x (the chain is one dp4a plus 3 ops against 5 ops), so the emulation costs 6.0x the instruction | yes, all three | +| RX 9070 XT (PC 1, gfx1201, eGPU), AMD OpenCL 2.0 AMD-APP 3683.0 (PAL,LC) | same job, same time | 701.4 (6.124 ms) | 480.8 (8.932 ms) | not in the OpenCL probe | 664.3 (6.465 ms) via `__builtin_amdgcn_sudot4(true, a, true, b, acc, false)`: the Adrenalin OpenCL C compiler accepts the clang builtin and emits `v_dot4_i32_iu8` | emulation 1.46x; the intrinsic 1.06x, so the emulation costs 1.38x the instruction | yes, all three; the older 3652.0 platform entry for the same card gave 696.2 / 501.7 / 683.6 | +| Ryzen 9800X3D gfx1036 (PC 1, integrated RDNA 2, 2 CUs) | same job | 40.6 (105.9 ms) | 15.8 (272.3 ms) | | `sudot4` does not build: "needs target feature dot8-insts" (RDNA 2 has `dot1-insts`' `v_dot4_i32_i8`, the `sdot4` builtin, which the probe did not try) | emulation 2.6x | alu and dot4e yes | +| Every PC device | `cl_khr_integer_dot_product` not listed on NVIDIA (OpenCL 3.0) or AMD (2.0); the pragma draws "unknown OpenCL extension" on both and the `dot(char4, char4)` kernel does not build | | | | | | | + +Reading across the three cards. Per dot4 at the hardware rate: the 5090 does 7.45 T dot4/s (one `dp4a` per step, 0.85 +of its ALU-chain step rate), the 9070 XT 0.66 T (0.95 of its ALU-chain rate), the M5 Max 0.55 T at best (the unsigned +emulation; no instruction). On the ALU chain the 5090 is 12.5x the 9070 XT and 10x the M5 Max; on hardware dot4 it is +11.2x the 9070 XT, so the family does not widen the AMD gap, and 13.6x the M5 Max, so Apple's emulation widens its gap +by 1.4x on this op (approximate: one probe shape, the ratios of best-of-3 numbers). The signed emulation is where the +vendors differ most: 7.1x the ALU step on NVIDIA, 4.7x on Apple, 1.46x on AMD (AMD's compiler and byte-permute +hardware make the four sign-extended products nearly free; the NVIDIA OpenCL compiler does not pattern-match the +emulation into `dp4a`, which the 6x gap between `dot4e` and `dot4_nv` shows). None of this is a hash-rate number: the +hash is bound by 128 dependent DRAM reads, and a family at W_new = 4 adds about 21 of these ops per hash per lane +against about 1.2 microseconds of memory latency per hash per lane (approximate), so the per-op penalties above turn +into hash-rate losses well under 5% on every card, to be measured with the family live. + +Reading of the Mac numbers. The ALU chain's 880 G steps/s on the M5 Max is the integer baseline (5 ops per step +counted, so about 4.4 T int ops/s, approximate; the 5090's 8,754 G steps/s is about 43.8 T, against the whitepaper's +104.8 peak INT32 TOPS which counts a multiply-add as two). A signed dot4 emulated as `int4(as_type(a))` products costs +4.7 of those steps; the unsigned form 1.6 steps. The 3x gap between the two is the sign extension (Metal lowers the +unsigned byte extraction to masks that fold into the multiplies, approximate reading of the result, not of the +compiled code). Both are far under the 8x bound of section 3, and the hash spends its time on DRAM reads, so a per-lane +`dot4` family would cost Apple a few percent at W_new = 4 (to be measured with the family live, not computed). The +Apple OpenCL `dot(char4, char4)` mismatch is the kind of thing the edge vectors of section 3 exist to catch. + +## 5. What is owed or unverified + +| Item | State | +|---|---| +| dp4a throughput on the RTX 5090 through NVIDIA OpenCL inline PTX | measured (section 4); the CUDA `__dp4a` form (`proto-cuda/dot4-probe.cu`) is unrun (no nvcc job tonight) and is a cross-check, not a gap | +| `sudot4` on the 9070 XT through Adrenalin's OpenCL C; `cl_khr_integer_dot_product` on the 3683.0 platform | measured: the builtin works and emits the instruction; the extension is not listed and the `dot(char4, char4)` kernel does not build | +| `sdot4` (`dot1-insts`) on RDNA 2 (gfx1036) | not tried; the probe only carries `sudot4` | +| Metal 4 `matmul2d` uchar x uchar into int on the M5 Max: wrap or saturate at the int32 edge, native or emulated, throughput | owed (a second Metal probe; the API needs a tensor set-up the dot4 probe does not have) | +| Metal Feature Set Tables: which Apple GPU families run int8 matmul2d natively | not read tonight | +| RDNA 3 and RDNA 4 ISA guides: the instruction text itself (names taken from LLVM and GPUOpen) | AMD's CDN refused the downloads tonight | +| `mm8` on AMD: the exact byte permutation between the PTX m8n8k16 fragment layout and the RDNA WMMA 16x16x16 layout | design, to be written with the kernel | +| The hash-rate cost of the family live at W_new = 4 on each vendor (the 5% rule of section 3) | owed, needs the generator change (not tonight) | +| Edge vectors of section 3 as files | owed, with the generator change | diff --git a/docs/analysis/prover-floor.md b/docs/analysis/prover-floor.md new file mode 100644 index 000000000..dc6be3da3 --- /dev/null +++ b/docs/analysis/prover-floor.md @@ -0,0 +1,152 @@ +# The prover floor: why a 12 GB card cannot prove on SP1 6.8.1's GPU server, and the patch + +5 October 2026, 22:00 UTC on (Josh: "execute if it will solve the issue"). Branch `prover-floor` +(worktree `igneum-wt-prover-floor`). The measured facts this starts from: `docs/plans/proving-v1.md` and the +bench-log entry "proving v1" (branch proving-v1): the GPU server holds 13.9 GB for an empty shard, 20.4 GB at the +adopted v1 shard, 28.3 GB flat from 20 M to 60 M cycles, and no environment knob moved the floor. Every figure +below is from the source at tag v6.8.1 (cloned to `vendor/sp1-6.8.1`, gitignored; the fork is the patch +`proving/prover-floor/sp1-gpu-6.8.1-floor.patch`) or from a PC 2 run named in `docs/bench-log.md` +("prover floor"). Sizes in GiB are computed from the source constants (4-byte field elements); sizes in MiB are +measured by `nvidia-smi` at 1 s. + +## Where the server is built and what it reads + +The SDK downloads `sp1_gpu_server_v6.8.1_x86_64.tar.gz` (133,750,780 bytes) from the SP1 release and runs it from +`$HOME/.sp1/bin/sp1-gpu-server` (`crates/cuda/src/server.rs` 19 to 30, 80 to 99). The source is in the same +repository: `sp1-gpu/crates/server` (the binary), built by `.github/workflows/release.yml` 234 to 314 on CUDA +12.8.1 with Go and protoc (`cargo build --release --bin sp1-gpu-server`). The binary takes no options +(`sp1-gpu/crates/server/src/main.rs` 15 to 18: `--version` only) and reads `CUDA_VISIBLE_DEVICES` (32 to 35); +everything else comes from the environment the host process passes it, through `SP1CoreOpts::default()` +(`crates/core/executor/src/opts.rs` 99 to 140: `SHARD_SIZE`, `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, +`MINIMAL_TRACE_CHUNK_THRESHOLD`, `TRACE_CHUNK_SLOTS`, `FULL_SIZE_SHARDS`) and the worker counts +(`crates/prover/src/worker/config.rs`). + +## The memory model, term by term + +Every device buffer is sized at construction from constants, not from the shard. The server builds the prover at +the first `Setup` request (`sp1-gpu/crates/server/src/server.rs` 126 to 137) through +`cuda_worker_builder_with_machine` (`sp1-gpu/crates/prover_components/src/builder.rs` 102 to 148): + +| Term | Where | Size | On the device | Moves with the shard | +|---|---|---|---|---| +| The gate | `builder.rs` 35 to 39: `gpu_memory_gb = ceil(total / GiB) + 4`; `panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB")` when under 24 | a 12 GB card reads 16, a 16 GB card 20: both refused before any allocation | | no | +| The core element threshold | `builder.rs` 41 to 48: `ELEMENT_THRESHOLD` = 2^28 + 2^27 = 402,653,184 elements (`opts.rs` 12) on a card reading over 30 (a 32 GB card reads 36); minus 2^26 + 2^25 + 2^24 = 285,212,672 on a card reading 24 to 30 (a 24 GB card). The environment's `ELEMENT_THRESHOLD` is read at `opts.rs` 129 and then OVERWRITTEN at `builder.rs` 48, which is why the sweep's `ELEMENT_THRESHOLD` rows changed nothing; `HEIGHT_THRESHOLD` survives (it is not overwritten), which is why the 2^25 + 2^20 row did | sets the next two terms | | no | +| The core trace area, one per shard in flight | `builder.rs` 70 to 71: `num_elts = element_threshold + 2^21` (`CORE_LOG_STACKING_HEIGHT` 21, `crates/prover/src/components.rs` 16) = 404,750,336; allocated on the device at `sp1-gpu/crates/jagged_tracegen/src/lib.rs` 484 to 500 (`allocate_and_initialize_traces`: `max_trace_size` felts + `max_trace_size / 2` u32 column index + 2^14 u32) | 6 bytes an element: **2.26 GiB** for the full threshold, 1.59 GiB for the 24 GB threshold | yes, in full, whatever the shard holds | no | +| The program's preprocessed traces (the proving key) | `sp1-gpu/crates/shard_prover/src/setup.rs` 47 to 58 and 107: the same `allocate_and_initialize_traces(max_trace_size)` at `Setup`, kept in the key cache for the connection's life (`server.rs` 139 to 143) | another **2.26 GiB**, held from `Setup` on | yes | no | +| The pinned host trace buffers | `sp1-gpu/crates/prover_components/src/components.rs` 99 to 103: 4 `PinnedBuffer` of `max_trace_size` felts per prover (core 4 x 1.51 GiB, recursion 4 x 0.5 GiB, shrink 4 x 0.125 GiB, wrap 4 x 0.32 GiB) | 9.8 GiB of pinned host RAM, not device memory (the WSL2 working set the bench saw) | no | no | +| The recursion trace area | `builder.rs` 15 and 95: `RECURSION_TRACE_ALLOCATION` = 2^27 elements, one per recursion tracegen (the recursion program's key at setup and its shard at prove) | 0.75 GiB each | yes | no | +| The shrink and wrap provers | `builder.rs` 16, 19, 117 to 127: 2^25 and 85,376,340 elements, built at `Setup` for every proof mode, used only by the Groth16 and PLONK path | host pinned at build; device only when a wrap runs (never, for a compressed proof) | no | no | +| The codewords (LDE) and the Merkle trees | `sp1-gpu/crates/basefold/src/fri.rs` 92 to 97: every stacked column of 2^21 rows encoded to 2^(21 + 1) rows (`log_blowup` 1); kept until the query phase unless `drop_ldes` (`builder.rs` 52: only on a 24 GB card with `FULL_SIZE_SHARDS`); the preprocessed codewords live in the key | 2 x the padded trace, so up to 2 x the term above | yes | yes, with the padded trace | +| The LogUp GKR layers | `builder.rs` 51: `recompute_gkr_trace = false`, so the first layer stays materialised (`sp1-gpu/crates/logup_gkr/src/tracegen.rs` 169 to 224) | of the order of the interaction count | yes | yes | +| The allocator | `sp1-gpu/crates/cuda/src/task.rs` 152 and 196: the device's default `cudaMallocAsync` pool with its release threshold at `u64::MAX`, so nothing freed is ever returned to the driver: `nvidia-smi` reads the high-water mark of everything live at once | | | | + +So at zero cycles the server already holds the proving key's 2.26 GiB, the shard's 2.26 GiB (both allocated at the +threshold, not at the shard's rows), their codewords and trees, and the recursion program's key and traces (2 x +0.75 GiB and their codewords): the 13.9 GB floor. The shard's own content only adds to the codewords, the GKR +layers and the working buffers, which is the 13.9 to 20.4 GB step from 0.3 M to 4.7 M cycles, and the flat 28.3 GB +from 20 M cycles is the threshold's padded area reached. The witness (5 to 22 KB) never appears. + +## What the patch does (`proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, three files) + +1. `builder.rs`: the panic is gone; the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks the element threshold + from a tier table (`element_threshold_for_budget`: over 30 as read, the full 402.6 M; 24 to 30, upstream's 24 GB + figure; 18 to 24 (a 16 GB card), 2^27 + 2^26 = 201.3 M; under 18 (a 12 GB card), 2^27 = 134.2 M); + `SP1_GPU_ELEMENT_THRESHOLD` sets it directly and `SP1_GPU_RECURSION_TRACE_ALLOCATION` the recursion buffer. + The chosen numbers are printed as a `FLOOR opts` line. Every other option is as upstream. +2. `jagged_tracegen/src/lib.rs`: with `SP1_GPU_FLOOR_LOG` set, every trace allocation prints its capacity and, + after the shard's traces are in, the elements actually used and the device memory in use. +3. `server.rs`: a `FLOOR memory` line (device used, free, total) after `Setup` and after every proof, with the + proof's time. + +Nothing in the proof changes: the element threshold only moves where the executor splits shards, exactly what +upstream's own 24 GB tier does with the same verifier and the same keys; the recursion program, the verifying key +and the pinned guest ids are untouched. The unpatched verifier (the pv1 host's SDK) is the one that verifies every +measured proof below. + +## The build (PC 2, WSL2 Ubuntu-24.04, job `floor-toolchain-1` then the build job) + +Toolchain found 22:10Z (job `floor-toolchain-1`, 4 s): nvcc 12.8 at `/usr/local/cuda-12.8`, cmake 3.28.3, gcc 13.3, +clang 18, protoc 3.21.12, cargo 1.99.0, no Go. The release workflow installs Go for the server's `native-gnark` +feature (the Groth16 and PLONK wrap through gnark), which a compressed proof never runs, so the build drops that +feature from `sp1-gpu/crates/server/Cargo.toml` and nothing else. `CUDA_ARCHS=86,89,120` (consequences reviewer +C26): the 12 GB tier is sm_86 (RTX 3060) and sm_89 (RTX 4070), the 16 GB tier sm_89 and sm_120 (RTX 5080), PC 2's +5090 is sm_120; the stock server lists sm_80, 86, 89, 90, 100 and 120, which a shipped build repeats. The recipe: +`tools/prover-floor/pc2-build-server.ps1` (generated by `make-build-playbook.sh` from the patch, so the two cannot +drift): clone the tag, `git apply` the patch, `touch` the three files, `cargo build --release --bin sp1-gpu-server` +niced with 8 jobs into `/opt/igneum-floor/target`, the binary copied to `/opt/igneum-floor/home/.sp1/bin/` (the SDK +spawns the server it finds under `$HOME/.sp1/bin`, so `HOME=/opt/igneum-floor/home` selects it and the live +`/root/.sp1/bin/sp1-gpu-server` stays as it is). + +What a measurement on PC 2 can and cannot say (C26). The server's allocation pattern is deterministic in the +budget it is given, so a run with `SP1_GPU_MEMORY_BUDGET_GB=12` on the 5090 shows the peak a 12 GB card's build +would ask for; it does not show that a 3060 proves it in time, nor what the card's display and driver hold. The +public line keeps "24 GB" until the on-order 12 GB card runs the same fixture. Every row names the arch list and +the card. + +What shipping it costs (C26). A patched server means the project signs and distributes its own build of SP1's +prover: the WSL2 package, the DMG's prover inputs, the K1-signed inputs and `evidence.md` carry it, and every SP1 +upgrade repeats the clone, patch, build and measurement. The verifying key and the pinned guest ids do not move +(the patch changes buffer sizes and the shard split, not the circuits), which the `verify-segment` and `--mode +compressed` VERIFIED lines of the unpatched host show on every row below. The packaging path is a row for the +proving plan before 0.3.12, not this branch. + +## Step 4 contingency, read not measured: RISC Zero's CUDA prover and its memory per segment + +If SP1 could not be brought under 11 GB, the alternative's floor is read from its operators' documentation (not +measured here; a PC 2 run would be the measurement): Boundless' prover guide +(https://docs.boundless.network/provers/performance-optimization) sets the segment size cap by VRAM as 8 GB: +po2 19, 16 GB: po2 20, 20 GB: po2 21, 40 GB: po2 22, with measured peaks po2 20: 13,835 MiB, po2 21: 22,905 MiB, +po2 22: 41,089 MiB; RISC Zero's PR 3761 adds `low_vram` and `pinned_witgen` to fit po2 22 on a 24 GB 4090. So +RISC Zero proves a 2^19-cycle segment inside 8 GB and a 2^20 one inside 16 GB, and a shard of 4.7 M cycles is +9 segments at po2 19 plus lift and join steps (times not on the page). Adopting it would cost a second guest (the +chain rule in the RISC Zero zkVM), a second pinned program id, a second verifier in the node and no shared +aggregation between the two formats: `docs/analysis/amd-proving.md` and the proving plan carry that row already. + +### The build, as it ran (job `floor-build-3`, 22:28:24 to 22:32:29Z) + +Three runs: `floor-build-1` (22:17Z) and `floor-build-2` (22:24Z) failed in 2 to 4 minutes on +`crates/recursion/gnark-ffi/build.rs:70`, "Failed to build Go library: NotFound" (no `go` on PC 2; the first run's +playbook lost its own log, a bug fixed before the second). `floor-build-3` fetched go1.27.1 (tarball sha256 +`63d339f0da5ab53635a56f2490a7984dfe12dfcff22ad749f63edaf590168445`, checked before unpacking under +`/opt/igneum-floor/go`, on the job's PATH only) and built in **240 s** (46 crates on the warm target of run 2, 8 +niced jobs, 16 cores). The binary: `/opt/igneum-floor/bin/sp1-gpu-server`, **166,768,224 bytes, sha256 +`5568108bf7fb9b0e525d8a08926b7046e51136ffaea53f0ca858631d0e938878`**, `--version` 6.8.1, `cuobjdump --list-elf` +sm_86, sm_89, sm_120 (the stock 251,306,680-byte server lists sm_80, 86, 89, 90, 100, 120 and compute_120 PTX). +The live `/root/.sp1/bin/sp1-gpu-server` (c2642ad1...) was never touched; the miners mined throughout. + +## Sweep 1 (job `floor-sweep-1`, 22:34:56 to 22:37:55Z): the shard term gone, a second floor found + +PC 2's RTX 5090 (32,607 MiB, idle 1,755 MiB with the miners stopped and the live prover off), the patched server +`5568108b...` (sm_86, sm_89, sm_120; the build above), the unpatched pv1 host `dae6b006...` as client and +verifier, one `--mode compressed --shard 0` per point, every server killed and its socket unlinked around every +point, peak = `nvidia-smi memory.used` at 1 s (the idle 1,755 MiB inside it), time = the compressed proof. +Every proof VERIFIED (1,272,897 bytes, verify 0.037 to 0.040 s), so the unpatched verifier accepts every proof +of the patched server. + +| Config (environment to the patched server) | Fixture | Cycles | Peak MiB | Prove s | Verified | +|---|---|---|---|---|---| +| control: `SP1_GPU_MEMORY_BUDGET_GB=32` (upstream's sizes) | empty live shard (block 83616) | 280,706 | 13,892 | 2.2 | yes | +| control | v1 shard (fees-v1-shards2 shard 0) | 4,717,439 | 20,516 | 4.2 | yes | +| 12 GB tier: budget 12 (threshold 2^27) | empty | 280,706 | 12,740 | 2.4 | yes | +| 12 GB tier | v1 shard | 4.7 M | 15,396 | 4.1 | yes | +| 12 GB tier + `SP1_WORKER_NORMALIZE_PROGRAM_CACHE_SIZE=1` | v1 shard | 4.7 M | 15,428 | 4.0 | yes | +| 16 GB tier: budget 16 (2^27 + 2^26) | v1 shard | 4.7 M | 18,628 | 3.8 | yes | +| `SP1_GPU_ELEMENT_THRESHOLD=67108864` (2^26) | empty | 280,706 | 12,772 | 3.1 | yes | +| 2^26 | v1 shard (split into 4 core shards) | 4.7 M | **12,708** | 5.3 | yes | +| `SP1_GPU_ELEMENT_THRESHOLD=33554432` (2^25) | v1 shard | 4.7 M | 12,836 | 8.5 | yes | + +Reading. The control reproduces the proving agent's curve (13.9 and 20.4 GB), so the patched server behaves as +the stock one at the stock sizes. The shard's term follows the threshold as the model says (20.5 GB at 402 M +elements, 15.4 at 134 M, 12.7 at 67 M), and then stops: 2^26 and 2^25 both sit at 12.7 to 12.8 GB for the empty +shard and the v1 shard alike. The `FLOOR memory after setup` line names the rest: **9,703 MiB in use before the +first shard** (2^26; 11,623 at the stock sizes), and the `FLOOR tracegen alloc` lines at Setup are five +allocations of 134,217,728 elements (the recursion keys, 0.75 GB each, each using 90,177,536 elements: 35.6 M +preprocessed and 54.5 M main at prove time), one of 33,554,432 (the shrink key, 0.19 GB) and one core key at the +threshold. The `NORMALIZE_PROGRAM_CACHE_SIZE` knob does not reach them (they are keys built at `Setup`, not the +program LRU). The time cost of the split: the v1 shard at 2^26 is 4 core shards and 5.3 s against 4.2 s (1.26x); +at 2^27 it is 4.1 s with no split. + +So after sweep 1 the binding term is the Setup-time keys allocated at full capacity, and patch v2 sizes every +trace buffer (keys and shards) to its padded need: `padded_trace_elements` in `jagged_tracegen/src/lib.rs` +(each phase pads to the next multiple of 2^21 rows, `generate_jagged_traces`'s "final padding"), applied in +`setup_tracegen` and `full_tracegen`, one stacking height of slack, `SP1_GPU_FLOOR_EXACT=0` restoring upstream. diff --git a/docs/analysis/proving-methods.md b/docs/analysis/proving-methods.md new file mode 100644 index 000000000..6cd843a99 --- /dev/null +++ b/docs/analysis/proving-methods.md @@ -0,0 +1,411 @@ +# Proving methods: why the prover needs 14 GB, what else exists, and how a 12 GB card gets to prove + +5 October 2026, from Josh at 22:05 UTC: "if this doesn't enable 12 GB cards, then do a full deep research task on proving +and see if there are different methods." "This" is the prover-floor agent's patch of SP1's GPU server (branch +`prover-floor`), running tonight. This document is research and reading, not measurement: every number of ours is from +`docs/bench-log.md` with its entry named; every claim about another system cites its repository file, its documentation +page or its paper, or is labelled approximate. Status words follow `docs/spec/00-overview.md` 0.2. Day estimates follow +Josh's rule of 3 October 2026: hours of agent time, never weeks. + +The facts this starts from (bench-log, "proving v1", 5 October 2026; `docs/plans/proving-v1.md`; `docs/analysis/amd-proving.md`): + +| Fact | Number | +|---|---| +| SP1 6.8.1's GPU server, an empty shard (280,706 cycles), the card to itself | 13,874 MiB peak, 2.2 s compressed | +| The adopted v1 shard (`S_p` 30,000 pgas, 4,717,439 cycles) | 20,434 MiB, 4.3 s; 22,210 MiB and 13.2 s beside the miner | +| The prototype shard (6.75 M pgas, 60.4 M cycles) | 28,307 MiB, 10.8 s; flat at 28.3 GB from 20 M cycles up | +| Aggregation, chained, per block, on a mining 5090 | 9.6 to 9.7 s; 2.2 to 2.5 s with the card to itself | +| The CPU path (PC 1, 16 cores) | 282 s a shard whatever its size, 29.5 to 30.5 GB RSS | +| AMD and Apple GPUs | no zkVM proves on AMD; RISC Zero has a Metal prover, SP1 does not | +| The promise | `site/litepaper.html`: "Target: shard size will be set so a 12 GB card proves one shard in about 20 seconds"; the design goal is every block proven within about a minute by the miners' own cards | + +## 1. The memory anatomy of a STARK-based zkVM prover, and why the floor is where it is + +### 1.1 What SP1 6.8.1 is + +SP1 6.x is not the univariate FRI STARK of the earlier SP1 releases (Succinct calls Hypercube the "first zkVM built entirely on a multilinear polynomial-based proof system", blog.succinct.xyz, sp1-hypercube, 20 May 2025). The crates it pulls say what it is: `slop-multilinear`, `slop-sumcheck`, +`slop-jagged`, `slop-stacked`, `slop-basefold`, `slop-whir` (the `~/.cargo/registry` of this Mac; `proving/igneum-prove/Cargo.toml` +pins `sp1-sdk = "=6.8.1"`). The architecture Succinct calls Hypercube: the execution trace is a set of multilinear +polynomials over the 31-bit KoalaBear field (`sp1-hypercube-6.8.1/src/verifier/config.rs`: `SP1BasefoldConfig = +Poseidon2KoalaBear16BasefoldConfig`), the constraints are checked by a zerocheck sumcheck and the lookups by a LogUp GKR +(`sp1-gpu/crates/zerocheck`, `sp1-gpu/crates/logup_gkr`), and the polynomial commitment is "jagged": every table's +columns, whatever their heights, are concatenated into one long vector, stacked into rows of height `2^log_stacking_height` +and committed with BaseFold, a FRI-like folding over a Reed-Solomon code (`sp1-gpu/crates/basefold/src/fri.rs`, +`slop-basefold-6.8.1/src/verifier.rs`). The proof system parameters, from `sp1-primitives-6.8.1/src/fri_params.rs` and +`sp1-prover-6.8.1/src/components.rs`: + +| Parameter | Value | Where | +|---|---|---| +| Field | KoalaBear, 31 bits, 4 bytes an element; extension degree 4 (16 bytes) | `sp1-primitives` | +| Core stage: Reed-Solomon blowup | `CORE_LOG_BLOWUP = 2`, so the codeword is 4x the data | `fri_params.rs:5` | +| Core stage: stacking height, maximum rows per table | `CORE_LOG_STACKING_HEIGHT = 21`, `CORE_MAX_LOG_ROW_COUNT = 22` | `components.rs:16,17` | +| Core shard limits (the executor's cut) | `MAX_SHARD_SIZE = 2^24` cycles, `ELEMENT_THRESHOLD = 2^28 + 2^27 = 402,653,184` trace elements, `HEIGHT_THRESHOLD = 2^22` rows | `sp1-core-executor-6.8.1/src/opts.rs:9-12` | +| Recursion (compress) stage | blowup 2 (`RECURSION_LOG_BLOWUP = 2`), stacking height 20, max rows 2^21 | `fri_params.rs:6`, `sp1-verifier-6.8.1/src/compressed/config.rs:1,2` | +| Shrink and wrap stages | blowup 3 (8x), 22 bits of grinding, stacking 18 and 21 | `fri_params.rs:17,18,7`, `components.rs:37-40` | +| Recursion arity | 4 proofs per compose step (`DEFAULT_MAX_COMPOSE_ARITY = 4`, `DEFAULT_MAX_REDUCE_ARITY = 4`) | `sp1-prover-6.8.1/src/worker/config.rs:183,193` | +| Workers | 4 core workers, 8 recursion prover workers, 4 recursion executors, 4 deferred workers, buffers of 4 to 8 | `worker/config.rs:188-205` | + +The stages a shard goes through (`sp1-prover-6.8.1/src/worker/controller/*.rs`): execute (the RISC-V executor cuts the +run into core shards at the thresholds above); core (one jagged-PCS proof per core shard, on the GPU); normalize and +compose (each core proof is verified inside a recursion program, then proofs are folded 4 at a time until one remains, +the "compressed" proof, 1,272,897 bytes for every shard we have proven, bench-log 4 and 5 October); deferred (what the +aggregator uses: `verify_sp1_proof` inside a guest, `proving/igneum-prove/aggregator/src/main.rs`); shrink and wrap +(to a BN254 STARK, then Groth16 or Plonk; not run here, ledger P3). + +### 1.2 The terms, and which scale with the shard + +Every STARK-family prover holds these buffers on the device at its peak, in some order and with some overlap. The +sizes below are from the constants of 1.1 and the allocation code of `sp1-gpu`; where a buffer's size is the actual +trace rather than the maximum, the row says so. + +| Term | What it is | Size rule | SP1 6.8.1 on a 32 GB card | Scales with the shard? | +|---|---|---|---|---| +| Main trace | the witness: one element per cell of every table the shard touched | `cells x 4 bytes`, where cells = sum over tables of rows x columns; the executor cuts a new core shard at 402,653,184 cells | the device buffer is allocated at the MAXIMUM, not the actual trace: `allocate_and_initialize_traces` takes `max_trace_size` and allocates `max_trace_size` felts plus `max_trace_size / 2` u32 of index (`sp1-gpu/crates/jagged_tracegen/src/lib.rs:484-503`), 6 bytes a cell; the core prover's `max_trace_size` is `element_threshold + 2^21` (`prover_components/src/builder.rs:70-71`): **2.26 GiB** on a card over 30 GB, 1.61 GiB on a 24 GB card (the threshold drops by 2^26 + 2^25 + 2^24 when memory is 30 GB or under, `builder.rs:41-45`) | no: fixed at the maximum shard, whatever the trace | +| Preprocessed trace | the program's own tables (the ELF as a `Program` AIR, the byte and range tables) | program size x its columns plus 2 x 2^16-class tables | small for a 2.8 MB guest ELF (`elf/manifest.json`); not isolated | with the guest, not the shard | +| Codeword (the LDE) | the stacked polynomial encoded at rate 1/4 for BaseFold | `stacked cells x 4 (blowup) x 4 bytes`; the stacked length is the actual cell count padded to a multiple of 2^21 | 4.7 M cycles: approximate, the actual trace; 60 M cycles: the shard is 7 to 15 core shards of up to 402 M cells, each encoded to 6 GiB at the blowup, one or more in flight | yes, up to the core-shard cap; past the cap the shard count grows and the per-shard term stays | +| Merkle commitment | Poseidon2 hashes of the codeword rows | `rows x 8 elements x 4 bytes x 2`, rows = 2^21 x blowup | about 0.5 GiB at full stacking, approximate | with the stacked rows | +| Zerocheck and GKR | the constraint sumcheck over the extension field, and the LogUp GKR layers | extension elements are 16 bytes; the sumcheck holds a folded copy of the trace in the extension field, which is 4x the base trace at the first round and halves each round | up to about 4x the live trace in the first round, approximate (`sp1-gpu/crates/zerocheck/src/primitives.rs:174,287`: `Buffer` of `new_total_length`) | yes | +| Recursion traces | the normalize and compose programs' own traces, verifying core proofs | fixed-shape programs: `RECURSION_TRACE_ALLOCATION = 2^27` cells (`builder.rs:15`), allocated at 6 bytes a cell: **0.75 GiB** per recursion prove, at blowup 4 a 2 GiB codeword plus its own zerocheck | fixed per recursion step; the number of steps is log4 of the core-shard count | no (per step) | +| Shrink and wrap traces | the two last stages, not run by us | 2^25 and 85,376,340 cells (`builder.rs:16,19`): 0.19 and 0.48 GiB | only when wrapping | no | +| Proving keys and program cache | the recursion programs (`vk_map.bin`, the normalize cache of 5 programs) and the shard program's setup | `DEFAULT_NORMALIZE_PROGRAM_CACHE_SIZE = 5` (`worker/config.rs:192`); the key setup took 14.6 s on the 5090 (bench-log 4 October) | not isolated | no | +| Pinned host buffers | the staging copies on the PC side | 4 core workers x `max_trace_size` x 4 bytes = **6.0 GiB** of pinned RAM, plus 4 x 0.5 GiB for recursion (`prover_components/src/components.rs:99-103`, `builder.rs:76,95`) | this is the 7.9 GB WSL2 working set measured on 5 October | no | +| The allocator | CUDA's default memory pool with its release threshold set to `u64::MAX` (`sp1-gpu/crates/cuda/src/task.rs:152,190-199`): freed blocks are never returned to the driver | `nvidia-smi` therefore reports the high-water mark of everything above, and it stays until the server exits | this is why the memory curve is flat between shards of different size | no | + +Two facts from this table explain the measurements: + +1. **The server refuses small cards by code.** `local_gpu_opts()` reads the card's total memory, adds 4 GB, and panics + under 24: `"Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"` (`sp1-gpu/crates/prover_components/src/builder.rs:35-38`). + A 16 GB card (16 + 4 = 20) and a 12 GB card (16) never start; a 20 GB card is the smallest that does. The 13.9 GB + floor measured on the 5090 is therefore not the whole story for a 12 GB card: on this build the card is refused + before any buffer is allocated. Any route through SP1's GPU server starts by removing this line. +2. **The environment knobs do not reach the floor** because the same function overwrites `element_threshold` with the + compile-time constant (`builder.rs:41-48`); only `HEIGHT_THRESHOLD` passes through, which is why the sweep's + `ELEMENT_THRESHOLD 2^26` rows changed nothing and `HEIGHT_THRESHOLD 2^20` took 5.4 GB off the 60 M-cycle shard + (bench-log, "the 12 GB requirement", 5 October 2026). The worker counts only slow the proof (11.4 s to 20.8 s) + because the device buffers are sized by `max_trace_size`, not by the worker count. + +### 1.3 Why the floor is 13.9 GB for an empty shard + +With the release threshold at `u64::MAX`, the peak is the high-water mark over the whole pipeline. For an empty shard +the core trace is small (280,706 cycles), so the fixed-shape terms dominate: the 2.26 GiB main-trace buffer allocated +at the maximum, the recursion step over a fixed-shape normalize program (a 0.75 GiB trace buffer, its 4x codeword in +the extension field for the zerocheck, its Merkle tree), the proving-key and program caches, and the deferred and +compose machinery that a compressed proof always runs once. The decomposition of the 13.9 GB into those terms is +approximate until the prover-floor agent's profile lands (branch `prover-floor`, tonight): the figure that is not +approximate is that none of it is the witness (5 to 22 KB a shard) and none of it is the shard's cycles (the same +13.9 GB at 280 k cycles and 556 k cycles, bench-log "the S_p curve"). + +The step from 13.9 GB (empty) to 20.4 GB (4.7 M cycles) is the live trace: the 4.7 M-cycle shard is one core shard +(its trace area is under 402 M cells, so it was not split; approximate from the memory curve, the cell count is not +logged by the host), and its codeword, zerocheck and GKR buffers are sized by its actual cells. The step from 20.4 GB +to 28.3 GB (20 M cycles and up) is the second and later core shards in flight at once: 4 core workers with a buffer of +4 (`worker/config.rs:188,189`) let several core shards' codewords exist at the same time; past 20 M cycles the pipeline +is full and the peak is flat, which is what the curve shows (28,371 MiB at 20 M cycles, 28,307 at 40 M and 60 M). + +### 1.4 The theoretical floor for our guest at the adopted shard + +If every buffer were sized to the shard rather than to the maximum, the adopted v1 shard (4.7 M cycles) would need, +approximate, from the rules of 1.2: + +| Term | Rule | Approximate bytes | +|---|---|---| +| Main trace, actual | 4.7 M cycles x about 60 cells a cycle (the `Add` and `Addi` tables cost 33 and 30 columns a row, a memory access adds 20 and a global interaction 241: `sp1-core-executor-6.8.1/src/artifacts/rv64im_costs.json`) | about 280 M cells, 1.1 GB | +| Codeword at blowup 4 | 4x | 4.5 GB | +| Zerocheck first round in the extension field | 4x base, halving each round | 4.5 GB at the peak round, falling | +| Merkle tree | rows x 32 bytes x 2 | 0.3 GB | +| Recursion step, fixed | 2^27 cells x 6 bytes plus its 4x codeword and extension copies | 2 to 3 GB, approximate | +| Keys and caches | | under 1 GB, approximate | +| Peak, if the core stage and the recursion stage do not overlap and the pool releases | | **about 10 to 11 GB**; about 6 GB if the blowup-4 codeword is replaced by a rate the sumcheck does not need (see 2.4) | + +So the adopted shard is, on paper, a 12 GB card's shard with the server re-sized and nothing else changed, and it is a +12 GB card's shard with 4 GB to spare if the shard is halved (`S_p` 15,000 pgas, 2.4 M cycles: the planner cuts at +transaction boundaries to any budget, `core/src/plan.rs`, and the fee switch of 5 October already moved `S_p` once). +What the paper figure does not say is the time: a smaller card proves slower, and 60 s with the miner running is the +bound (section 3). The prover-floor agent is measuring the real figure; this section says what it should find and why. + +### 1.5 RISC Zero's anatomy, for comparison + +RISC Zero is the FRI STARK the textbooks describe, and its constants make the same table easy to read +(`~/.cargo/registry`, `risc0-zkp-3.0.4/src/lib.rs`, `risc0-circuit-rv32im-4.0.4/src/zirgen/defs.rs.inc`): + +| Term | Value | Where | +|---|---|---| +| Field | BabyBear, 31 bits; extension degree 4 | `risc0-core` | +| Segment size | `DEFAULT_SEGMENT_LIMIT_PO2 = 20` (1,048,576 cycles), `MIN_CYCLES_PO2 = 13`, `MAX_CYCLES_PO2 = 24`; `DEFAULT_MAX_PO2 = 22` for the verifier | `risc0-circuit-rv32im-4.0.4/src/execute/mod.rs:39`, `risc0-zkp-3.0.4/src/lib.rs:35-38`, `risc0-zkvm-3.0.4/src/receipt.rs:884` | +| Trace width | data 211 + accum 103 + code 1 = 315 columns; globals 90, mix 36 | `defs.rs.inc:7-11` | +| Blowup | `INV_RATE = 4`; 50 queries; FRI fold 16 | `risc0-zkp-3.0.4/src/lib.rs:41-51` | +| Recursion | lift, join and resolve programs at `RECURSION_PO2 = 18` rows | `risc0-zkvm-3.0.4/src/host/recursion/prove/mod.rs:58` | +| The GPU buffers | `check + ctrl + data + accum + mix + out` elements x 4 bytes, printed by the CUDA HAL at `eval_check` | `risc0-circuit-rv32im-4.0.4/src/prove/hal/cuda.rs:181-200` | + +From those constants the trace of a segment is `2^po2 x 315 x 4` bytes and its LDE 4x that, so, approximate: po2 18 is +0.3 GB of trace and 1.5 GB with the LDE, po2 19 is 3.1 GB, po2 20 is 6.2 GB, po2 21 is 12.3 GB, before the check +polynomial, the extension-field accumulators and the Merkle trees. Two things follow. A RISC Zero segment at the +default 2^20 is in the same class as one SP1 core shard, not smaller. And a RISC Zero segment at 2^18 or 2^19 is a +2 to 4 GB object: the only reason a 12 GB card could not prove one is the fixed overhead of the recursion circuits +(2^18 rows each) and the allocator, which is the measurement the prover-floor agent takes on PC 2 if SP1 cannot go +under 11 GB. Section 2.2 carries the documented numbers. + +### 1.6 Which terms the shard size can move, and which it cannot + +| Lever | Moves | Does not move | +|---|---|---| +| Our `S_p` (pgas per shard) | the live trace, the codeword, the zerocheck: everything in 1.2 marked "yes" | the maximum-sized buffers, the recursion step, the keys, the pool | +| SP1's `HEIGHT_THRESHOLD` (the one knob the server honours) | the rows per table in one core shard, so the live buffers | the fixed terms (measured: 13,861 MiB on the empty shard with every knob at its minimum) | +| A server patch: size `max_trace_size` to the shard, release the pool, one core worker | the 2.26 GiB buffer, the high-water mark, the in-flight count | the recursion step's fixed shape and the key caches | +| A different proof system | the blowup (sumcheck-only and linear-code systems have none, 2.4), the recursion shape | the trace itself: a RISC-V cycle costs tens of cells in every zkVM | + +## 2. Every current proving route + +Read 5 October 2026, 22:10 to 23:00 UTC, by four research agents and this one; every cell names its page or file. +"Not documented" means the project publishes no figure, which for a memory floor is itself the finding. + +### 2.1 The zkVMs with a GPU prover + +| Prover | Proof system, field, chunk | GPU support and the documented minimum memory | Throughput, on what | Verification of the recursive proof; wrapper | Licence | State, October 2026 | +|---|---|---|---|---|---|---| +| **SP1 6.8.1** (ours) | Hypercube: multilinear, jagged PCS, BaseFold, LogUp GKR; KoalaBear; core shards of up to 2^24 cycles and 402 M cells (section 1) | CUDA only. Docs: "24GB or more VRAM", compute capability 8.0+, Linux x86_64 (docs.succinct.xyz, hardware-acceleration page). Code: panic under 20 GB physical (`builder.rs:35-38`). Measured here: 13.9 GB floor, 20.4 GB at the adopted shard, 28.3 GB at the prototype shard. Issue #2950: two clients on a 48 GB L40S hold 41 to 43 GB; a single 6 GiB tensor allocation failed | 4.3 s for the adopted shard, 10.8 s for the prototype one on a 5090 (bench-log); Succinct: 99.7% of Ethereum blocks under 12 s on 16 x RTX 5090 (blog.succinct.xyz, 18 Nov 2025) | compressed proof 1,272,897 bytes, verified in 0.032 to 0.040 s here (`--mode verify-segment`); Groth16 about 260 bytes and about 270 k gas, Plonk about 868 bytes and 300 k gas (docs, proof-types page); the Groth16 wrap needs about 14 GB of host RAM, Plonk about 60 GB (hardware-requirements page) | Apache-2.0 or MIT for the repository including `sp1-gpu/` (`LICENSE-APACHE`, `LICENSE-MIT` at the root; no separate licence under `sp1-gpu/`); `sp1-cluster` is Business Source 1.1 | v6.8.1 of 24 Sep 2026 is the latest tag; mainnet for Ethereum proving since 19 Feb 2026 (blog); AMD port PR #2668 closed unmerged 20 Mar 2026; no Metal, Vulkan or WebGPU | +| **RISC Zero 3.0.x** | FRI STARK (DEEP-ALI), BabyBear, Poseidon2, blowup 4, 50 queries; segments of `2^po2` cycles, default po2 20, allowed 13 to 24 (section 1.5); lift, join, resolve recursion at 2^18 rows; keccak as a separate circuit | CUDA and **Metal** (`risc0/sys/kernels/zkp/metal/`; on Apple silicon the Metal path is on automatically, `risc0/zkvm/build.rs`). Documented memory per segment: Bento design page, 1 M cycles 9 to 10 GB, 2 M 17 to 18 GB, 4 M 32 to 34 GB; Boundless performance page, the largest `SEGMENT_SIZE` per card: 8 GB card po2 19, 16 GB po2 20, 20 GB po2 21, 40 GB po2 22; docs: "less than 10 GB available: change the segment size limit" (dev.risczero.com, local proving); `env.rs:190-192`: "lowering this value by 1 will cut memory consumption by about half". PR #3761 (June 2026): po2 22 did not fit a 24 GB 4090 until the `low_vram` buffer reuse | 4090: 808 kHz at po2 21, 1,207 kHz at po2 22 with PR #3761 (end to end to a succinct receipt); Apple M2 Pro about 14 kHz on the 2023 datasheet, approximate (the page was unreachable tonight); real-time Ethereum on about 160 x 4090 (blog, approximate) | succinct receipt 222,668 bytes, constant; about 100 ms to verify, approximate (`gsr-stark-verifier` PR #5, mirrored docs); Groth16 seal 256 bytes, about 200 to 300 k gas, approximate; the Groth16 wrapper is x86 only, not on Apple silicon (docs) | Apache-2.0 or MIT, CUDA and Metal kernels included (`risc0/sys/kernels/zkp/cuda/eltwise.cu:1-13`); Bento is BSL 1.1 with a change date already passed | v3.0.6 of 17 Jul 2026 on the maintained line; `main` is 5.0.0 with no release body; `RISC0_PROVER=actor` multi-GPU scheduler experimental since 3.0.1 (`r0vm/src/actors/factory.rs:183-195` carries measured per-po2 memory tokens: po2 18 = 8, 19 = 10, 20 = 15, 21 = 24; lift and join = 3) | +| **Airbender** (Matter Labs) | DEEP STARK, FRI, Mersenne31; chunks of 2^22 cycles; Boojum then FFLONK wrap (docs.zksync.io, airbender page) | CUDA only. "any GPU with 22GB RAM" for production (zksync.io/airbender); the code has memory presets `GiB21` (24 GB cards) and `GiB30`, raised from 29 because Ethereum blocks failed to allocate at 29 GiB (PR #448, `gpu/execution_prover/src/prover/config.rs`); the final SNARK is CPU with about 150 GB of RAM (`docs/gpu.md`) | one H100: 21.8 MHz base layer, 8.5 MHz end to end, about 35 s an Ethereum block (June 2025 post); ethproofs.org today: 4 x 5090 2.3 s average | FFLONK over BN254 on chain; gas not published | MIT or Apache-2.0 | v0.6.0-rc.2; Veridise audit Feb to Apr 2026; live for ZKsync Atlas chains | +| **ZisK** (Polygon spin-out) | eSTARK over Goldilocks (pil2-stark), Poseidon2, approximate; main instance 2^22 to 2^23 rows, chunks of up to 2^22 steps (PR #1238) | CUDA only, CUDA 12.9+; **no VRAM floor documented**; workers need about 32 GB of host RAM, the assembly emulator 64 GB (docs, limits and distributed pages); Cysic's Venus fork submits from one RTX 4090 | 4 x 5090: p99 9.62 s on Ethereum blocks (Aug 2026); 24 x 5090 6.56 s average (Nov 2025) | PLONK wrapper verified by Solidity (`zisk-contracts`); 128-bit claimed | Apache-2.0 or MIT | v1.3.1-alpha, 30 Sep 2026, "undergoing security and correctness audits" (README) | +| **OpenVM 2.0** (Axiom) | SWIRL: sumcheck, zerocheck, LogUp GKR, stacked reduction into WHIR; BabyBear; segments by metered trace height (blog.openvm.dev/2.0) | CUDA only; "at least 24GB of VRAM": L40, 4090, L40S, 5090 (blog.openvm.dev/openvm-gpu) | 11.4 MHz on one 5090, 139 MHz on 16; 2.1 preview: 4 x 5090 p99 9.7 s | STARK proof under 300 KB; Halo2-KZG wrapper, 316 k gas; the Halo2 wrap 8.1 s on a 5090 | MIT or Apache-2.0, GPU prover included | v2.0.2 of 14 Aug 2026; zkSecurity audit of SWIRL; Scroll's prover builds on it | +| **Pico** (Brevis) | Plonky3 STARK, KoalaBear default; chunk size a parameter with no documented default | CUDA via `pico-gpu`; **no VRAM figure published**; every run on 32 GB 5090s | Prism 2.1: 16 x 5090 over two machines, 4.87 s average on Ethereum blocks | Groth16 via gnark | core MIT or Apache; **`pico-gpu` is BUSL-1.1** and "not recommended for production" (its README) | v2.1.2, Aug 2026; Sherlock audit | +| **Ziren** (ZKM, MIPS) | Plonky3-class, KoalaBear, LogUp GKR, WHIR | CUDA 12, compute capability 8.6+, "24 GB VRAM or higher"; the GPU prover is a Docker image pinned by digest, source "planned H1 2026" (docs.zkm.io prover page; an independent evaluation of v1.1.4 says the GPU path is not open) | one 5090: 5.9 MHz on a 288 M-cycle block; 4 GPUs 3.1 to 3.3x | compressed proof 603 KiB; Groth16 or PLONK | core MIT or Apache; GPU image licence unspecified | v1.2.7; no public audit cited | +| **Stwo / S-two** (StarkWare) | Circle STARK over Mersenne31; blowup 1 (rate 1/2), 70 queries, 26 bits of grinding | **CPU SIMD first** (AVX2, AVX-512, NEON, WASM); GPU through ICICLE-Stwo (Ingonyama): about 3 GB of trace in GPU memory, out of memory from 2^23 rows; a WebGPU port of the constraint evaluation (zkSecurity blog, April 2025); client-side proving under 1 GB after a spill allocator (third-party PR) | 620 k Poseidon2 a second on an M3 laptop | via a Cairo verifier (proofs of proofs); sizes not published here | Apache-2.0 | live on Starknet mainnet since 3 Nov 2025; no RISC-V guest of its own (Nexus 3.0 is the RISC-V zkVM on it, BUSL-1.1 until 2029) | +| **Jolt** (a16z) | sumcheck and lookups (Lasso lineage, Twist and Shout memory checking); PCS Dory over BN254 by default, or **Akita**, a lattice commitment over a 128-bit prime field (Sep 2026, "Lattice Jolt"); RV64IMAC; no continuations (the book's recursion page is "under construction") | **No CUDA in the public repository** (LayerZero's "Jolt Pro" CUDA port is private); **Metal**: PR #1938 merged 30 Sep 2026 (the `jolt-metal` runtime crate), PR #1733 (the full prover on Metal) still a draft. Memory: "about 200 bytes per cycle" with Akita (a16z substack, Sep 2026); the book: "under 2 GB of memory per million cycles"; a streaming prover bounded to "a few GBs" is planned, not shipped | over 2 M cycles a second on a laptop CPU with Akita, over 10 M with Metal on a Apple laptop (a16z substack, Sep 2026); PR #1733: M5 Max, 2^25 cycles in 19.8 s, 2^27 in 77 s at an 89.4 GiB footprint | proof about 50 KB (Dory) or 65 to 80 KB (Akita); verify sub-second, approximate; on-chain 1.3 to 2 M gas estimated in 2024; no Groth16 wrapper shipped | MIT or Apache-2.0 | `v0.3.0-alpha` is the last tag (1 Oct 2025); README: "not suitable for production use"; no audit | +| **Ceno** (Scroll) | GKR tower prover, BabyBear, WHIR or BaseFold PCS; RV32IM | CUDA, but the real HAL is in a **private** `ceno-gpu` repository (the public one is a mock); no memory numbers | 2 GPUs 1.6x over one (PR #1403, Sep 2026) | via OpenVM recursion to Halo2 | Apache-2.0 | README: "under construction and not suitable for use in production" | +| **Nexus 3.0** | on Stwo (Circle STARK, M31) | no GPU path documented | none published | not published | **BUSL-1.1** until 10 Feb 2029 | last push 6 Jan 2026; folding (Nova family) abandoned June 2025 for the STARK | +| **Valida** (Lita) | Plonky3 STARK | no GPU; a CUDA port "underway" in July 2025 | none current | not published | Apache or MIT | dormant since Sep 2025; documented soundness issues in its own benchmarks page | +| **Powdr** | no longer a zkVM: `powdrVM` archived; powdr is autoprecompiles on OpenVM | OpenVM's | OpenVM's | OpenVM's | MIT or Apache | tooling layer; "DO NOT USE FOR PRODUCTION" | +| **Binius / Binius64** (Irreducible) | binary-field SNARK, BaseFold-style FRI over GF(2^64) words | CPU SIMD only; the FPGA work was dropped 9 Sep 2025 ("FPGAs underperformed GPUs"); no GPU | ECDSA aggregation about 5x over SP1 and R0VM on L40S GPUs, on CPU (the page carries methodology corrections) | hash-based; no recursion shipped | Apache-2.0 or MIT | **the company shut down 12 Nov 2025**; the only zkVM on it (PetraVM) is archived | +| 2026 entrants | Cysic Venus (a ZisK fork with cudaGraph tuning and an FPGA backend, Apache or MIT, "do not use in production"); Zilkworm (Erigon's C++ guest on Airbender, 2 x 5090 9.3 s); zkDTVM (evmone guest, 4 x 5090 4.7 s, no public docs); Delphinus zkWasm (Halo2 on BN254, a 4090 minimum plus 58 GB of host RAM); Miden (Goldilocks STARK, client-side, Metal via `miden-gpu`, mainnet alpha planned); Boojum (2023 claim of proving on a 16 GB card, superseded by Airbender) | none states a floor under 24 GB on a GPU | | | | | + +The reading of the table. No shipped zkVM documents a GPU floor under 24 GB except RISC Zero, whose memory is a +function of a runtime knob (`segment_limit_po2`) and is published per card size by Boundless. SP1's 24 GB is a line +of code, not a property of the proof system: Airbender, OpenVM and Pico all pad to the card they tune on, and all +three say 24 or 32 GB because their market is Ethereum blocks on 5090 clusters. The real-time race has collapsed to 2 +to 4 consumer cards per block (ethproofs.org, 5 October 2026), which is why nobody is tuning for a 12 GB card: the +customer buys 5090s. Igneum's customer is the miner who already owns the card, so Igneum has to do the tuning itself. + +### 2.2 The sumcheck and GKR family against FRI STARKs, in memory terms + +| Family | What it holds at the peak | Blowup | The GPU figure today | Source | +|---|---|---|---|---| +| FRI STARK (RISC Zero, Airbender, ZisK, Pico, Stwo, SP1 3 and 4) | trace, its Reed-Solomon codeword at the blowup, the Merkle trees, the DEEP quotient in the extension field | 4x (RISC Zero, Airbender), 2x (SP1 3 and 4, approximate), 2x (Stwo at rate 1/2) | RISC Zero: 9 to 10 GB per 1 M cycles (Bento) | section 1.5; Boundless pages | +| Sumcheck with a hash-based PCS (SP1 Hypercube, OpenVM SWIRL, Ceno, Ziren) | the trace as multilinears, the extension-field folded copies of the zerocheck and GKR, and the BaseFold or WHIR codeword of the stacked polynomial (still a Reed-Solomon encoding, at 4x in SP1, section 1.1) | 4x of the stacked data in SP1; WHIR's rate is a parameter | SP1: the fixed 13.9 GB plus about 6.5 GB for a 4.7 M-cycle shard (measured) | section 1 | +| Sumcheck with a curve or lattice PCS (Jolt) | the trace and the one-hot columns; **no codeword at all**: Dory commits by MSM and Akita by lattice hashing, so memory is bytes per cycle with no blowup | none | no GPU figure: 200 bytes a cycle on CPU (Akita), so the adopted 4.7 M-cycle shard is about 0.9 GB of prover RAM, approximate (derived) | a16z substack, Sep 2026; the Jolt book, streaming page | +| GKR (Expander, Ceno) | the circuit witness layer by layer; no codeword for the inner layers | none inside; a PCS for the inputs | Expander: 16 MB per Keccak, approximate | Polyhedra blog (returned 530 tonight) | +| Linear-code PCS (Ligero, Brakedown, Ligerito, Blaze) | one encoded matrix and one Merkle tree; linear time, no FFT | rate 1/2 to 1/4 | no prover memory benchmarks found; Linea's Vortex is the only production use | eprint 2021/1043, 2025/1187, 2024/1609; `linea-monorepo/prover/protocol/compiler/vortex` | +| Binius (binary field) | words of GF(2^64) and a BaseFold FRI | 2x to 4x | none; CPU only; company closed | irreducible.com posts | + +The memory law in one line: a FRI or BaseFold prover holds `blowup x trace` plus the trace itself plus extension-field +working copies, so 8 to 12 bytes per cell at the peak; a Jolt-class prover holds the trace and its lookups at about 4 +bytes per cell and commits without encoding. The figure that matters for us is not the ratio but the absolute: our +adopted shard is small enough (about 280 M cells, section 1.4) that a FRI-class prover sized to it fits a 12 GB card, +and a Jolt-class one fits a phone. The reason SP1 does not fit today is section 1.3, not the proof system. + +### 2.3 Folding schemes + +| Scheme | Prover memory per step | The verifier at the end | Field | GPU | Fit for Igneum | +|---|---|---|---|---|---| +| Nova, SuperNova, HyperNova, ProtoStar, Mova; Sonobe as the library | one step's witness plus the running instance: tiny by construction (eprint 2021/370) | an IVC proof of O(F) group elements, compressed by a SNARK: Sonobe's decider is Groth16 over BN254 with KZG, about 11.9 M constraints for a 500 k-constraint step (sonobe.pse.dev, decider page); MicroNova about 2.2 M gas (eprint 2024/2099) | curve cycles (Pasta, BN254 and Grumpkin) | partial: sppark MSM on Pasta, a GPL-3 `cuda-nova` for BN254; Sonobe lists GPU as a plan | **no**: a RISC-V step over a 256-bit curve cycle costs two MSMs per step, the opposite of the hash-based consumer-card design decision (design 5.6), and the verifier changes to pairings | +| Nexus zkVM 1 and 2 | the prover ran "on as little as 1 GB of RAM" (whitepaper, approximate) | a curve SNARK | curve cycle | none | **abandoned by its own author**: Nexus 3.0 (25 June 2025) moved to a Circle STARK, "proofs are smaller, faster to generate" (StarkWare blog) | +| LatticeFold, LatticeFold+, Neo, SuperNeo; Nightstream as the zkVM | one step plus an accumulator, lattice commitments over 64-bit fields | a Spartan-class decider | Goldilocks named; BabyBear and KoalaBear not | none; Nethermind's LatticeFold is a "proof-of-concept prototype" whose benches take 48 h; Nightstream is "research software, not production-ready" with its RV32IM prototype removed | **not before 2027 at the earliest**; the first candidate that folds small-field STARK steps | +| Arc, WARP (hash-based accumulation of Reed-Solomon proximity claims) | small: Merkle openings per step (eprint 2024/1731, 2025/753) | a FRI-style accumulator check | any STARK field | none; no public implementation found | the right primitive on paper for folding RISC-V STARK shards with a hash-based verifier; nothing to adopt | +| Mangrove, Nebula | 390 MB peak at 2^24 gates (Mangrove, eprint 2024/416); pay-per-use steps (Nebula) | curve SNARK | curve cycle | none | research | + +What folding would mean for a shard: the shard prover would hold one transaction's step at a time and the memory +floor would vanish; the price is a curve-based decider at the end of every shard (seconds on a CPU, a different +verifier in the node, pairings on the light-client path), and no production code over our field. Today's small-field +zkVMs get their bounded memory from segmenting and recursion (2.4), not from folding. Folding is a watch item, not a +route. + +### 2.4 Continuations and segment proving at small sizes + +| Prover | The segment knob | What a 2^18 or 2^19 segment costs | Can our shard be cut that way inside the guest? | +|---|---|---|---| +| RISC Zero | `segment_limit_po2`, runtime, 13 to 24 (`env.rs:181-186`); the recursion lifts every segment at 2^18 rows and joins them in a tree | po2 19 is the documented fit for an 8 GB card and po2 20 for a 16 GB card (Boundless); the scheduler's measured tokens put po2 18 at about a third of po2 21 and a lift or join at an eighth (`factory.rs`); the 4.7 M-cycle shard at po2 19 is 9 segments, 9 lifts and 8 joins | yes, with no guest change: the zkVM cuts at the limit on its own; the shard statement is unchanged and one succinct receipt comes out | +| SP1 | `HEIGHT_THRESHOLD` (honoured) and `ELEMENT_THRESHOLD` (overwritten by the server, section 1.2); `SHARD_SIZE` up to 2^24 cycles | the live buffers shrink (22.9 GB against 28.3 GB on the prototype shard at `HEIGHT_THRESHOLD 2^20`, bench-log) and the fixed 13.9 GB does not; the compose tree folds 4 proofs at a time | yes, the same way; but the floor is the server's, so the cut buys nothing until the server is re-sized (route A) | +| Our own planner | `S_p` in pgas, a consensus parameter changed by the fee-switch pattern (`docs/plans/fee-switch-devnet.md`); the cut is at transaction boundaries (`core/src/plan.rs`) | halving `S_p` halves the live trace and doubles the shard count; the aggregator verifies one deferred proof per shard (1.66 M cycles for 4 shards, bench-log 4 October) and the chained aggregation is one per block whatever the count (9.7 s on a mining 5090) | yes, already implemented; a transaction above `S_p` stays one shard and the zkVM's own continuations cover it (spec 7.6 item 1) | +| Jolt | none: monolithic; streaming planned | n/a | no | + +### 2.5 Distributed proving across several small cards + +| System | How one execution is split | Per-card memory | Several cards on one host | What it means for four 12 GB cards | +|---|---|---|---|---| +| SP1 cluster (`sp1-cluster`, BSL 1.1) | by core shard: `ProveShard`, `RecursionReduce`, `RecursionDeferred` and `ShrinkWrap` tasks go to GPU workers, `CoreExecute` and the Groth16 or Plonk wrap to CPU workers (`crates/prover-types/src/lib.rs:31-41`); artifacts through Redis and S3 | "only certain GPUs with >= 24GB RAM are supported" (`infra/charts/sp1-cluster/values-example.yaml`); one task holds one whole card | yes: one GPU node process per card (`gpu{0..7}` services in the docker-compose deployment page); the local server itself supports device 0 only (`task.rs:160`, "only device 0 is supported at the moment"), one server per `CUDA_VISIBLE_DEVICES` | the split is by core shard, and the adopted shard is ONE core shard (section 1.3), so there is nothing to split across cards; the floor per card is unchanged. The cluster is a throughput tool, and its code is BSL | +| RISC Zero Bento (Boundless) | by segment onto Redis; `gpu_prove_agent` spawns one prove agent per card with `CUDA_VISIBLE_DEVICES`; the same `SEGMENT_SIZE` for every card, "the lowest common denominator"; joins form a tree (docs.boundless.network, performance-optimization and bento pages; `compose.yml:62-113`) | by `SEGMENT_SIZE` (2.1): 8 GB po2 19, 16 GB po2 20 | yes, documented: one 16 GB card 264 kHz, two 431 kHz (sub-linear, "bound by bus bandwidth, memory") | **the one documented configuration**: four 12 GB cards at po2 19 or 20 take segments off one queue and the joins fold them; the per-card floor is the segment, and the cost of small segments is the lift and join count (9 lifts and 8 joins for the adopted shard at po2 19) | +| RISC Zero `RISC0_PROVER=actor` | one process, several cards, a token budget per card from measured memory per po2 (`r0vm/src/actors/factory.rs`) | per po2 | yes, experimental since 3.0.1 | the same model without Bento's services | +| Pico Prism 2.0 | a global task queue across two machines, 16 x 5090, 100 Gbps between them (Brevis blog, May 2026) | not published | yes | no figure | +| OpenVM | metered execution on the CPU, segments to GPUs, an aggregation tree, "clusters with hundreds of GPUs" (docs, distributed-proving page) | 24 GB | yes | no 12 GB path | +| ZisK | coordinator and stateless workers; "splits the trace into pieces, proves each in parallel on separate machines, and aggregates"; the first worker aggregates a binary tree (docs, distributed execution page) | not documented | yes, `--gpu` per worker | no figure | +| Ceno | shards round-robin by `shard_id % device_count`; a shard never split across cards; one CUDA context per device (PR #1403) | not stated | yes | the same model | +| Column-split of one trace across cards (FRIttata eprint 2025/1285, HyperFond 2025/1349, deVirgo arXiv 2210.00264, Pianist 2023/1271, Cirrus 2024/1873, SumFold 2025/1653) | the sumcheck or FRI itself is distributed, each worker holding a slice of the columns or rows and exchanging small messages | a slice | research code or CPU clusters only | nothing shipped; the one route that would let four 12 GB cards hold what one 32 GB card holds for a SINGLE core shard, and nobody has it in a zkVM | + +The reading. Every shipping system splits by rows (segments, shards, chunks), proves each on one card, and folds +with recursion. So "four 12 GB cards do what one 32 GB card does" is true for throughput (four shards in flight, or +four segments of one shard, then a join tree) and false for a single unit that exceeds one card: that unit must be cut +smaller, by the zkVM's segment knob (RISC Zero) or by our planner (`S_p`). For Igneum the units are already small and +independent (a shard, assigned by sortition), so the rig's natural mode is one prover process per card, each taking +its own shard. The distributed route therefore costs nothing in protocol and lands as an app change (section 3, route C). + +### 2.6 Proof systems that run on AMD or Apple + +| Target | What exists | Status | Source | +|---|---|---|---| +| Apple, RISC Zero Metal | the full STARK prover (rv32im, keccak, recursion) on Metal, automatic on Apple silicon; the Groth16 wrap x86 only | shipped, maintained (PR #3761's June 2026 matrix lists "metal (Mac M-series): build, run"); the only speed published is a 2023 M2 datasheet (14 to 93 kHz, approximate); nothing for M3, M4 or M5 | `risc0/sys/kernels/zkp/metal/`, dev.risczero.com local-proving page | +| Apple, ICICLE Metal (Ingonyama) | MSM, NTT, sumcheck on Metal since v3.6.0 (Mar 2025); "missing API implementations for Poseidon and Poseidon2 hashes, Merkle tree" at that release; v4.0.0 of 11 Jul 2025 is the latest | a library, closed-source backends under a free research licence (dev.ingonyama.com, install_gpu_backend page); no STARK prover built on it for Metal | ICICLE releases, the Metal blog | +| Apple, Jolt Metal | PR #1938 merged 30 Sep 2026 (the runtime and field kernels); PR #1733, the prover itself, a draft: M5 Max 2^25 cycles in 19.8 s, 3.2x over its CPU | the fastest Apple number anyone has published, in a draft | github.com/a16z/jolt pulls 1733 and 1938 | +| Apple, Stwo | CPU SIMD with NEON; ICICLE-Stwo promises Metal | CPU path shipped; no RISC-V guest of its own | stwo README, Ingonyama blog | +| Apple, Miden | `miden-gpu` on Metal | Cairo-class VM, not RISC-V | hackmd (bobbinth) | +| AMD, sppark | "A limited support for AMD's RDNA and CDNA GPUs" (README); SP1's tree carries no HIP build | a library | github.com/supranational/sppark | +| AMD, SP1 PR #2668 | an external port to RDNA3 and RDNA4 with "a caching memory allocator to work around hipMallocAsync leak bug" | **closed unmerged 20 Mar 2026** | github.com/succinctlabs/sp1/pull/2668 | +| AMD, OpenVM stark-backend HIP fork | `cuda2hip.hpp` so the same `.cu` builds under nvcc and hipcc, native `mont32_t.hip`, tested on gfx1100, targets MI300X and 7900 XTX | merged 15 Sep 2026 in a fork (Okm165/stark-backend PR #2), not upstream | the PR | +| AMD, Goldilocks NTT and STARK on ROCm | 19.19 ms NTT at 2^27 on an RX 7900 XTX; a Goldilocks STARK backend on HIP | research posts | ethresear.ch, qingming-g64-ntt and stark-g64 | +| Vulkan and WebGPU | ICICLE's Vulkan build (Jan 2025) with no installable backend; zkSecurity's WebGPU Stwo (5x on constraint evaluation, 2x end to end, no 64-bit integers in WGSL); ZPrize WebGPU MSM | prototypes; nothing proves a RISC-V shard | the pages named | + +Said plainly, as `docs/analysis/amd-proving.md` said it: on 5 October 2026 no zkVM proves on an AMD GPU, and the only +Apple prover that ships is RISC Zero's. The AMD work that exists is two ports of CUDA STARK kernels through a HIP shim, +one closed, one in a fork; both are days of agent work to revive against a given tree, and PC 1's RX 9070 XT (gfx1201) +is the card to measure on. + +## 3. For each route: the change to our guest, the aggregator and the node's verifier; the cost; the risk; 12 GB under 60 s + +What the node verifies today: SP1 compressed proofs through `igneum-prove-host --mode verify` and `verify-segment` +(`vendor/igneum-node-pv1/igneum/exec/src/proving.rs:41-46, 878-926`), the pinned ids read at start and named in the +native statement (`program_ids`, `IGNEUM_PROOF_PROGRAM_IDS`), the record bound in a BLS-signed `ProofRecord` (version +1) or `SegmentRecord` (version 2) with the proof's SHA-256 (spec 7.7 item 1, 7.8 item 3). A different proof system +means a new `ProofSystem` version (design 5.6), a new pinned id, a second verifier command, and the record's version +field telling the node which. The swap procedure of design 5.6 (test vectors, 90% signalling, a 3-month overlap with +both verifiers, a wrap of the last old proof) is the path for any of the rows below that change the family. + +The 60-s test. The litepaper's minute, the launch target of 20 to 60 s behind the tip, and the mine-and-prove +measurement that a shared card proves 3 to 4x slower (bench-log, `chain-pc2-pv1c`). No 12 GB card has run any +prover in this repository; the 12 GB times below are approximate, scaled from the 5090 by memory bandwidth (an RTX +3060 at 360 GB/s and an RTX 4070 at 504 GB/s against the 5090's 1,792 GB/s, NVIDIA's published figures, approximate), +which is the term a STARK prover is bound by. They are the numbers the first 3060-class run replaces. + +| Route | Guest | Aggregator | Node verifier and record | Cost (agent time) | Risk | Reaches 12 GB with a real shard under 60 s? | +|---|---|---|---|---|---|---| +| **A. Re-size SP1's GPU server** (the prover-floor agent's patch, running tonight): remove the 20 GB panic (`builder.rs:37`), size `max_trace_size` to the shard (honour `ELEMENT_THRESHOLD`, or set the core allocation from the shard's measured cells), one core worker and a buffer of 1, a release threshold so the pool returns memory between stages, `drop_ldes` on; build with `CUDA_ARCHS` for Ampere, Ada and Blackwell | none: the same ELF, the same pinned id (the verifying key hashes the program and its preprocessed tables, not the server's buffer sizes; `HEIGHT_THRESHOLD` only shortens tables below the verifier's 2^22 maximum) | none: the compressed proof format and the aggregator guest are unchanged | none: the same `--mode verify`; the record format unchanged | hours to one day: a fork of `sp1-gpu/crates/prover_components` and `jagged_tracegen` (Apache or MIT), the 11-min cross-build, a per-card profile in `provedefault.rs`, a CI check that the fork's constants match the pinned verifier's | low on the protocol, medium on the build: the fixed recursion stage may hold the floor near 8 to 9 GB (section 1.3, approximate) and the first measurement says whether 11 GB is reached; a fork of `sp1-gpu` to carry forward on every SP1 release; the server rejects nothing it cannot hold, so an out-of-memory shard must fail cleanly and be left (the pool's rule today) | **memory: likely for the adopted shard** (10 to 11 GB on paper, section 1.4), **not** for the prototype shard (28 GB of live trace). **Time: prove-only yes** (4.3 s on the 5090 scales to about 15 to 22 s on a 3060 and 10 to 15 s on a 4070, approximate); **mine-and-prove on a 12 GB card: no at `S_p`** (3 to 4x on a shared card puts a 3060 at 45 to 90 s, approximate, and the miner's 1.7 GB on top of 11 GB does not fit), yes at `S_p/2` on a 4070 if the floor lands under 9 GB (approximate). The measurement decides; this is the route the gate waits on | +| **B. Halve `S_p`** (30,000 to 15,000 pgas, the fee-switch pattern): more and smaller shards | none | none: one deferred proof per shard, so 2x the shards per block; the chained aggregation stays one per block (9.7 s mining, 2.5 s alone) | none | hours: a fee-table change and a rollout plan like `fee-switch-devnet.md` | low: more records per block (the coinbase carries at most 8 shard records, spec 7.7 item 2, so `B_p / S_p` must stay at 8 or under); the assignment window and sortition unchanged | **alone, no**: the floor is the server's (13.9 GB at 0 cycles). **With A, it is the dial** that moves a 12 GB card from prove-only to mine-and-prove, and a 16 GB card to a comfortable fit | +| **C. One prover process per card on a rig** (the distributed route): the app runs one `sp1-gpu-server` per NVIDIA card (`CUDA_VISIBLE_DEVICES`, the per-device socket of `sp1-cuda/src/client.rs:211`, `.cuda().with_device_id(n)`), one host process per card, each taking its own assigned shard; the rig installer already picks cards (`igneum-rig-lib.sh`, `prover_decision`) | none | none: shards are independent units by design (spec 7.2); the aggregator runs on the biggest card | none | one day: the app's prover loop per card (`prover.rs` runs one loop today), the Settings and tile per card, the rig installer's prover unit per card, the socket cleanup per device (the root-socket rule of 5 October) | low; the throughput is per card, the host RAM 6 GB of pinned buffers per server (section 1.2), so a 4-card rig needs 32 GB of RAM or route A's smaller buffers | **it does not move the floor**: each card still needs A. It is the route that makes four 12 GB cards worth four shards a cycle, and it ships with A, not instead of it. Splitting ONE shard across cards is not a route: the adopted shard is one core shard (2.5), and column-split provers are research | +| **D. RISC Zero as proof system version 2** (CUDA and Metal; segments at po2 19 or 20) | a second guest: `core/` is plain Rust and ports as is; the precompile patches differ (SP1's `sha3` and `k256` patches against RISC Zero's `sha2`, `k256` and keccak circuit); the shard statement bytes unchanged; a second pinned ELF and image id in `elf/manifest.json` | a RISC Zero aggregator guest using composition (`env::verify` of the shard receipts, dev.risczero.com composition page); the chain rule (N verifies N-1) inside the family; **a block's shards must be one family**, and a chain cannot cross families inside the proof: a family switch lands at a segment boundary as a fresh chain (spec 7.8 item 6 already allows one after an unproven segment; the rule gains "or at a proof-system version change") | a second verifier mode (`--mode verify-r0`, the `risc0-zkvm` verifier, pure Rust, about 100 ms, 222 KB receipts); the record's `version` selects the family; the native statement names the family's pinned id; both verifiers in the node through the overlap of design 5.6 | 3 to 4 days: guest port and pinning 1, aggregator and chain rule 1, node verifier and record version 1, app profile and host modes 0.5, test vectors and the fast-time harness 0.5; plus the measurement day on PC 2 | medium: two proof systems in consensus for the overlap; RISC Zero's Groth16 wrap is x86 only (the light-client path of ledger P3 stays on SP1 or waits); a 222 KB receipt per shard against 1.27 MB today is a gain; the recursion tree per shard (9 lifts and 8 joins at po2 19) is extra time on small cards; `main` is at 5.0.0 with no release body, so the pin is 3.0.6 | **memory: yes by documentation** (po2 19 for an 8 GB card, po2 20 for 16 GB; 9 to 10 GB per 1 M cycles), the first documented sub-12 GB prover. **Time: approximate**: a 4090 does 808 kHz at po2 21, so the adopted shard is about 6 s on a 4090-class card and about 20 to 30 s on a 3060-class one at po2 19, prove-only; beside the miner over 60 s on a 3060, near it on a 4070. The prover-floor agent's PC 2 run is the first real number | +| **E. Airbender, OpenVM, ZisK, Pico, Ziren** as version 2 | a new guest each (RISC-V, except Ziren's MIPS); OpenVM's and ZisK's toolchains are the most complete | each has its own recursion; OpenVM's aggregation and Halo2 wrap are the most documented | a new verifier each (STARK under 300 KB for OpenVM; PLONK or FFLONK for ZisK and Airbender) | 4 to 6 days each | the same two-family cost as D with no memory gain: 21 GiB (Airbender), 24 GB (OpenVM, Ziren), undocumented (ZisK, Pico); Pico's and Ziren's GPU code is BUSL or closed | **no**: none documents a floor under 21 GiB; the race is tuned for 5090 clusters | +| **F. Jolt (Lattice Jolt) as version 2**: a sumcheck prover with no codeword; CPU and Metal | a new guest (RV64IMAC, Jolt's toolchain; no keccak precompile today, approximate, so the trie hashing costs more cycles than in SP1) | **none exists**: no recursion or continuation shipped, so the aggregator would verify N shard proofs natively and the chain rule would live in the native statement until Jolt's recursion lands | a Dory verifier (BN254 pairings, about 50 KB, sub-second, approximate) or an Akita verifier (lattice, 65 to 80 KB); no on-chain verifier shipped | 5 to 8 days for the guest, the verifier and the record; the aggregator question has no answer in the code | high: alpha software, no audit, no production user, no recursion; the proof system of the miner's CPU, not of its card | **memory: yes by a wide margin** (about 0.9 GB for the adopted shard at 200 bytes a cycle, approximate). **Time on a CPU: about 2 to 3 s** for 4.7 M cycles at over 2 M cycles a second (a16z, Sep 2026, laptop CPU; approximate for our guest), **on Metal under 1 s** (PR #1733's 2^25 in 19.8 s on an M5 Max, approximate). The numbers are the best in this document and the software is not shippable | +| **G. Folding** (Nova family, lattice folding) | a step circuit per transaction or per opcode group | a decider per shard | pairing or lattice verifier | weeks of research, no code over our field | the family that Nexus left | **no** today; the watch item for 2027 | +| **H. AMD through a HIP port of SP1's kernels** (PR #2668 revived against 6.8.1, or the `cuda2hip` shim of the OpenVM fork) | none | none | none: the same SP1 proofs | 3 to 5 days plus PC 1's RX 9070 XT to measure; the `hipMallocAsync` leak needs the caching allocator the PR carried | medium: a kernel port with no upstream; the sppark NTT has a limited HIP path and cuPQC none | memory as route A (the same buffers); **time unmeasured on any AMD card**; the one route that gives AMD miners the 20% pool share | +| **I. Apple through RISC Zero Metal** (route D's Metal half) | as D | as D | as D | inside D's 3 to 4 days | the 2023 M2 figure (14 kHz, approximate) says 5 minutes for the adopted shard; an M5 Max is not measured by anyone | **memory: yes** (unified memory, 64 GB on the M5 Max). **Time: unknown**; the Mac measure lock run is the number | + +## 4. The ranked recommendation + +| Rank | Route | Why | Gate | +|---|---|---|---| +| **1. Soonest to 12 GB with the least change: A, with B as the dial and C for rigs** | re-size SP1's GPU server; keep the guest, the aggregator, the verifier and the pinned ids exactly as they are; set `S_p` from the first 12 GB measurement; one prover per card on rigs | nothing in consensus moves; the work is a fork of two Apache crates and an app profile; it is already running tonight; every other route costs days and adds a second verifier | the prover-floor agent's rows: the adopted shard under 11 GB alone and the time on the first 3060-class or 4070-class card, prove-only and beside the miner. If under 11 GB and under 60 s prove-only: ship 0.3.12 with the 12 GB tier as prove-only and `S_p/2` measured for mine-and-prove. If not under 11 GB: route D | +| **2. The fallback if A misses 11 GB, and the Apple route either way: D, RISC Zero as version 2** | the only shipped prover with a documented sub-12 GB configuration and a shipped Metal path; Apache or MIT including the kernels; 222 KB receipts | the swappable interface was built for this (design 5.6) and the node already names the pinned id in the statement, so a second family is a version, not a redesign; the cost is 3 to 4 days plus the overlap | PC 2's po2 19 and 20 rows (memory, time per segment, lift and join) tonight; the Mac's Metal row | +| **3. Best in five years: the sumcheck family without a codeword (Jolt-class), or the sumcheck-plus-WHIR family SP1 and OpenVM already converge on** | Jolt proves the adopted shard in seconds on a laptop CPU at under 1 GB of memory, which is the only route that gives AMD-only, Apple and 8 GB machines the proving share with their existing hardware; its verifier is small (50 to 80 KB); its licence is MIT or Apache. It is alpha with no recursion, so not before it ships a stable release with continuations and an audit. SP1 Hypercube and OpenVM SWIRL are the same mathematics with a hash-based PCS and a GPU today, which is why staying on SP1 now loses nothing in that direction | do not adopt now; re-read Jolt and the Arc or WARP accumulation line at every 6-month era draw (design 5.6's swap procedure needs 90% signalling and a 3-month overlap, so the lead time is the schedule) | a stable Jolt tag with recursion, an audit, and a CUDA or merged Metal prover | +| **The interface question** | yes: `ProofSystem` is versioned (`VERSION`, `program_id`, `verify_segment`), the record carries `version`, the node reads pinned ids at start and names them in the native statement, and the overlap procedure keeps both verifiers in the node for 3 months with `B_p` from the stricter table. What is missing for two families at once is small and named: the record version selecting the verifier command, the fresh-chain rule at a version change, and the shard plan carrying the family per block so a block's shards are homogeneous (the aggregator folds one family). Those three items are in route D's day of node work | so the answer to "ship one now and move to the other later" is yes, and route A ships nothing that has to be undone | | + +The honest statement of what this ranking does not know: no 12 GB card has run any prover here. Route A's time +figures are bandwidth scaling, labelled approximate; route D's are a 4090 figure scaled the same way. The first 3060 +or 4070 in this repository replaces both columns, and the plan is to borrow or buy one this week (a 4070 is the +common 12 GB card of 2026; a 3060 the common older one; both are the gate's named class, design R2). + +## 5. The tier consequences, and the public line while the change is made + +Every number carries its consequences (CLAUDE.md, 5 October 2026). The table says what each tier has today on SP1 +6.8.1, what route 1 (A plus B plus C) gives it if the gate is met, what route 2 (D) adds, and what only route 3 would +give. "Today" is measured; the rest is the routes' expected outcome, labelled, until the measurement. + +| Tier | Today (measured, bench-log 5 October) | Route 1: re-sized SP1 server, `S_p` as the dial, one server per card | Route 2: RISC Zero version 2 | Only route 3 (sumcheck without a codeword) | +|---|---|---|---|---| +| Home miner, one 8 GB NVIDIA card | mines; proves nothing (the server panics under 20 GB) | proves nothing at `S_p` (the floor's fixed terms, 8 to 9 GB approximate, leave no room); perhaps empty shards | prove-only at po2 19 (Boundless' 8 GB tier), the miner paused per shard; time approximate 30 to 60 s | mines and proves on its CPU | +| Home miner, one 12 GB card (3060, 4070) | mines; proves nothing; the litepaper's gate card | **prove-only at `S_p`** if the floor lands under 11 GB (expected, section 1.4): about 15 to 22 s a shard, approximate; **mine-and-prove at `S_p/2`** on a 4070 if the floor is under 9 GB, approximate; on a 3060 the shared card misses 60 s, approximate, so its default is prove-only with the miner paused per shard (the 16 GB rule of `provedefault.rs` today, moved down a tier) | prove-only at po2 20 (16 GB tier) or po2 19; mine-and-prove not inside 60 s on a 3060, approximate | mines and proves, CPU | +| Home miner, one 16 GB card (5080, 4080, 4060 Ti 16 GB) | an empty shard alone (13.9 GB); nothing beside the miner | **mine-and-prove at `S_p`** (11 GB plus the miner's 1.7 GB), about 7 to 12 s a shard alone and 20 to 40 s beside the miner, approximate | mine-and-prove at po2 20 | the same | +| Home miner, one 24 GB card (4090, 3090) | the adopted shard alone (20.4 GB) and beside the miner (22.2 GB, approximate for the card); the prototype shard never | mine-and-prove at `S_p` with 10 GB to spare; the prototype shard (28 GB live) only if the devnet's fee switch has passed, which it has from DAA 210,000 | the same with Metal irrelevant | the same | +| Home miner, one 32 GB card (5090) | everything, measured | everything, with more shards in flight if the pool releases between stages | the same | the same | +| Rig, several NVIDIA cards | one prover on the biggest card (`prover_decision`) | **one server per card**, each its own shard; the aggregator on the biggest card; host RAM 6 GB pinned per server today, under 2 GB with route A's buffers | the same model (Bento's) | the same | +| Pool user | through the pool; who proves is open (spec 09) | unchanged | unchanged | unchanged | +| AMD-only (RX 9070 XT, 7900 XTX) | mines; proves nothing on the card; the CPU path 282 s a shard at 30 GB | unchanged until route H (a HIP port, 3 to 5 days, measured on PC 1's 9070 XT) | unchanged: RISC Zero is CUDA and Metal only | mines and proves on its CPU | +| Apple silicon (M-series) | mines (26.7 MH/s on the M5 Max); the SP1 CPU prover 41 to 55 s for an empty shard, 272 s for a small one | unchanged | **proves on the GPU through Metal** (64 GB unified memory on an M5 Max holds any segment); the time is the measurement | proves in seconds on Metal (Jolt's draft PR figure, approximate) | +| Windows under 32 GB of RAM | off (the WSL2 prover held 7.9 GB) | the pinned buffers fall with `max_trace_size`, so a 16 GB PC likely qualifies, approximate; measure | RISC Zero's CUDA path also runs in WSL2 | | + +The deadlines these fit (spec 7.2 item 3, the litepaper, `proving-v1.md`): the 10-s exclusive window is the 5090's +alone; a 12 GB card at 15 to 22 s proves its assigned shards in the open phase and is paid when no faster card took +them, which on a chain with few 5090s is most of the time; the minute of the litepaper holds for prove-only 12 GB +cards and for mine-and-prove 16 GB cards; the 600-s unproven deadline holds for every tier above the CPU path. + +### The public line while the change is made + +The litepaper's sentence today ("Target: shard size will be set so a 12 GB card proves one shard in about 20 +seconds") is a target and says so (fud-ledger P1, overclaim 27). What this document adds, for `site/litepaper.html`, +`site/miner.html` and the app's Proving tile, in the copy law: + +> Proving runs on NVIDIA cards with 24 GB or more today. A build for 12 GB and 16 GB cards is being measured: the +> memory is the prover's buffers, not the shard, and the fix is a smaller build of the same prover. AMD and Apple +> cards mine. A second prover with an Apple path exists and is the fallback. + +And the rule for the next status line, whichever way the measurement goes: the number, the card it was taken on, and +the tier it moves, in one sentence, the day it is taken. + +### What this document does about it + +| Consequence | Action | Owner | +|---|---|---| +| The gate card has never run a prover here | get a 4070 or 3060 into the measurement loop this week; until then every 12 GB figure stays approximate | coordinator; Josh for the card | +| Route A's gate | the prover-floor agent's rows (asked for by message tonight); if under 11 GB, `provedefault.rs` gains the 12 GB prove-only and 16 GB mine-and-prove tiers and the rig installer one server per card | prover-floor agent, then the proving engineer | +| Route D's measurement | RISC Zero 3.0.6 at po2 19 and 20 on PC 2 (CUDA) and on this Mac (Metal), the same shard statement run natively: memory, time per segment, lift and join, receipt size | prover-floor agent (PC 2); a Mac measure job for Metal | +| The two-family node items (record version selects the verifier, fresh chain at a version change, one family per block) | spec 7.8 gains the three rules when route D starts; nothing changes before | execution engineer | +| AMD | route H is a 3-to-5-day job with a measurement on PC 1's 9070 XT; opened as a plan when route A's result is in | execution engineer | +| The public line | the paragraph above to the site and the tile with the next site pass | site-pages owner | + +## Sources + +Our own: `docs/bench-log.md` entries "proving v1: segment records, the chain rule, the unproven rule" (5 October 2026), +"the SP1 CPU prover on PC 1" (5 October), "shard proving on the RTX 5090" (4 October); `docs/plans/proving-v0.md`, +`proving-v1.md`; `docs/analysis/amd-proving.md`; `docs/spec/07-execution.md` 7.2, 7.6, 7.7, 7.8; `docs/design/execution-layer.md` +5.1 to 5.7; `proving/igneum-prove` (`host/src/proof_system.rs`, `program/src/main.rs`, `aggregator/src/main.rs`, +`elf/manifest.json`); `vendor/igneum-node-pv1/igneum/exec/src/proving.rs`; `app/igneum-app/src/provedefault.rs`, `prover.rs`. + +SP1 6.8.1, read from `~/.cargo/registry/src/index.crates.io-*/` and the vendored tree `vendor/sp1-6.8.1` (commit +c84ada1e, 24 Sep 2026) on the `prover-floor` worktree: `sp1-core-executor-6.8.1/src/opts.rs`, `src/utils.rs`, +`src/artifacts/rv64im_costs.json`; `sp1-prover-6.8.1/src/components.rs`, `src/worker/config.rs`, `src/shapes.rs`; +`sp1-primitives-6.8.1/src/fri_params.rs`; `sp1-verifier-6.8.1/src/compressed/config.rs`; `sp1-hypercube-6.8.1/src/verifier/config.rs`; +`sp1-cuda-6.8.1/src/server.rs`, `src/client.rs`; `sp1-gpu/README.md`, `sp1-gpu/crates/prover_components/src/builder.rs`, +`src/components.rs`, `sp1-gpu/crates/jagged_tracegen/src/lib.rs`, `sp1-gpu/crates/shard_prover/src/prover.rs`, +`sp1-gpu/crates/cuda/src/task.rs`, `src/device.rs`, `sp1-gpu/crates/sys/lib/runtime/mem_pool.cu`, `sp1-gpu/crates/zerocheck/src/primitives.rs`. +Web: docs.succinct.xyz (hardware-acceleration, hardware-requirements, proof-types, security-model, provers introduction, +cluster architecture, docker-compose deployment); blog.succinct.xyz (sp1-hypercube, real-time-proving-16-gpus, +sp1-hypercube-is-now-live-on-mainnet); github.com/succinctlabs/sp1 releases v6.0.0 to v6.8.1, issues #2674, #2930, +#2950, #2969, pulls #2631, #2668, #2723, #2917, #2974; github.com/succinctlabs/sp1-cluster (README, LICENSE, +`infra/charts/sp1-cluster/values-example.yaml`, `crates/worker/src/config.rs`); eprint 2025/917 (jagged polynomial commitments). + +RISC Zero: `~/.cargo/registry` crates `risc0-zkp-3.0.4/src/lib.rs`, `risc0-zkvm-3.0.4/src/receipt.rs`, `src/host/recursion/prove/mod.rs`, +`risc0-circuit-rv32im-4.0.4/src/execute/mod.rs`, `src/zirgen/defs.rs.inc`, `src/prove/hal/cuda.rs`; github.com/risc0/risc0 +`risc0/zkvm/src/host/client/env.rs`, `risc0/zkvm/Cargo.toml`, `risc0/zkvm/build.rs`, `risc0/sys/kernels/zkp/{cuda,metal}/`, +`risc0/r0vm/src/actors/factory.rs`, `risc0/circuit/recursion/src/lib.rs`, releases v2.0.0, v3.0.1, v3.0.6, pull #3761; +dev.risczero.com (local-proving, composition); docs.boundless.network (bento, performance-optimization, quick-start); +github.com/boundless-xyz/boundless (`compose.yml`, `bento/README.md`, `bento/LICENSE-BSL`); github.com/ekrembal/gsr-stark-verifier pull 5; l2beat.com/zk-catalog/risc0. + +Others: zksync.io/airbender, docs.zksync.io airbender and proving pages, github.com/matter-labs/zksync-airbender (README, +`docs/gpu.md`, pull #448), veridise.com (the Airbender audit); 0xpolygonhermez.github.io/zisk (introduction, limits, +distributed execution, installation), github.com/0xPolygonHermez/zisk (README, pull #1238, `zisk-contracts`); +blog.openvm.dev (2.0, 2.0-production, 2.1, openvm-gpu, v1), docs.openvm.dev (security-model, distributed-proving, sdk); +pico-docs.brevis.network, github.com/brevis-network/pico and pico-gpu (README, LICENSE), blog.brevis.network (Prism 1.0, +2.0, 2.1); docs.zkm.io (prover, performance), github.com/ProjectZKM/Ziren, zkm.io (the independent evaluation of v1.1.4), +eprint 2026/2330; github.com/starkware-libs/stwo and stwo-cairo (README), ingonyama.com (ICICLE-Stwo, the Starknet +partnership, ICICLE Metal v3.6), dev.ingonyama.com (install_gpu_backend), blog.zksecurity.xyz/posts/webgpu, starkware.co +(S-two 2.0.0, Nexus on S-two), theblock.co (S-two on Starknet); github.com/a16z/jolt (README, book: intro, dory, akita, +streaming, recursion, blindfold; pulls #1733, #1938; tags), a16zcrypto.substack.com ("How to prove software ran +correctly", Sep 2026), a16zcrypto.com (jolt-6x-speedup, 64-bit-proving-jolt, zkvm-jolt-zero-knowledge, faqs-on-jolts-initial-implementation), +eprint 2025/611; github.com/scroll-tech/ceno (README, Cargo.toml, pull #1403), ceno-gpu-mock, scroll.io (Ceno post), +osec.io ("zkVMs' unfaithful claims"); github.com/nexus-xyz/nexus-zkvm (README, LICENSE), blog.nexus.xyz (roadmap); +lita.gitbook.io (Valida architecture, benchmarks); github.com/powdr-labs/powdr; irreducible.com (announcing-binius64, +reinventing-irreducible, irreducible-shutting-down), github.com/binius-zk/binius64, eprint 2026/1656; eprint 2021/1043, +2022/1010, 2024/1609, 2025/1187, 2024/1586, 2024/185 (linear-code commitments, WHIR, Vortex), github.com/Consensys/linea-monorepo; +PolyhedraZK/Expander and blog.polyhedra.network (returned 530 tonight); eprint 2021/370, 2024/2099, 2024/1220, 2024/416, +2024/1605, 2025/247, 2025/294, 2026/242, 2024/1731, 2025/753, 2026/1371 (folding and accumulation), sonobe.pse.dev, +github.com/privacy-scaling-explorations/sonobe, NethermindEth/latticefold, LFDT-Nightstream/Nightstream; eprint 2023/1271, +2024/1208, 2024/1873, 2025/1349, 2025/1653, 2025/1285, 2018/691, arXiv 2210.00264, 2602.16338 (distributed proving); +github.com/supranational/sppark, github.com/Okm165/stark-backend pull 2, ethresear.ch (qingming G64 NTT and STARK on ROCm); +github.com/cysic-labs/venus, erigon.tech (Zilkworm), github.com/DelphinusLab/prover-node-docker, hackmd.io/@bobbinth +(Miden), ethproofs.org/clusters (5 October 2026). diff --git a/docs/analysis/scratch-soundness.md b/docs/analysis/scratch-soundness.md new file mode 100644 index 000000000..8093f0bb1 --- /dev/null +++ b/docs/analysis/scratch-soundness.md @@ -0,0 +1,408 @@ +# Layer 3 soundness: the per-warp scratch with read-modify-writes + +5 October 2026 (night), cryptographer role, Counter ASIC 2.0 plan step 4 (`docs/plans/counter-asic-2.md`). Branch +`ca2-soundness` on top of `readwidth` b970dda (the scratch as a class parameter, 32 or 128 KiB per warp). Tests: +`igneum-pow/tests/scratch.rs`; Metal runs through `proto-metal/packbench` on the M5 Max; commands and counts in +`docs/bench-log.md` (entry of the same date). Nothing here touches the lottery hash as shipped: variant 5 is behind +`LoadClass::scratch(k, kb)` and is never emitted by generator version 2. + +Every figure below is measured (machine, date, command named) or cited; "approximate" marks a figure from memory. + +## 0. The five findings + +| # | Question | Finding | Status | +|---|---|---|---| +| 1 | Is what is written uniform and beyond a chip's precomputation? | The fill is a bijection of the lane nonce, the rewrite a bijection of the fold value in each word; written words show no bit bias over 3 to 12 million rewrites per class (worst 3.63 sigma of 6). The fill IS precomputable, by design, and at 64 slots 78.5 percent of reads are fill reads. | sound as a function; see 2 for what that means | +| 2 | Does any short cut avoid the writes? | No short cut inside a unit: a slot after d read-modify-writes needs all d fold values (replay test). But the live state is bounded by the read-modify-write count, not by the scratch size, because CPU verification resets the scratch per unit: 64 to 320 bytes per lane at scr2 to scr8, whatever the nominal 32 KiB, 128 KiB or 1 MiB. The named chip (cache mirror plus recompute) keeps that in SRAM at under 5 percent of its mirror and its gain does not move at any share under the 6 GB cap. | NOT sound as an anti-chip layer | +| 3 | Is the verifier's one-warp simulation exact? | Exact when the GPU's lazy per-unit tag is unique over the arena's life and the arena holds no stale tag. The kernels rely on this and neither host guarantees it (no clear at allocation, no clear at the 32-bit wrap of the tag counter, 16.4 minutes on a 5090). With the host contract of section 4.3 the simulation is exact: 14 edge packs twice, 200 fuzz packs, consecutive units on one warp and the wrap inside a launch all match the CPU on Metal (228 of 228); a broken tag and a broken fill are caught (3 of 3). | sound with a host contract; today it is luck | +| 4 | The attack surface of the writes | Out of bounds: impossible by the mask, 42 of 42 emitted kernels pass the static check, which catches six deliberate breaks. Aliasing: none, lane-major arenas disjoint by (warp, lane), two logical units of a wave64 get two arenas. Ordering: one lane, one slot, program order; no cross-lane sharing, no atomics needed. Alignment: 16-byte slots at 16-byte offsets from a 256-byte-aligned base. Wrap: identical to the CPU, tested at the launch level. | sound | +| 5 | What a conformance vector must carry | The class and geometry, the fill and rewrite, the host contract (tags, clearing, groups a multiple of warps), two consecutive units on one warp with a forced slot collision, a unit in the top 256 nonces with the wrap inside the launch, and the fingerprint declared independent of the warp count. The standard three-unit vectors catch a broken tag only through base 1,000,000 and would miss it at a 1 MiB scratch. | defined in section 6 | + +Recommendation (section 10): do not adopt layer 3 as the plan states it (read-modify-writes taken from the 16 +dataset loads). It replaces latency-bound dataset reads with cache-bound ones for the GPU, costs the named chip +nothing it cannot keep in a few megabytes of SRAM, and leaves that chip's gain at 2.4x at every share. The lever +that moves that chip is the mixer multiplier of the M16 analysis (x2 brings it to 1.2x, x4 to 0.6x, under the +verifier's 10 ms gate). If a scratch is kept for another reason, add the read-modify-writes beside the 128 loads, +never in their place, and ship the host contract and the vector of section 6 with it. + +## 1. What the branch implements + +| Piece | Where | What | +|---|---|---| +| Class | `igneum-pow/src/generator.rs:170-230` | `LoadClass { scratch: Some(k), scratch_kb }`: `k` of the 16 memory slots are `Op::Scratch`; `scratch_kb` KiB per warp of 16-byte slots, lane-major, `slots = kb x 2` per lane (32 KiB: 64, 128 KiB: 256); `scratch_slot_mask() = slots - 1` | +| Draw | `generator.rs:488-491` | the first `k` of the 16 drawn load slots become scratch ops (a uniform k-subset); the source register follows the fresh-source rule like a load | +| Fill | `igneum-pow/src/verify.rs:30` | `scratch_fill(seed, base, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot x 0x9e3779b1 + (j + 1) x 0x85ebca77)`, j in 0..2 | +| Fold | `verify.rs:18` | `x = dst ^ w0; x = (rotl(x, 11) x 0x9e3779b1) ^ w1; x = (rotl(x, 11) x 0x9e3779b1) ^ w2; dst = x` (the read-width fold over the three data words) | +| Rewrite | `verify.rs:41` | the slot becomes `(x ^ w1, rotl(x, 7) ^ w2, x + w0)` | +| CPU model | `verify.rs:48-100`, `:305-312` | `ScratchModel`: per (lane, slot) a written bit and three words; an unwritten slot reads as its fill; one model per unit, so a unit starts from the fill | +| Acceptance | `igneum-pow/src/accept.rs:202-215, 374` | a scratch site that reads one slot in all 32 lanes rejects the program (lane-constant site); scratch slots carry bit 31 in the address list and are left out of the distinct-address bound, which now covers the dataset loads only | +| GPU statement | `igneum-pow/src/emit.rs:143-157` | `s_ = rN & mask; v_ = 16-byte load of slot s_; m_ = (v_.x == tag) ? ~0 : 0; w = (v_.yzw & m_) \| (fill & ~m_); fold; dst = x_; 16-byte store of (tag, x_ ^ w1_, rotl(x_, 7) ^ w2_, x_ + w0_)` in Metal, CUDA and OpenCL | +| Persistent prologue | `emit.rs:159-175` | `lane = tid & 31; warp_ = tid >> 5; arena = scratch + (warp_ x 32 + lane) x words_per_lane; for (g_ = warp_; g_ < groups; g_ += nwarps_) { gbase = baseNonce + g_ x 32; tag = salt + g_; ... }` | +| Hosts | `proto-metal/packbench.swift:144-164`, `proto-opencl/host.c:1025-1033, 1268` | the arena is allocated and never written by the host; `salt` starts at 1 and advances by the launch's unit count; no clear at allocation, none at the wrap | + +The constraint of the night (coordinator, 5 October 2026): the whole working set on an 8 GB card stays under 6 GB +(1 GiB table, the layer 5 hot table, the scratch of every resident warp, buffers), which caps the scratch at tens of +KiB per warp. On an RTX 5090 at full occupancy (170 SMs x 64 warps = 10,880 warps, approximate hardware maximum; +the measured version 2 kernel ran 24 warps per SM, 4,080 warps, `docs/bench-log.md` M11, 4 October 2026): + +| Scratch per warp | 10,880 warps | 4,080 warps (measured occupancy) | Table + scratch at 10,880 | Under 6 GB with a 1 GiB table | +|---|---|---|---|---| +| 32 KiB | 340 MiB | 128 MiB | 1,364 MiB | yes | +| 128 KiB | 1,360 MiB | 510 MiB | 2,384 MiB | yes | +| 1 MiB (the first experiment) | 10,880 MiB | 4,080 MiB | 11,904 MiB | no | + +## 2. Question 1: uniformity of what is written + +### 2.1 As functions + +The fill of word j of slot s for lane nonce n is `splitmix32(((n ^ seed[j]) + s x 0x9e3779b1 + (j + 1) x 0x85ebca77))`. +`splitmix32` is a bijection of its 32-bit input; for fixed (seed, s, j) the input is a bijection of n. So over any +2^32 consecutive nonces every 32-bit value appears once as the fill of (s, j): uniform. Test +`fill_is_a_bijection_of_the_nonce`: 2^16 consecutive nonces give 2^16 distinct words for 7 slots x 3 word +positions; the fill of lane l at base b equals the fill of lane 0 at base b + l; it wraps with the nonce +(base 0xffffffe0, lane 32 equals nonce 0). + +The rewrite `(x ^ w1, rotl(x, 7) ^ w2, x + w0)` is, for fixed old content w, a bijection of the fold value x in +EACH word. Test `rewrite_is_a_bijection_of_the_fold_value`: 2^16 consecutive x give 2^16 distinct words in each +position for 16 random w. Consequence: a uniform x gives a uniform word in every position, and the three words +are three images of the same x, so a rewritten slot carries exactly 32 bits of new state behind 96 bits of +storage (from w and any one written word, x is recovered; the test checks all three inversions). + +The fold value x is `fold(dst, w)`, a bijection of `dst` for fixed w (xor, then rotate-multiply-xor twice; the +multiplier is odd). So the written words are uniform whenever `dst` is, and `dst` is a register of the running +program. + +### 2.2 The attack: what a chip can precompute + +The fill is a pure function of (seed, nonce, slot): precomputable, and meant to be (the verifier computes it too). +A chip never stores a fill; it computes it in about 10 integer operations when a slot is first touched. The written +words depend on `dst`, the register state at that instruction, which depends on every earlier instruction of the +hash, including the dataset loads. Nothing about them is precomputable before the hash runs. This is the whole of +what question 1 can give: the writes are as unpredictable as the registers. What that is worth is question 2. + +### 2.3 The stats run (the `TESTS.md` section 3 shape) + +Test `written_words_unbiased_and_rehit_rates`, M5 Max, 5 October 2026, `cargo test --test scratch`: for each +class, programs of `igneum-genesis`, `igneum-genesis/stats1`, `igneum-genesis/stats2`, 2^11 units each (196,608 +hashes per class), closed-form dataset, every read-modify-write traced (`verify::interpret_warp_scratch`). Ones +count per bit of every written word and of the change each rewrite makes (written XOR read), sigma = sqrt(N)/2, +limit 6 sigma like the acceptance rule's output check. + +| Class | Slots per lane | RMW per hash per lane | Rewrites traced | Max bias, written words (sigma) | Max bias, written XOR read (sigma) | +|---|---|---|---|---|---| +| scr2k32 | 64 | 16 | 3,145,728 | 2.61 | 3.40 | +| scr4k32 | 64 | 32 | 6,291,456 | 2.18 | 3.81 | +| scr8k32 | 64 | 64 | 12,582,912 | 3.63 | 2.25 | +| scr2k128 | 256 | 16 | 3,145,728 | 3.36 | 2.19 | +| scr4k128 | 256 | 32 | 6,291,456 | 2.71 | 3.68 | +| scr8k128 | 256 | 64 | 12,582,912 | 2.73 | 2.60 | + +576 bit positions (6 classes x 3 words x 32 bits) at under 4 sigma is what fair coins give. Verdict: no structural +bias in what is written. Like `TESTS.md` section 3 this is a sanity check, not a proof of strength. + +## 3. Question 2: no short cut avoids the writes + +### 3.1 Inside a unit: the chain is dependent + +Slot s of lane l, touched d times in a unit, holds `w_d = rewrite(x_d, w_{d-1})`, `w_0 = fill`, with +`x_i = fold(dst_i, w_{i-1})`. `x_i` depends on the slot content before it, which depends on every earlier fold +value of that slot; and `dst_i` is the register state, which the earlier fold values entered. Test +`slot_is_replayable_from_its_fold_values`: a slot after 64 read-modify-writes is reproduced from the fill and the +64 fold values; dropping one diverges. So a chip cannot skip a write and still read the slot later. It has three +ways to hold a slot, all exact: + +| Store | Bytes per lane | Cost on a re-hit | +|---|---|---| +| Dense: every slot, 12 data bytes plus a valid bit | 12 x slots: 776 (64 slots), 3,104 (256), 24,832 (2,048) | one SRAM read | +| Sparse: only touched slots, 12 bytes plus a slot index | about 13 x distinct: 185 to 820 (table below) | one lookup | +| Implicit: only the fold values, 4 bytes plus a slot index per read-modify-write, replay on a re-hit | 5 x 8k: 80 (scr2), 160 (scr4), 320 (scr8) | d rewrites of 5 integer ops | + +The implicit store is smaller than the dense one whenever `slots > 8k / 3`: at scr4 above 10.7 slots, at scr8 +above 21.3. So "the smallest scratch at which keeping it implicitly is dearer than storing it" is 8k/3 slots per +lane, 2.7 to 5.3 KiB per warp at scr4 to scr8. Every size on the table, 32 KiB and above, is past it: a chip +keeps the scratch implicitly in 80 to 320 bytes per lane at any nominal size, and the replay cost is bounded by +the re-hit depth, which the next table measures. + +### 3.2 The re-hit rate at 64 and 256 slots (and at 2,048) + +Measured in the same test run (every read-modify-write of 196,608 hashes per class traced; a re-hit is a read of a +slot the same unit wrote earlier). Birthday: `distinct = S (1 - (1 - 1/S)^n)` for n uniform draws from S slots. + +| Class | S | n = RMW per hash | Distinct slots, birthday | Re-hits, birthday | Re-hit %, birthday | Re-hit %, measured | Max chain depth seen | Slot histogram against uniform | +|---|---|---|---|---|---|---|---|---| +| scr2k32 | 64 | 16 | 14.26 | 1.74 | 10.9 | 12.58 | 7 | chi2 z 22,023; hottest slot 2.74x, coldest 0.83x | +| scr4k32 | 64 | 32 | 25.33 | 6.67 | 20.8 | 21.47 | 8 | z 10,880; 1.87x, 0.92x | +| scr8k32 | 64 | 64 | 40.64 | 23.36 | 36.5 | 36.99 | 9 | z 7,587; 1.39x, 0.91x | +| scr2k128 | 256 | 16 | 15.54 | 0.46 | 2.9 | 3.84 | 5 | z 19,146; 5.10x, 0.82x | +| scr4k128 | 256 | 32 | 30.14 | 1.86 | 5.8 | 6.25 | 6 | z 9,632; 3.06x, 0.90x | +| scr8k128 | 256 | 64 | 56.72 | 7.28 | 11.4 | 11.89 | 6 | z 5,873; 1.98x, 0.90x | +| 1 MiB (not run) | 2,048 | 32 | 31.76 | 0.24 | 0.8 | | | | + +Two readings. First, the slot a read-modify-write addresses is the low 6 or 8 bits of a program register, and +those bits are not uniform: `or` sets them, `mul` clears them, so one slot of 256 is addressed 5.1 times as often +as the mean and the re-hit rate runs 2 to 33 percent above the birthday rate. For the dataset the same bias on the +low bits of a 28-bit address is harmless (it moves the read inside an item); for a 64-slot scratch it concentrates +the chain. Second, the chain depth is small: at scr4k32 the deepest slot in 196,608 hashes saw 8 earlier +read-modify-writes; a replay costs at most 8 x 5 integer operations, against about 1,170 for one dataset item. + +### 3.3 The live state is bounded by the read-modify-write count, not by the size + +The verifier evaluates one unit from nothing but (program, day, nonce group): `ScratchModel::new` per unit, +`verify.rs:296`. Every conforming GPU must therefore start every unit from the fill, which the tag does +(section 4). So no state crosses a unit boundary, and the state a unit can ever read back is what it wrote itself: +at most 8k slots per lane. The nominal size only sets how often those 8k writes land on the same slot (the table +above). The scratch's "memory" is 8k x 16 bytes per lane of touched slots, 256 bytes to 1 KiB at scr2 to scr8, +and a chip holds it implicitly in 80 to 320 bytes. + +The attack of rolling back or sharing scratch between units has nothing to take: a unit starts from the fill +whatever ran before it, so a chip that clears 64 valid bits per unit has rolled back, and nothing one unit wrote +is readable by another. The CPU verifier is that chip. + +### 3.4 The named chip, and what the scratch costs it + +The strongest chip the plan has priced (coordinator, 5 October 2026): the whole 256 MiB cache on the die, computing +every dataset item on the fly. Its cache SRAM, from `docs/analysis/sram-mirror.md` revision 2 (`ca2-analysis` +e6085c6), headline at shipped-product density / bit-cell lower bound, dollars per good die approximate: 164 / 83 +mm^2 and $30 / $13 at N7 (shipped density from AMD 3D V-Cache, 64 MB on 41 mm^2, Hot Chips 2021); 128 / 64 mm^2 and +$46 / $21 at N5, N3E and Intel 18A (TSMC N5 HD macro 31.8 Mib/mm^2 after assist overhead, SemiAnalysis, December +2022); 106 / 54 mm^2 and $56 / $26 at N2; with a 96 MB hot table 226 / 114 at N7, 175 / 89 at N5, 146 / 74 at N2. +The chip's cache cost in the table below is the N5 headline, 128 mm^2 and $46 per good die. It computes every item +through the mixer (`docs/analysis/m16-recompute-attacker-2026-10-05.md`: 128 items per hash, about 1,170 integer +operations per item, 150,000 per hash; at a 50 T op/s integer budget equal to a 5090's, approximate, 0.33 Ghash/s). +Against the measured version 2 rate of the RTX 5090, 139.7 MH/s (`docs/bench-log.md` M11, 4 October 2026), that is +2.4x before any fixed-function factor, 7x with the 3x the M16 analysis allows (approximate). + +Units in flight on that chip. It has no DRAM latency to cover: every one of its 1,024 cache reads per hash is an +on-die SRAM read. Its hash latency is the dependent chain: 128 items x (8 dependent SRAM reads plus 9 mixer +applications). At about 10 ns per on-die read and about 40 ns per 130-operation mixer on a 16-wide integer +pipeline at 2 GHz (both approximate), an item is about 0.4 us and a hash about 50 us; at 0.33 Ghash/s that is +about 17,000 hashes in flight, 530 units of 32 lanes. A tighter pipeline halves it. The GPU covers DRAM latency (40 to 48 ns row +cycle, MEMSYS 2018, more under load) with 130,560 lanes in flight at the measured occupancy (4,080 warps x 32), 348,160 at +full occupancy, that is 8 to 20 times more lanes than the chip needs. + +What the scratch costs that chip, per variant, with the arithmetic: + +Chip cache mirror: 128 mm^2, $46 per good die (N5 headline; 64 mm^2, $21 bit-cell lower bound). Chip scratch SRAM at +the same two densities (2.1 MB/mm^2 headline, 4.2 MB/mm^2 lower bound at N5): + +| Variant | Dataset loads per hash | Chip ops per hash | Chip rate at 50 T op/s | 5090 rate | Chip gain | Chip scratch SRAM at 17,000 lanes, implicit store | Same, dense 64-slot store | Dense store as mm^2, headline / lower bound (N5) | Share of the 256 MiB mirror (any density) | +|---|---|---|---|---|---|---|---|---|---| +| scr0 (control), 128 loads | 128 | 150,000 | 333 MH/s | 139.7 measured | 2.4x | 0 | 0 | 0 | 0 | +| 12.5% replaced (scr2) | 112 | 131,400 | 381 | 160 projected (128/112 x 139.7) | 2.4x | 1.4 MB | 13 MB | 6.2 / 3.1 mm^2 | 4.9% | +| 25% replaced (scr4) | 96 | 112,800 | 443 | 186 projected | 2.4x | 2.7 MB | 13 MB | 6.2 / 3.1 | 4.9% | +| 50% replaced (scr8) | 64 | 75,600 | 661 | 279 projected | 2.4x | 5.4 MB | 13 MB | 6.2 / 3.1 | 4.9% | +| 12.5% added (16 RMW beside 128 loads) | 128 | 150,200 | 333 | 139.7 or below | 2.4x or more | 1.4 MB | 13 MB | 6.2 / 3.1 | 4.9% | +| 25% added | 128 | 150,400 | 332 | 139.7 or below | 2.4x or more | 2.7 MB | 13 MB | 6.2 / 3.1 | 4.9% | +| 50% added | 128 | 150,800 | 332 | 139.7 or below | 2.4x or more | 5.4 MB | 13 MB | 6.2 / 3.1 | 4.9% | +| 256-slot dense store (128 KiB class), any share | | | | | | | 53 MB | 25 / 12.6 | 20% | + +How the rows are computed: a read-modify-write costs the chip about 12 integer operations (fold and rewrite) and +one SRAM access; replacing a load removes an item derivation (1,170 operations); the 5090's rate for a replaced +load is projected from the measured distinct-load bound (the card's rate tracks distinct dataset loads per hash, +`docs/bench-log.md` 3 October, 23.7 G loads/s at 1 GiB; the readwidth agent's M5 Max measurement of the night, +relayed by the coordinator, shows the same: 27.7 MH/s at v2 to 29.4-31.7 at 25 percent replaced and 44.4-49.1 at 50 +percent, 32 KiB per warp). The scratch SRAM is 17,000 lanes x 80 to 320 bytes (implicit) or x 776 bytes (dense at +64 slots) or x 3,104 bytes (dense at 256 slots); its share of the mirror is a ratio of bytes, 4.9 or 20 percent, +whichever density is used for both; the implicit store (the chip's cheaper choice at every size, section 3.1) is +0.5 to 2 percent. The chip's gain is set by operations per dataset item and the GPU's distinct-load bound, and the +scratch touches neither. + +Plain answer to the coordinator's question: no read-modify-write share under the 6 GB cap, replaced or added, +brings the named chip under 2x. The share would be chosen as the smallest at which the chip falls under 1.5x, and +there is none: the gain is 2.4x at 0, 12.5, 25 and 50 percent, 32 or 128 KiB. This changes nothing about the public +claim that layer 3 would have changed: the claim must rest on the mixer, not on the scratch. + +The lever that does move that chip, from the M16 table, beside it: + +| Mixer cost multiplier | Chip ops per hash | Chip rate | Gain against 139.7 MH/s, no fixed-function factor | With a 3x factor (approximate) | CPU verify per warp (M16 table, scaled from 0.41 to 1.2 ms) | 5090 daily dataset build | +|---|---|---|---|---|---|---| +| x1 (today) | 150,000 | 333 MH/s | 2.4x | 7.2x | 0.4 to 1.2 ms | 13.4 ms | +| x2 | 300,000 | 167 | 1.2x | 3.6x | 0.8 to 2.4 ms | 27 ms | +| x4 | 600,000 | 83 | 0.6x | 1.8x | 1.6 to 4.8 ms | 54 ms | +| x8 | 1,200,000 | 42 | 0.3x | 0.9x | 3.3 to 9.6 ms | 107 ms | + +The mixer multiplier leaves the honest hash rate untouched (the miner pays the mixer once a day), costs the chip +linearly, and is bounded by the 10 ms verification gate (x8 is at the gate's edge on this core, and the 2019-class +core of O-1.14 is unmeasured). The scratch costs the honest GPU a measured share of its rate when it spills the +cache and nothing when it does not, and costs the chip a few megabytes. The comparison is not close. + +### 3.5 Where the GPU's writes would cost DRAM latency, and why that does not help + +The GPU's hot scratch footprint is not the nominal size either: it is the slots in-flight units have touched, +about `warps x 32 lanes x distinct slots x 16 bytes` (x 2 at a 32-byte sector, approximate): on the 5090 at 4,080 +resident warps and scr4, 25.3 slots at 64 or 30.1 at 256, 53 to 63 MB of slots, 100 to 125 MB in sectors, around +the card's 96 MiB L2 (`docs/bench-log.md`, 3 October). The readwidth agent's M5 Max rows (coordinator's message: +the rate rises with the share at 32 and 128 KiB) show the scratch sitting in that chip's caches at 4,096 warps. +To push the writes to DRAM latency the hot footprint must pass the last-level cache at the resident count: +`96 MiB / 4,080 warps = 24 KiB per warp`, which at 512 bytes of touched slots per lane per read-modify-write slot +means `8k x 512 B > 24 KiB`, k above 6 (above 48 read-modify-writes per hash) at ANY nominal size on the table, or +a higher resident count. That fits the 6 GB cap (it is the hot set, not the arena, that matters), and it costs +the honest miner a DRAM-latency read-modify-write per slot (a DRAM row cycle is 40 to 48 ns across DDR4, GDDR5 and +HBM2, Li, Reddy and Jacob, MEMSYS 2018; the loaded latency a GPU kernel sees is higher, approximate; DRAM latency +improved 1.3x in two decades while bandwidth improved 20x, Chang 2017, so no memory technology an attacker could +buy removes it, and no shipped mining chip has used HBM or stacked memory) while the named chip still keeps the same +hot set in a few megabytes of SRAM at 8 to 20 times fewer lanes in flight. The write path cannot be made to cost the +chip more than the GPU, because the GPU must keep 8 to 20 times more of it live. + +## 4. Question 3: the verifier's one-warp simulation is exact + +### 4.1 Lazy fill on both sides + +The CPU initialises lazily with a written bit per (lane, slot), one model per unit. The GPU initialises lazily with +a 32-bit tag in word 0 of each 16-byte slot: a slot whose tag equals the unit's tag reads as written, any other +reads as the fill (`emit.rs:143-157`). There is no explicit fill and no reset between units of a persistent warp +(`emit.rs:159-175`: the loop over `g_` keeps the arena). The two agree if and only if, when a unit first touches a +slot, that slot does not already carry the unit's tag. That is: + +1. Tags are unique over the life of the arena's contents (`tag = salt + g_`, `salt` the host's running counter). +2. The arena holds no word equal to a live tag in a slot's tag position before the unit writes it. + +### 4.2 The attacks (the bug classes) + +| Case | What happens | Today | +|---|---|---| +| Recycled allocation | A fresh process starts `salt` at 1 (`packbench.swift:144`, `host.c:1027`). If the driver hands back the previous process's arena with its contents (Metal, CUDA and OpenCL do not promise zeroed memory, approximate), slots tagged 1..N from the old run match the new run's first units exactly, and those units read stale words instead of the fill: a CPU mismatch on every colliding slot. | not guarded; passes on this Mac because fresh allocations read as zero in practice and tag 0 is never issued (luck, not contract) | +| Tag counter wrap | `salt` is 32 bits and advances by units per launch. A 5090 at 139.7 MH/s runs 4.37 M units/s, 2^32 units in 984 s: the counter wraps every 16.4 minutes on one card (81.8 minutes on the M5 Max at 28 MH/s). After the wrap a slot whose LAST writer carried the repeated tag reads as written. With 10,880 arenas each slot is rewritten about 395,000 times between two uses of one tag (at 64 slots a unit leaves a slot untouched with probability 0.60; 0.60^395,000 is 0), so on a full card the wrap is harmless in practice; on a one-warp launch repeated 2^32 times it is not. | not guarded | +| Tag 0 on zeroed memory | A host that starts `salt` at 0 gives unit 0 the tag 0, which a zeroed arena carries in every slot: unit 0 reads zeros for every first touch. | both hosts start at 1; nothing in the pack says they must | +| `groups` not a multiple of the warp count | Warps run different trip counts; the OpenCL local-memory exchange path carries a barrier inside the loop (spec 1.9), so a short warp hangs or desynchronises. | `packbench` refuses it; `host.c` rounds the batch | + +### 4.3 The host contract that makes the simulation exact + +A host of a scratch class MUST: allocate the arena as `warps x 32 x words_per_lane` words and zero it; issue tags +from a 32-bit counter that starts at 1 and advances by the unit count of every launch; zero the arena again before +any launch whose tags would pass 2^32 - 1 (tag 0 is never issued); launch `groups` as a multiple of the warp count. +The zeroing costs one memset of the arena (340 MiB at 32 KiB x 10,880 warps) every 2^32 units, 16 minutes on a +5090. This is the class fix for all four rows: with it the GPU's tag test and the CPU's written bit are the same +predicate. + +### 4.4 The tests (Metal, M5 Max, 5 October 2026) + +Two consecutive units on one persistent warp and the wrap inside a launch (`packbench --warps 1`, +`--batch-base 4294967040`, the option added on this branch); the hand-built edge programs that force every +read-modify-write of a hash onto one slot (so two consecutive units on one arena collide on every slot); the +deliberate breaks. Results in section 7.2. On the CPU, the same edge programs against an independent hand model +(a second interpreter with its own slot store, `tests/scratch.rs`): 56 of 56 cases match, and the hand model with +its rewrite words swapped mismatches on every case (the comparison has teeth). + +## 5. Question 4: the attack surface of the writes + +| Surface | Argument | Test | +|---|---|---| +| Out of bounds | `s_ = rN & (slots - 1)`, so `s_ < slots`; the lane's arena is `(warp_ x 32 + lane) x 4 x slots` words from the base, the access is `arena + 4 x s_ + 0..3`, the largest index is `warps x 32 x 4 x slots - 1`, the host's allocation. The emitter has one scratch template (`emit.rs:143`) and it masks. | `scr_packs_regenerate_and_pass_the_static_scratch_check`: 42 of 42 emitted kernels (7 scr packs x 6 files, the OpenCL bound file carrying two kernels) regenerate byte for byte from program.json and pass the text check: k masked slot definitions with the class mask, k tagged stores, 3k fill calls, one arena definition with the class stride, one tag definition, no `scratch[`; six deliberate breaks caught (section 8) | +| Aliasing between lanes | Lane-major: lane l of warp w owns words `[(32w + l) x 4S, (32w + l + 1) x 4S)`; two (w, l) pairs give disjoint ranges. Inside the range a slot is 4 words at `4 x s_`, so two slots of one lane are disjoint too. | the `lanevar` edge program: one init-dependent slot per lane, 32 lanes at 64 slots share slots in pairs by the birthday bound; any cross-lane aliasing would change the fold; 128 of 128 lanes on Metal (section 7.2) | +| Wave64 (two logical units in one hardware wave) | `warp_ = tid >> 5`, so the two halves get `warp_ = 2w` and `2w + 1`, two arenas; `gbase` and `tag` are per `g_`, per half. | not run on wave64 hardware (the OpenCL emulator's persistent launch is on the readwidth commit; unverified here) | +| Determinism: alignment | A slot is 16 bytes at byte offset `16 x (lane_base + s_)`; the arena base is the buffer base: Metal, CUDA and OpenCL allocations are at least 128-byte aligned (CUDA 256, OpenCL `CL_DEVICE_MEM_BASE_ADDR_ALIGN` at least the largest built-in type, approximate from memory), so every 16-byte vector access is aligned. | Metal: every run of section 7 | +| Determinism: ordering | A lane's two read-modify-writes of the same slot in one hash are a load and a store, then a load and a store, from one thread to one address: program order within a thread holds in every model. No other thread touches the slot (aliasing row), so no atomics, fences or barriers are needed and none are emitted. | `slot0` and `sixteen` edge programs: 64 and 128 dependent read-modify-writes on one slot per lane per hash, standalone and as the second unit on a warp | +| Determinism: vendors | The statement is integer only: xor, rotate by immediate, multiply, add, a 16-byte load and store. Bit-exact across Metal, CUDA and OpenCL by construction; measured only on Metal here. | Metal; CUDA and OpenCL runs are PC jobs (not mine tonight) | +| 32-bit nonce wrap | `gbase = baseNonce + g_ x 32` and `nonce = baseNonce + gid` wrap in 32-bit arithmetic; `scr_fill(gbase + lane)` wraps like the CPU's `base.wrapping_add(lane)`; `out[gid]` indexes by launch position, not by nonce. An aligned unit never straddles 2^32 (spec 1.9), so the wrap case is a launch whose unit SEQUENCE crosses it. | `packbench --batch-base 4294967040 --batch-log2 9`: 16 units from 0xffffff00, the ninth at gbase 0; fingerprint identical at 1 and 4 warps (section 7.2); every fuzz pack runs that launch | + +## 6. Question 5: what a vector for the scratch class must carry + +Before a scratch pack can be a conformance vector (plan step 4, "only then a vector"), it must carry, beyond what +`igneum-program-pack-3` carries today: + +1. The class in the program id and the pack (`scrk`: it is, `program_id_class`, `generator.rs:400-412`) + and the geometry (slots per lane, words per lane, bytes per warp: it is, `program.h`). +2. The fill and the rewrite as text (it is, `program.json` "scratch"). +3. The host contract of section 4.3 as text in `program.h` and `program.json`: tag counter from 1, zero at + allocation and at the wrap, `groups` a multiple of the warp count. Not there today. +4. Vectors that exercise the tag path, which the three standard units do not reliably: two consecutive units on + one warp (bases 0 and 32 in one one-warp launch) for a program whose consecutive units collide on a slot. At + 64 slots any generated program collides (25 touched of 64 per unit; the broken-tag run of section 8 was caught by + base 1,000,000, a warp's 16th unit, and NOT by a two-unit launch whose vectors lack base 32). At 2,048 slots two + consecutive units share a touched slot with probability about 0.4 (32 x 32 / 2,048 expected overlaps = 0.5), so + the standard vectors would miss a broken tag at the 1 MiB size with probability about 0.6 per unit pair. The + edge programs `slot0` and `sixteen` collide on every slot at every size: a vector set should carry one. +5. A unit in the top 256 nonces with the launch crossing 2^32 (`--batch-base` near the top, at least two warps). +6. The batch fingerprint declared independent of the warp count (`8c07620f4d9adefd` for scr4k32 at 2^12 nonces + from base 0 at 1, 2 and 128 warps, section 7.2): unit independence is the property the per-unit reset gives, and + a fingerprint that moved with the warp count would mean a unit read another unit's slot. + +## 7. Tests and results + +### 7.1 CPU (`igneum-pow/tests/scratch.rs`, `cargo test -j4 --test scratch`, M5 Max, 5 October 2026, 3.6 s) + +| Test | What | Result | +|---|---|---| +| `rewrite_is_a_bijection_of_the_fold_value` | 16 random slot contents x 2^16 consecutive fold values, each written word distinct; the three inversions | pass | +| `fill_is_a_bijection_of_the_nonce` | 7 slots x 3 words x 2^16 nonces distinct; lane and base interchange; wrap | pass | +| `written_words_unbiased_and_rehit_rates` | 6 classes x 3 seeds x 2^11 units, every rewrite traced: bias within 6 sigma (worst 3.63), re-hit rate within 0.9x to 2x of birthday, slot histogram, depth histogram | pass (tables of sections 2.3 and 3.2) | +| `edge_programs_match_the_hand_model` | 7 edge programs x 2 geometries x 4 bases (0, 32, 0x7ffffff0, 0xffffffe0) against an independent hand model; the slots driven and the re-hit counts as built; the mutated hand model mismatches | 56 of 56 pass, 56 of 56 teeth | +| `scr_packs_regenerate_and_pass_the_static_scratch_check` | 7 scr packs: program and program id from program.json, 6 kernel texts byte for byte, static scratch check on all 42, the pack's vectors from the CPU; six deliberate breaks caught | pass | +| `fuzz_scr_programs_cpu` | 200 generated programs over the six classes, generator contract and acceptance on every one, 4 units each (one in 0..224, one around 2^31, one in the top 256 nonces, one uniform), traced run equal to the untraced run, every slot inside the lane; writes the 214 packs for Metal with `IGNEUM_SCRATCH_PACKS_OUT` | pass; 200 of 200 have a unit in the top 256 | +| `slot_is_replayable_from_its_fold_values` | 64 dependent read-modify-writes replayed from the fill and the fold values; one dropped diverges | pass | + +The rest of the crate: 33 of 34 lib tests and all pack tests pass; `verify::tests::fold_and_wide_fetch` fails on the +readwidth tip itself (`verify.rs:508`, `k as u32 * 0x9E37_79B1` overflows under the test profile's overflow +checks; the readwidth agent's test, reported to its owner, not touched here). + +### 7.2 Metal (`proto-metal/packbench` built from this branch, M5 Max, 5 October 2026, under `with-lock.sh run`) + +| Run | Launch | Expected | Result | +|---|---|---|---| +| scr4k32, standard pack | 2,048 warps, 2^24 nonces, 1 batch | 3 of 3 standalone, 3 of 3 in batch | PASS, fingerprint `3d1af881bd978fb9`; 1.8 s wall for the whole run (compile, cache, 1 GiB build, vectors, batch) | +| scr4k32, warp-count independence | 2^12 nonces (128 units) at 1, 2 and 128 warps | one fingerprint | `8c07620f4d9adefd` at all three, PASS | +| scr4k32, wrap inside the launch | 512 nonces from 0xffffff00 at 1 and 4 warps | one fingerprint, the base-0 vector inside the window after the wrap | `8e9e233234d3a297` at both, in-batch 1 of 1, PASS | +| scr4k32, broken tag (`tag = salt`), standard vectors | 2,048 warps, 2^24 | the base-1,000,000 vector (warp 530's 16th unit) fails | standalone 3 of 3, in batch 2 of 3, overall FAIL (caught) | +| scr4k32, broken tag, two units on one warp | 1 warp, 2^6 | nothing to catch it: the standard vectors have no base 32 | standalone 3 of 3, in batch 1 of 1, PASS (missed: the point of section 6 item 4) | +| 14 edge packs (7 programs x 32 and 128 KiB), run A | 1 warp, 2^6 (units at bases 0 and 32 on one arena) | 4 of 4 standalone, 2 of 2 in batch each | 14 of 14 PASS (56 of 56 standalone units, 28 of 28 in batch) | +| 14 edge packs, run B | 1 warp, 2^9 from 0xffffff00 (16 units on one arena, the wrap inside) | 4 of 4 standalone, 3 of 3 in batch each | 14 of 14 PASS (56 of 56, 42 of 42) | +| edge `slot0` at 32 and 128 KiB, broken tag (`tag = salt`) | 1 warp, 2^6 | standalone 4 of 4, in batch 1 of 2, FAIL | as expected at both geometries: the second unit read the first's slot 0 and FAILED; the standalone units passed | +| edge `slot0` at 32 KiB, broken lazy fill (`m_` forced to all ones: a first touch reads the stale words) | 1 warp, 2^6 | standalone fails | 0 of 4 standalone, 0 of 2 in batch, FAIL (lane 0 of base 0: GPU `64b49aeb987dae69`, expected `9ff3a2021f66b5be`) | +| 200 fuzz packs (scr2k32 29, scr4k32 26, scr8k32 42, scr2k128 32, scr4k128 26, scr8k128 45; datasets 64 MiB, 256 MiB, 1 GiB) | 2 warps, 2^9 from 0xffffff00 (8 units per warp, the wrap inside) | 4 of 4 standalone, 2 of 2 in the window, 200 of 200 PASS | 200 of 200 PASS: 800 of 800 standalone units (25,600 hashes), 400 of 400 in batch; 91 s for the 200 runs | + +Totals on Metal: 228 of 228 runs PASS where a pass was expected, 3 of 3 FAIL where a failure was built in. + +## 8. Deliberate breaks (the watcher rule) + +| Break | Where | Caught by | Evidence | +|---|---|---|---| +| One slot mask dropped (Metal) | copy of scr4k32 `program.metal` | static check: "masked slot followed by the load: 3, expected 4" | test output | +| Mask 63 changed to 127 on every RMW (Metal) | same | "masked slot followed by the load: 0, expected 4" | test output | +| Arena stride 256 changed to 128 words (Metal) | same | "arena definition: 0, expected 1" | test output | +| A stray `arena[0]` and `scratch[1]` access (Metal) | same | "arena mentions: 10, expected 9; direct scratch indexing: 1, expected 0" | test output | +| One slot mask dropped (OpenCL, CUDA) | copies of scr4k32 `kernel.cl`, `kernel.cu` | "masked slot followed by the load: 3, expected 4" | test output | +| Wrong class geometry or RMW count or kernel count passed against a right text | the same text | the check fails | test output | +| `tag = salt` (every unit of a launch shares the tag) | copy of scr4k32 `program.metal`, on the GPU | the base-1,000,000 vector in a 2,048-warp batch | `vectors standalone 3/3, in batch 2/3`, overall FAIL | +| the same on the `slot0` edge pack, two units on one warp | on the GPU | in-batch 1 of 2 | bench-log entry | +| lazy fill broken (`m_` all ones) | copy of the `slot0` edge pack, on the GPU | standalone vectors | bench-log entry | +| The hand model's rewrite words swapped | `tests/scratch.rs` | every edge case mismatches | 56 of 56 | + +The out-of-bounds break (mask dropped) was not run on the GPU on purpose: Metal does not bounds-check device +buffers (`TESTS.md` section 5), so a run would read another lane's or another buffer's words and "did not crash" would +prove nothing. The static check is the guard, as it is for the dataset mask. + +## 9. What is unverified + +1. CUDA and OpenCL runs of the scratch packs on NVIDIA and AMD (PC jobs, reserved for the readwidth agent tonight); + the 5090's rate per variant, so the "projected" column of section 3.4 is the distinct-load bound, not a + measurement. Wave64 hardware for the two-arena argument. +2. The chip-side latency figures of section 3.4 (10 ns SRAM read, 40 ns mixer) are approximate; the conclusion + does not depend on them: at ten times the in-flight count the scratch is still under a sixth of the mirror. +3. The recycled-allocation case was not reproduced (it needs a driver that hands back live contents); the argument + is that nothing forbids it and the contract of 4.3 removes it. +4. The slot-bias finding (section 3.2) was measured on three seeds per class; the hottest-slot ratio will vary by + program. +5. `verify::tests::fold_and_wide_fetch` on the readwidth tip (section 7.1). + +## 10. Recommendation + +1. Layer 3 is sound as a construct: the written words are uniform, the chain inside a unit has no short cut, the + kernels cannot write out of bounds, and with the host contract of section 4.3 the CPU's one-warp simulation is + exact (14 edge packs, 200 fuzz packs, the wrap, consecutive units on one arena, on Metal). +2. Layer 3 is not sound as a chip-resistance layer, at the capped size or at any size: CPU verification resets the + scratch per unit, so its live state is 8k slots per lane whatever the arena, a chip keeps it implicitly in 80 to + 320 bytes per lane, and the named chip (on-die cache mirror plus recompute) keeps its whole scratch in 1.4 to + 13 MB of SRAM at 530 units in flight, 3 to 5 percent of its mirror. Its gain stays at 2.4x (7x with a 3x + fixed-function factor, approximate) at 0, 12.5, 25 and 50 percent, replaced or added, 32 or 128 KiB. No share + under the 6 GB cap brings it under 2x. +3. Taking the read-modify-writes from the 16 dataset loads makes the hash less memory-hard for everyone: the GPU + measured faster at every share on the M5 Max (readwidth rows), and the chip's operations per hash fall with the + loads. If a scratch is kept at all, add it beside the 128 loads. There is no reason found here to keep one. +4. The lever that moves the named chip is the M16 mixer multiplier: x2 to 1.2x, x4 to 0.6x against the measured + 5090 rate, at 0.8 to 4.8 ms of verification per warp against the 10 ms gate. Decision 2 should price that + against the gate on the 2019-class core (O-1.14) rather than layer 3. +5. If Josh keeps layer 3 for a reason outside this analysis: ship the host contract in the pack, add the four + vector items of section 6 (consecutive units with a forced collision, the wrap launch, the warp-count-independent + fingerprint, the contract text), and run the CUDA and OpenCL twins of section 7.2 on the PCs before the class + becomes a genesis rule. diff --git a/docs/analysis/sram-mirror.md b/docs/analysis/sram-mirror.md new file mode 100644 index 000000000..e1798fa88 --- /dev/null +++ b/docs/analysis/sram-mirror.md @@ -0,0 +1,283 @@ +# Layer 6: the SRAM mirror of the cache against published SRAM density, year 0 to 10 + +5 October 2026 (night), Counter ASIC 2.0 (`docs/plans/counter-asic-2.md`, layer 6), branch `ca2-analysis`. Every figure +below is either cited (paper, vendor document, URL, date) or labelled approximate. Nothing here is a measurement of a +chip. Numbers in this file were computed with the arithmetic shown; the script is in section 10. + +Revision 2 (same night): the first draft priced the mirror from bit-cell area times a 0.70 array factor. The +coordinator's chip-economics research (sources below) showed that shipped cache-only dies land at about half that +density once assist circuits, redundancy, TSVs, power and test are in. Every table now carries two columns: the +shipped-product density as the headline and the bit-cell figure as the lower bound. The conclusion did not move; the +cost per die rose 2 to 3x. + +## 1. The question + +The lottery hash derives every dataset item from a 256 MiB cache (spec 01 sections 1.5 and 1.8). A chip that holds the +cache in on-die SRAM can recompute items instead of reading the dataset (ledger M16, the recompute attacker). Layer 6 +asks whether the cache size, as the specification schedules it, keeps that SRAM mirror unaffordable for ten years of +the genesis schedule, and if not what growth rule would. + +Two things also sit in a chip's SRAM budget if it mirrors the full read-only working set: the layer 5 hot table (32, +64 or 96 MB, a class parameter on `readwidth` b970dda, coordinator's note of 5 October) beside the 256 MiB cache. The +per-warp scratch of layer 3 (32 or 128 KB per warp, written, not read-only) is not mirrorable and is left out of the +mirror; it is counted in the 6 GB working-set budget in section 7. + +## 2. What the specification schedules for the cache + +| Quantity | Rule | Where | +|---|---|---| +| Dataset | 2 GiB at genesis plus 0.5 GiB per year (`N_d` grows about 23 KiB per day) | spec 01 section 1.13.3, Designed | +| Cache | 256 MiB, "prototype value, to be fixed at gate 1"; the rule that fixes it: "the cache must exceed the largest on-chip cache of any card that mines, and 96 MiB of L2 on the 5090 is the figure to beat" | spec 01 sections 1.5 and 1.16 | +| Cache growth | None. No section of `docs/spec/` grows the cache (grep of `docs/spec` for cache growth, schedule, doubling: only the dataset rule of 1.13.3 and the README's "growth" word, which refers to it) | this analysis, 5 October 2026 | + +So the plan's layer 6 row ("already in the design; confirm the schedule") is half right: dataset growth is in the +design, cache growth is not. The cache is flat at 256 MiB for every year of the schedule as the spec stands. M16's +closing line names the rule the cache should get ("exceeds what one die can hold, and grows") as a gate 1 decision +that has not been taken. + +## 3. SRAM density, cited: bit cells per node and shipped cache dies + +### 3.1 Bit cells + +| Node (vendor) | HD 6T bit cell, um^2 | Raw density, Mbit/mm^2 (1/cell) | Year of volume (approximate) | Source | +|---|---|---|---|---| +| N7 (TSMC) | 0.027 | 37.0 | 2018 | WikiChip, "TSMC Details 5 nm" (ISSCC/IEDM disclosures), https://fuse.wikichip.org/news/3398/tsmc-details-5-nm/ | +| N5 (TSMC) | 0.021 | 47.6 | 2020 | same (two N5 cells: HD 0.021, HP 0.025) | +| N3B (TSMC) | 0.0199 | 50.3 | 2022 to 2023 | WikiChip, "IEDM 2022: Did We Just Witness The Death Of SRAM?", https://fuse.wikichip.org/news/7343/iedm-2022-did-we-just-witness-the-death-of-sram/ (TSMC's IEDM 2022 N3 paper) | +| N3E (TSMC) | 0.021 | 47.6 | 2023 | same; Tom's Hardware, "TSMC's 3nm Node: No SRAM Scaling", https://www.tomshardware.com/news/no-sram-scaling-implies-on-more-expensive-cpus-and-gpus | +| N2 (TSMC) | 0.0175 | 57.1 | 2025 to 2026 | TSMC at IEDM 2024, reported by Tom's Hardware, https://www.tomshardware.com/tech-industry/tsmc-shares-deep-dive-details-about-its-cutting-edge-2nm-process-node-at-iedm-2024-35-percent-less-power-or-15-percent-more-performance ; ISSCC 2025 paper "A 38.1Mb/mm2 SRAM in a 2nm-CMOS-Nanosheet Technology", https://research.tsmc.com/page/memory/4.html | +| Intel 18A | 0.021 | 47.6 | 2025 to 2026 | ISSCC 2025 paper 29.2, "A 0.021 um^2 High-Density SRAM in Intel 18A RibbonFET Technology with PowerVia", https://www.researchgate.net/publication/389644177 ; IEEE Spectrum 26 Feb 2025, https://spectrum.ieee.org/sram-intel-tsmc | +| Samsung SF3 / SF2 | not disclosed as a bit cell area in anything found tonight (Samsung's ISSCC papers give assist circuits and macro figures, not the HD cell) | | | search of ISSCC 2021 to 2025 coverage, 5 October 2026; left out of the tables | + +The stall. N3B's cell is 5% smaller than N5's and N3E's is the same size as N5's (0.021 um^2 both): zero SRAM +scaling from N5 to N3E (WikiChip IEDM 2022 article above; Tom's Hardware above; SemiAnalysis "TSMC's 3nm Conundrum", +https://newsletter.semianalysis.com/p/tsmcs-3nm-conundrum-does-it-even). N2's nanosheet cell recovers 17% (0.021 to +0.0175 um^2). So across 2020 to 2026 the HD bit cell shrank once, by 17%. + +Macro density from the bit cell. WikiChip's and SemiAnalysis's convention is bit-cell density times about 0.70 for +the assist and periphery overhead (SemiAnalysis, December 2022: TSMC N5 HD SRAM macro 31.8 Mib/mm^2 after about 30% +assist overhead; WikiChip's 31.8 Mib/mm^2 for the 0.021 um^2 cell is the same arithmetic). The two ISSCC 2025 macros +bracket it: TSMC N2 38.1 Mb/mm^2 at a 0.0175 um^2 cell is 67%; Intel 18A 38.1 Mb/mm^2 array density and 34.3 Mb/mm^2 +for the volume macro at a 0.021 um^2 cell are 80% and 72%. That is a macro on a test chip. It is the LOWER BOUND on +die area, not the die. + +### 3.2 Shipped cache dies (what a whole die of SRAM really holds) + +| Product | SRAM | Die | Node | MB per mm^2 | Source | +|---|---|---|---|---|---| +| AMD 3D V-Cache (Zen 3 SRAM chiplet) | 64 MB | 41 mm^2 | TSMC 7 nm | 1.56 | AMD at Hot Chips 33, reported by Tom's Hardware, August 2021, https://www.tomshardware.com/news/amd-unveils-more-ryzen-3d-packaging-and-v-cache-details-at-hot-chips ("the 3D V-Cache SRAM measures 41 mm^2", "64 MB of 7 nm SRAM"); the densest cache-only die that has shipped | +| Graphcore GC200 (with compute) | 900 MB | 823 mm^2 | 7 nm | 1.09 | coordinator's chip-economics research, 5 October 2026 (vendor figures) | +| Groq TSP | 220 MB | 725 mm^2 | 14 nm | 0.30 | same | + +The V-Cache die is a pure SRAM die with its TSVs, redundancy, test and power: 1.56 MB/mm^2 at N7 against the bit-cell +figure 37.0 Mbit/mm^2 = 4.6 MB/mm^2 and the 0.70-macro figure 3.2 MB/mm^2. The shipped die is 0.48 of the macro +figure. The headline column below scales the V-Cache density to other nodes by the bit-cell ratio (0.027 / cell), an +approximation that assumes the periphery and TSV overheads scale with the cell, which they do not fully (so the +headline column is itself slightly optimistic for the attacker at N5 and below). + +### 3.3 GPU on-die SRAM, the reticle, wafer prices + +GPU on-die SRAM for scale: the RTX 5090 carries 96 MB of L2 (98,304 KB) on a 750 mm^2 TSMC 4N die with 92.2 billion +transistors; the full GB202 has 128 MB; the RTX 4090 had 72 MB and the RTX 3090 6 MB (NVIDIA, "RTX Blackwell GPU +Architecture" whitepaper v1.1, appendix table "L2 Cache Size", https://images.nvidia.com/aem-dam/Solutions/geforce/blackwell/nvidia-rtx-blackwell-gpu-architecture.pdf). +At the V-Cache density scaled to N5 (2.0 MB/mm^2) that L2 is about 48 mm^2 of the 750 (6%), approximate. The +RX 9070 XT carries 64 MB of Infinity Cache plus 8 MB of L2 (vendor figures, approximate, bench-log "the 9070 XT on the +eGPU"). + +Reticle: the EUV field is 26 x 33 mm = 858 mm^2, about 830 mm^2 usable after scribe lanes (SemiAnalysis, "Die Size +And Reticle Conundrum", https://newsletter.semianalysis.com/p/die-size-and-reticle-conundrum-cost ; WikiChip "Mask", +https://en.wikichip.org/wiki/mask). The 5090's 750 mm^2 is 90% of it. + +Wafer prices (approximate; TSMC publishes none, every figure is supply-chain reporting): N7 about $9,500, N5 and N3 +about $20,000 (Silicon Analysts, "Wafer Pricing by Node", September 2026, https://siliconanalysts.com/data/wafer-pricing); +N2 about $30,000 (Tom's Hardware, https://www.tomshardware.com/tech-industry/semiconductors/tsmc-could-charge-up-to-usd45-000-for-1-6nm-wafers-rumors-allege-a-50-percent-increase-in-pricing-over-prior-gen-wafers). + +## 4. Die area to mirror the cache, per node, two columns + +Headline = V-Cache density (41 mm^2 per 64 MiB at N7) scaled by the bit-cell ratio. Lower bound = bits / (raw +density x 0.70). Columns: the 256 MiB cache alone, the cache plus the 96 MB hot table of layer 5 (as MiB), and the +larger caches of the options in section 7. Area in mm^2; a figure over 830 is split into the dies shown. + +| Node | 256 MiB, headline | 256 MiB, lower bound | 256 + 96, headline | 256 + 96, lower bound | 512 MiB, headline / lower | 1 GiB, headline / lower | 4 GiB, headline / lower | +|---|---|---|---|---|---|---|---| +| N7 | 164 | 83 | 226 | 114 | 328 / 166 | 656 / 331 | 2,624 (4 dies) / 1,325 (2 dies) | +| N5 | 128 | 64 | 175 | 89 | 255 / 129 | 510 / 258 | 2,041 (3 dies) / 1,031 (2 dies) | +| N3B | 121 | 61 | 166 | 84 | 242 / 122 | 483 / 244 | 1,934 (3 dies) / 977 (2 dies) | +| N3E, Intel 18A | 128 | 64 | 175 | 89 | 255 / 129 | 510 / 258 | 2,041 (3 dies) / 1,031 (2 dies) | +| N2 | 106 | 54 | 146 | 74 | 213 / 107 | 425 / 215 | 1,701 (3 dies) / 859 (2 dies) | + +One reticle (830 mm^2) holds, at the headline density, 1.3 GiB of SRAM at N7, 1.6 GiB at N5, N3E and 18A, 1.9 GiB at +N2 (lower-bound column: 2.5, 3.2, 3.9 GiB). + +Against the figures the ledger carries: M16's "100 to 300 mm^2" (low end from a 0.02 um^2 cell with overhead, high +end from wafer-scale parts at about 1 MB per mm^2) brackets the headline 106 to 164 mm^2 well; the plan's "about +45 mm^2 at a leading node" is below even the lower bound and should be read as the bit-cell area with no overhead. +The right figures for the ledger are 106 to 164 mm^2 (shipped density) with 54 to 83 mm^2 as the floor. + +## 5. Cost per good die, two columns + +Dies per 300 mm wafer by the usual approximation pi x 150^2 / A minus the edge term pi x 300 / sqrt(2A); yield by +Poisson exp(-A x D0) with D0 = 0.1 defects per cm^2 (an assumption, approximate; SRAM arrays carry redundancy so +real yield is higher, which lowers these costs). Cost per good die = wafer price / (dies x yield). Packaging, test, +the logic beside the SRAM and the design (masks at N5 and below run into the tens of millions of dollars, +approximate) are not in these numbers; they are per-die silicon only. Headline / lower bound in each cell. + +| Node, wafer price | 256 MiB | 256 + 96 MiB | 1 GiB | 4 GiB | +|---|---|---|---|---| +| N7, $9,500 | 164 mm^2, 379 dies, yield 0.85: $30 / $13 | $44 / $19 | $224 / $75 | $896 (4 dies) / $456 (2 dies) | +| N5, $20,000 | 128 mm^2, 495 dies, 0.88: $46 / $21 | $68 / $30 | $306 / $111 | $1,512 (3 dies) / $621 (2 dies) | +| N3B, $20,000 | 121 mm^2, 524 dies, 0.89: $43 / $20 | $63 / $28 | $280 / $103 | $1,371 (3 dies) / $569 (2 dies) | +| N3E, 18A, $20,000 | $46 / $21 | $68 / $30 | $306 / $111 | $1,512 / $621 | +| N2, $30,000 | 106 mm^2, 600 dies, 0.90: $56 / $26 | $81 / $37 | $343 / $131 | $1,641 (3 dies) / $696 (2 dies) | + +Reading. The silicon for a 256 MiB mirror is $30 to $56 per die at shipped density (2 to 3x the first draft's +figure), under $90 with the hot table. A funded chip programme pays that without noticing: it was never the SRAM +that priced the recompute attacker out, and the plan's premise for layer 6 ("the SRAM mirror stays unaffordable") +does not hold for the cache as a mirror and did not hold at genesis either. A 1 GiB cache is a 425 to 656 mm^2 die +($224 to $343), affordable too; 4 GiB is a 3 to 4 die part at about $900 to $1,600 of silicon, which is a different +product but not an impossible one (the attacker's problem at that size is the 1,024 dependent cross-die reads per +hash, section 6). + +## 6. What the mirror buys the attacker, year by year + +From M16 (`docs/analysis/m16-recompute-attacker-2026-10-05.md`): with the cache on die the attacker recomputes 128 +items per hash at about 1,170 integer operations and 8 dependent 64-byte cache reads each, about 150,000 operations +and 1,024 dependent SRAM reads per hash. At a 5090-class integer budget (about 50 T op/s, approximate) that is +0.33 Ghash/s against the honest 141 Mhash/s projected for version 2 programs: 2.4x at equal silicon before any +fixed-function factor, 3x to 6x with one (approximate). The SRAM is 106 to 164 mm^2 of that chip at the headline +density (14 to 22% of a 750 mm^2 die; the m16 model's 13 to 40% band holds), so the mirror is cheap and the recompute +route is bound by integer throughput, not by SRAM. + +The layer 5 hot table changes nothing in that arithmetic: the hot table is read-only and derived from the day key +like the cache, so a chip mirrors it in the same SRAM (another 32 to 96 MB, 24 to 48 mm^2 at N5 headline) and reads +it at SRAM latency, which is exactly what a GPU's L2 does with it. Layer 5 taxes the DRAM-only chip (the one without +SRAM); it does not tax the SRAM chip. + +Dataset growth does not touch the recompute attacker: the attacker never holds the dataset. It taxes the +partial-store attacker (O-1.6, the time-memory curve, not drawn) and the honest card. + +Year by year under the schedule as it stands (flat 256 MiB), the mirror's area at the best node available that +year, headline density. Node years are approximate; the density trend from 2018 to 2025 is 37.0 to 57.1 Mbit/mm^2 +raw, 1.54x in 7 years, about 6% per year, and it came in one step (N2); the extrapolation past 2026 assumes that +average holds (approximate, and optimistic for the attacker: A16 and A14 have no disclosed SRAM cell yet). + +| Year | Calendar (approximate) | Dataset, GiB | Cache (spec) | Best node | Mirror of the cache, headline (lower bound), mm^2 | With a 96 MiB hot table, headline, mm^2 | Mirror as a share of a 750 mm^2 die | +|---|---|---|---|---|---|---|---| +| 0 | 2027 | 2.0 | 256 MiB | N2 (cited) | 106 (54) | 146 | 14% | +| 1 | 2028 | 2.5 | 256 MiB | N2 or A16 | 103 (52) | 142 | 14% | +| 2 | 2029 | 3.0 | 256 MiB | trend | 95 (48) | 130 | 13% | +| 3 | 2030 | 3.5 | 256 MiB | trend | 89 (45) | 123 | 12% | +| 4 | 2031 | 4.0 | 256 MiB | trend | 84 (43) | 116 | 11% | +| 5 | 2032 | 4.5 | 256 MiB | trend | 79 (40) | 109 | 11% | +| 6 | 2033 | 5.0 | 256 MiB | trend | 75 (38) | 103 | 10% | +| 7 | 2034 | 5.5 | 256 MiB | trend | 71 (36) | 97 | 9% | +| 8 | 2035 | 6.0 | 256 MiB | trend | 67 (34) | 92 | 9% | +| 9 | 2036 | 6.5 | 256 MiB | trend | 63 (32) | 87 | 8% | +| 10 | 2037 | 7.0 | 256 MiB | trend | 59 (30) | 82 | 8% | + +Reading. A flat cache's mirror shrinks from 14% to 8% of a large die over the decade, and a 5090-class consumer GPU +already carries 96 MB of L2 on one die with the full GB202 at 128 MB; at the 2020 to 2025 pace of GPU L2 growth +(6 MB, 72 MB, 96 MB on the three NVIDIA flagships in the whitepaper table) a consumer GPU could hold 256 MiB on die +within the decade. The spec's own rule for the cache ("must exceed the largest on-chip cache of any card that +mines") would then be broken by a flat cache. That is the real reason to grow it: not to price a chip out (section +5 shows the SRAM cannot do that) but to keep the cache out of every GPU's own cache, so the honest hash stays +DRAM-latency-bound and the recompute route stays a route only a custom chip can take. + +## 7. Answer to the layer 6 question, and the options + +Does the flat 256 MiB cache keep the SRAM mirror unaffordable through year 10? No. It is affordable at year 0 ($30 to +$56 of silicon per die at shipped density, section 5) and gets cheaper. What keeps the recompute attacker near 1x is +M16's integer arithmetic and the mixer-cost lever (4x the mixer cost puts the equal-silicon gain at 0.36x, bounded +by the CPU verify gate), not the cache size. The cache size does one other job, keeping the cache larger than any +GPU's L2, and that job needs growth. + +Options for the cache rule, with the honest costs each implies. Verifier fill time is 0.2 s per 256 MiB on one core +(spec 1.12: "a 0.2 s CPU cache fill", from the measured 175 to 190 ms of section 1.8.3), scaled linearly; the +verifier holds the whole cache (section 1.11), so its memory is the cache size plus the program and the interpreter. +GPU fill: 0.67 ms per 256 MiB on the 5090 (section 1.8.3), linear. The GPU dataset build (13.4 ms per 1 GiB on the +5090, section 1.8.3) depends on the dataset size, not the cache size; a larger cache spreads the build's 8 dependent +reads per item over more memory, which on a GPU means more of them miss L2 and the build slows by some factor +between 1x and the L2-to-DRAM latency ratio, which is a measurement to take (approximate; owed). Mirror area is at N2 +headline density (lower bound in brackets), the node of the first years; at the trend's year-10 density divide by +about 1.8. + +| Option | Rule | Cache at year 0 / 4 / 10 | Mirror at N2, headline (lower bound), year 0 / 4 / 10, mm^2 | Dies at year 10 (830 mm^2 reticle), headline | Verifier fill, one core, year 0 / 10 | Verifier memory, year 10 | GPU cache fill (5090), year 10 | Keeps the cache above a 96 MB L2 at year 10 | Keeps it above a 256 MB L2 | +|---|---|---|---|---|---|---|---|---|---| +| A, as specified | flat 256 MiB | 256 / 256 / 256 MiB | 106 (54) / 106 / 106 | 1 | 0.2 / 0.2 s | 256 MiB | 0.7 ms | yes, 2.7x | no | +| B | cache = dataset / 8 (today's ratio) | 256 / 512 / 896 MiB | 106 (54) / 213 (107) / 372 (188) | 1 | 0.2 / 0.7 s | 896 MiB | 2.3 ms | yes, 9.3x | yes, 3.5x | +| C | cache doubles when the dataset doubles (the dataset's own clock: year 4, then year 12) | 256 / 512 / 512 MiB | 106 (54) / 213 (107) / 213 (107) | 1 | 0.2 / 0.4 s | 512 MiB | 1.3 ms | yes, 5.3x | yes, 2x | +| D | cache = dataset / 4 | 512 / 1,024 / 1,792 MiB | 213 (107) / 425 (215) / 744 (376) | 1 | 0.4 / 1.4 s | 1.75 GiB | 4.7 ms | yes | yes, 7x | +| E, one reticle | cache sized so the mirror exceeds one reticle at the node of the day: 2 GiB at N2 headline density (section 4; 4 GiB on the lower bound), growing with density | 2 GiB / about 2.3 / about 3.5 GiB | 850 / 850 / 850 (by construction) | 2 | 1.6 / 2.8 s | 3.5 GiB | 5.4 / 9.4 ms | yes | yes | + +Where the working set enters (coordinator's budget: 1 GiB table + hot table + scratch for every resident warp + +buffers under 6 GB on an 8 GB card): the cache is not in the miner's working set at hash time (the dataset is built +from it once a day and the cache can be dropped or kept), so options A to D do not move that budget; the dataset's own +growth does (2 GiB at genesis, 4 GiB at year 4, 7 GiB at year 10, which is past an 8 GB card at about year 8 on its +own). Option E's 2 GiB cache would have to be built on the card and dropped, which is fine for a 16 GB card and tight +on an 8 GB one at build time (2 GiB cache + 2 GiB dataset + hot table). The per-warp scratch at 170 SMs x 64 warps +(approximate, readwidth) is 340 MB at 32 KB and 1.36 GB at 128 KB per warp; with the 1 GiB table, a 96 MB hot table +and buffers that is 1.5 to 2.5 GB at the prototype dataset size, 2.5 to 3.5 GB at the 2 GiB genesis size, inside +6 GB either way. + +Recommendation. Option C (the cache doubles when the dataset doubles) is the one that keeps the spec's own rule true +with the smallest verifier cost: it ties the cache to a clock the spec already has, keeps `AND MASK` (a power of two +every step, which is the 1.13.3 option (b) argument again), costs the verifier 0.4 s and 512 MiB at year 4 and nothing +more until year 12, and keeps the cache 2x above a 256 MB GPU L2 if one appears. It does not price a chip out; nothing +about cache size does (section 5). The lever that does is the mixer cost multiplier of M16, which is the gate 1 +decision to take beside this one. Option B is the same idea in a smooth form and costs the verifier 0.7 s at year 10. +Option E is the only one that makes the mirror a multi-die part and it costs every verifier 1.6 s and 2 GiB at +genesis (at the headline density; the lower-bound density would ask for 4 GiB and 3.2 s), which fails the spirit of +the 10 ms verify gate (the fill is once a day, but a light node joining pays it on every day it syncs across). + +Decision for Josh, at gate 1: A, B, C, D or E above, together with M16's mixer multiplier. Nothing here changes a +vector today: the cache size is a prototype value of spec 1.16 and the growth rule would be a new sentence in 1.13.3. + +## 8. Why the latency bound is the property to lean on (citations behind the plan's rule) + +The plan's "what stays true" paragraph says DRAM latency is the same physics for everyone and bandwidth per watt is +what a custom memory chip buys. The sources behind that: + +| Claim | Figure | Source | +|---|---|---| +| Random-access DRAM latency is the same across memory types | Row cycle time 40 to 48 ns across DDR4, GDDR5 and HBM2 | Li, Reddy and Jacob, "A Performance and Power Comparison of Contemporary DRAM Architectures", MEMSYS 2018 (coordinator's chip-economics research, 5 October 2026) | +| Latency does not scale, bandwidth does | DRAM latency improved about 1.3x in two decades while bandwidth improved about 20x | K. Chang, "Understanding and Improving the Latency of DRAM-Based Memory Systems", PhD thesis, CMU, 2017 (same research) | +| No mining chip has bought latency with exotic memory | No shipped mining chip has used HBM or stacked memory; the Ethash chips used DDR3, GDDR6 and undisclosed types | same research; the Ethash chip gain of about 3x in the plan came from bandwidth per watt, not latency | +| The honest hash is latency-bound on every card measured | The hash runs within a few percent of 1/128 of each card's dependent random-read ceiling (5090, 9070 XT, M5 Max) | `docs/bench-log.md`, "the 9070 XT on the eGPU", 5 October 2026 (measured) | + +Reading for layer 6: an SRAM mirror beats DRAM latency by about 10x per read (a 64 MiB buffer inside the 9070 XT's +Infinity Cache chased at 9.2 G loads/s against 2.5 in GDDR6, the same bench-log entry; the 5090's L2 at 5.8x the +hash rate of its 1 GiB dataset, M16), which is why the recompute attacker is bound by the 1,024 dependent SRAM reads +and the 150,000 integer operations per hash and not by the SRAM's size or price. The cache size decides whether the +mirror is one die or several (section 4); it does not decide whether the mirror exists. + +## 9. What is cited, what is approximate, what is owed + +| Item | Status | +|---|---| +| Bit cells for N7, N5, N3B, N3E, N2, Intel 18A | cited (section 3.1) | +| Shipped cache-die density (AMD V-Cache 64 MB on 41 mm^2 at 7 nm; Graphcore GC200; Groq TSP) | cited (section 3.2; V-Cache checked against Tom's Hardware's Hot Chips 33 report, 5 October 2026; the Graphcore and Groq rows are from the coordinator's research and were not re-checked tonight) | +| Scaling the V-Cache density to other nodes by the bit-cell ratio | approximate, stated | +| Samsung SF2 or SF3 bit cell | not found; left out | +| Array efficiency 0.70 | WikiChip's and SemiAnalysis's convention, bracketed by two ISSCC 2025 macros (67 to 80%); a macro figure, used only as the lower bound | +| Wafer prices | approximate, supply-chain reporting, cited | +| D0 = 0.1 per cm^2, Poisson yield | assumption, stated | +| Node years and the 6% per year density trend past 2026 | approximate, extrapolated from cited 2018 to 2025 points | +| GPU L2 sizes | cited (NVIDIA whitepaper); AMD Infinity Cache approximate | +| Latency citations (MEMSYS 2018, Chang 2017, mining-chip memory types) | from the coordinator's research, not re-read tonight | +| Recompute attacker arithmetic | M16, which is itself arithmetic on measured rates, not a chip measurement | +| Dataset-build slowdown at a larger cache on a GPU | owed, a measurement (5090 at a 512 MiB and 1 GiB cache) | +| The on-die emulation of M16 (inline kernel with a 64 MiB cache inside the 5090's L2) | still a PC job (M16) | + +## 10. The arithmetic + +``` +MiB = 2^20; bits = cache_MiB * MiB * 8 +headline_mm2 = cache_MiB * (41 / 64) * (cell_um2 / 0.027) (V-Cache: 41 mm2 per 64 MiB at N7, scaled by cell) +raw_Mbit_per_mm2 = 1 / cell_um2 (1e6 cells per mm2 per um2 of cell) +lower_bound_mm2 = bits / (raw * 0.70 * 1e6) +dies_per_wafer = pi * 150^2 / area - pi * 300 / sqrt(2 * area) +yield = exp(-area_mm2 * 0.001) (D0 = 0.1 per cm2) +cost_per_good_die = wafer_price / (dies * yield); over 830 mm2: k = ceil(area / 830) dies of area / k, cost x k +reticle_GiB = 830 / (mm2 per MiB) / 1024 +``` +Run on 5 October 2026 with Python 3 on the M5 Max; the printed tables are the ones above, rounded. diff --git a/docs/bench-log.md b/docs/bench-log.md index 0ecb53182..cc01b90ec 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1551,6 +1551,148 @@ What is measured: one BLS12-381 aggregate signature over 16 summed G1 keys plus Reading (the NEW finding, ledger C4). With the module off GHOSTDAG alone converges on the heavier chain and the losing side's records re-determine (F24 works when the chain moves). With the module on the overlay holds during the split (A, with 30% of the frozen table, locks nothing; B locks 7 and 8) and then fails at the heal in the shipped node: B's certificates for blocks off n0's chain are "kept pending until the chain decides (no lock at this index)", n0's chain never decides because GHOSTDAG keeps its heavier tip and nothing turns the certificate into a fork-choice constraint, and once n0's last lock (index 7, DAA 209) is one window old (DAA 329) the frozen table stops applying on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys are 100% of A's own window (B's post-cut blocks are red there) and n0 locks 10, 11, 12 alone; B's certificates for 10 and 11 then log CONFLICTING on n0 (n0 log, 17:27:04 to 17:29:54 BST). A finality fork from a 96-s honest partition, no attacker, table intact at the heal; the 150-s run and the v2 control end the same way. The spec's fork choice ("GHOSTDAG among tips through all certified checkpoints", 3.5) is therefore implemented only for certificates over blocks already on the node's chain. Fix named in the ledger entry: verify an off-chain certificate against the table at its own block and let it constrain fork choice (a certificate-driven reorg), then re-determine. Raw: `scratchpad fud-a/c4-results-*.md`, node logs `c4-on90-tmp/`, `c4-v2-control-tmp/`. +## 5 October 2026 (night), read width of the lottery hash: 4, 16 and 64-byte loads, a per-load mix, a written scratch; three cards (gate 1 experiment, cryptographer) + +Branch `readwidth` (commits 019b014, b970dda, 4badcee, a9e002c, d0018cf and the entry commit); plan and recommendation in `docs/plans/read-width.md`. Nothing here changes consensus: every class sits behind `igneum-pow --class` and the default class is generator version 2 byte for byte (`igneum-pow/tests/packs.rs` passes on the four pinned packs after every commit). Question (Josh, after "the 9070 XT on the eGPU" above): would wider reads keep the latency-bound random-access property while closing the vendor gap. Additions from the coordinator: a per-load width drawn from an era-fixed mix, and a written per-warp scratch (measurement only, no soundness claim). + +**What a class does** (`igneum-pow/src/generator.rs` `LoadClass`, `verify::fold_words`, the three emitters): a load of W words reads the W-word-aligned address `(src AND MASK) AND NOT (W - 1)` and folds every word into `dst` (`x = dst ^ w0; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]`); W = 1 is the lottery hash exactly (`w4` = pack `bcc1248b10cc90f2`). A mix class draws W per load with one extra `below(100)` roll per instruction. A scratch class `scrk` turns `k` of the 16 memory slots into read-modify-writes of a 16-byte slot of the lane's share of a `kb` KiB per-warp scratch (kernels run persistent warps, one per block or work-group; a slot reads as a seed-and-base fill until the unit writes it, behind a per-unit tag). Program ids carry the class. Dependent chain and 32-lane unit unchanged. + +**Correctness**: 23 packs (`proto-cuda/packs-readwidth/`, Rust CPU reference vectors). Every pack passed its three vector units and the cache and dataset checks on Metal (M5 Max, `proto-metal/packbench`), Apple OpenCL (`--bench-pack`), the RTX 5090 (NVRTC, `igneum-worker-cuda --bench`) and, the 16 width and mix packs, the RX 9070 XT (`igneum-worker-opencl --bench-pack`); the 2^24 batch fingerprints agree across all four runtimes on every pack (for example w16 `e7c890445b47af60`, w64 `836e56e7d496e980`, mixB-2 `a18ac73098c76007`). The clang CUDA emulation (w16, w64, w64x4, mixA-0, mixB-0: 3 of 3 units standalone and 2 of 2 in batch at 2 warps per block) and the clang OpenCL emulation (the same five plus scr2k32 and scr8k128, sub-group 32 and, width packs, wave64 with sub-group shuffles) pass with equal fingerprints per configuration. Acceptance rule on the classes: 60 candidates per class, rejection 0 to 14 of 60 (w16 and w64 as v2; the mixes the same; the scratch classes' distinct-address bound now covers dataset loads only, since a 64-slot lane scratch repeats slots by design). CPU verifier (M5 Max, one core, avg of 50 units, `igneum-pow bench --class`): v2 0.604 ms, w16 0.610, w64 0.630, w64x4 0.160, mix50-35-15 0.620, mix25-50-25 0.614, scr0k32 0.600 (1.004 on a loaded re-run), scr2k32 0.657, scr4k32 0.458, scr8k32 0.317, scr2k128 0.535, scr4k128 0.458, scr8k128 0.311; per hash divide by 32. The wide reads cost the verifier nothing (a lane's words lie in one item); scratch ops replace item derivations and make it cheaper. + +**Probes** (`--memprobe`, dependent random reads at 1024 MiB, G reads/s, best over lanes in flight; 4 B = the hash's pattern; the 5090 and 9070 XT with the card off in the app, the Mac through Apple OpenCL under a load average of 5 to 10): + +| Card | 4 B chase | 16 B | 64 B | 64 B as GB/s | coalesced stream GB/s (rated) | integer chain | +|---|---|---|---|---|---|---| +| RTX 5090 (PC 2, CUDA) | 17.5 to 18.2 | 18.0 to 19.9 | 9.1 to 15.7 (9.1 at 4 M lanes) | 584 | 1,579 (1,792) | 39.0 T op/s | +| RX 9070 XT (PC 1, eGPU, OpenCL) | 2.42 to 2.66 | 2.43 to 2.73 | 2.47 to 2.87 | 158 | 636 (640) | 6.2 T op/s | +| Apple M5 Max (Apple OpenCL, approximate) | 3.50 | 3.51 | 3.51 | 225 | 522 | | + +Reading: on the 9070 XT and the M5 Max a 64-byte dependent read costs exactly what a 4-byte one costs (the line is fetched either way); on the 5090 a 64-byte read costs about two 4-byte reads (two 32-byte sectors) and the 64 B chase at full occupancy sits at 584 GB/s, a third of the stream. + +**Hash rates** (5 timed dispatches of 2^24 nonces after a warm-up; Metal and the 9070 XT by device time, the 5090 by wall time around the stream sync; the PC cards switched off in the app for the run and restored, PC 1's 5090 and the integrated chip kept mining; the Mac under other agents' builds, load 4 to 9, so its absolute numbers carry that; the share = measured / (the card's probe ceiling at the class's widths / loads per hash)): + +| Class | dataset B/hash | RTX 5090 MH/s (share) | RX 9070 XT MH/s (share) | M5 Max Metal MH/s (share) | 5090 / 9070 | +|---|---|---|---|---|---| +| v2 (w4, the lottery hash) | 512 | 136.1 (0.96) | 18.15 (0.87) | 27.74 (1.01) | 7.5x | +| w16 | 2,048 | 139.8 (0.90) | 17.90 (0.84) | 28.26 (1.03) | 7.8x | +| w64 | 8,192 | 71.9 (0.58) | 17.59 (0.78) | 28.27 (1.03) | 4.1x | +| w64x4 (32 loads) | 2,048 | 275.3 (0.56) | 75.19 (0.84) | 109.7 (1.00) | 3.7x | +| mix50-35-15, 6 programs: min / median / max (spread of median) | 1,664 to 3,680 | 99.5 / 114.2 / 121.0 (18.8%) | 17.45 / 18.76 / 18.83 (7.4%) | 25.36 / 27.26 / 28.43 (11.3%) | 6.1x | +| mix25-50-25, 6 programs | 2,240 to 5,024 | 95.9 / 107.3 / 119.8 (22.3%) | 17.84 / 18.45 / 18.85 (5.5%) | 23.21 / 24.68 / 25.21 (8.1%) | 5.8x | + +Scratch (variant 5; N persistent warps; 5090: 2,048 warps launched against a resident capacity of 4,080 = 24 blocks/SM x 1 warp/block x 170 SMs at `--block-warps 1`, the occupancy query unchanged by the allocation (24 before and after); Metal: 2,048 to 16,384 warps swept, best shown; arena = N x per-warp size; the whole working set = 1 GiB dataset + 256 MiB cache + 128 MiB output + arena, under 2 GB on every row): + +| Class (k of 16 slots, KiB per warp) | scratch ops/hash | dataset B/hash | RTX 5090 MH/s (vs scr0, share) | M5 Max Metal MH/s (vs scr0) | RX 9070 XT MH/s | 5090 arena / working set | +|---|---|---|---|---|---|---| +| scr0k32 (control, persistent loop, no RMW) | 0 | 512 | 139.1 (0, 0.98) | 28.25 (0) | 17.88 (control, 0.86) | 64 MiB / 1.4 GiB | +| scr2k32 (12.5%) | 16 | 448 | 114.4 (-18%, 0.80) | 26.14 (-7%) | 14.65 (-18%) | 64 MiB / 1.4 GiB | +| scr4k32 (25%) | 32 | 384 | 109.8 (-21%, 0.76) | 31.74 (+12%) | 14.00 (-22%) | 64 MiB / 1.4 GiB | +| scr8k32 (50%) | 64 | 256 | 122.1 (-12%, 0.82) | 49.08 (+74%) | 14.17 (-21%) | 64 MiB / 1.4 GiB | +| scr2k128 (12.5%) | 16 | 448 | 110.1 (-21%, 0.77) | 26.24 (-7%) | 14.07 (-21%) | 256 MiB / 1.6 GiB | +| scr4k128 (25%) | 32 | 384 | 98.0 (-30%, 0.68) | 28.08 (-1%) | 13.14 (-27%) | 256 MiB / 1.6 GiB | +| scr8k128 (50%) | 64 | 256 | 72.8 (-48%, 0.49) | 35.44 (+25%) | 12.03 (-33%) | 256 MiB / 1.6 GiB | + +The 9070 XT rows are 2,048 persistent warps (4,096 within 1 percent), arena 64 MiB at 32 KiB and 256 MiB at 128 KiB, working set 1.4 and 1.6 GiB; its control (17.88, the persistent loop) equals its v2 rate (18.15) within 2 percent, and every RMW share costs it 18 to 33 percent: on AMD a scratch op is a dependent 16-byte read plus a write into a region the 64 MB Infinity Cache does not hold for 2,048 warps, so it is memory work there as on the 5090, not the cached op it is on Apple. Apple OpenCL on the same scratch packs (wall time, `--bench-pack --warps 2048`): scr0k32 27.85, scr2k32 28.58, scr4k32 32.43, scr8k32 47.93, scr2k128 25.67, scr4k128 27.24, scr8k128 32.93 MH/s, the Metal shape within 4 percent, fingerprints equal. Bytes moved per scratch op: 16 read + 16 written (the tag word included); per hash at 50 percent, 1,024 read + 1,024 written beside 256 of dataset reads. The 5090 at 4,096 launched warps (above its 4,080 resident) lost 2 to 26 percent (scr8k32 90.0 MH/s), so the rows above are the in-capacity launch. + +**Readings.** (1) Same count, wider: the vendor gap does not move at 16 B (7.8x) because on the 9070 XT a 4-byte read already costs a 64-byte line and on the 5090 a 16-byte read costs one 32-byte sector, the same as 4 bytes: the memory systems do identical work, only the fold's input grows. At 64 B the gap closes to 4.1x, entirely by the 5090 losing half its rate (its share falls to 0.58 and its DRAM traffic reaches 589 GB/s, 37 percent of the stream: bandwidth, not latency, bounds it), while the 9070 XT and the M5 Max do not move. (2) Fewer, wider (w64x4): 3.7x, but every card runs 4x faster because the dependent chain is 32 loads long instead of 128; the 5090 sits at a 0.56 share (bandwidth), so a chip with more bandwidth per dollar than a GPU gains, which is the Ethash shape the design avoids. (3) The mix: the hour-to-hour spread is 7 to 22 percent of the median per card (the 5090 the widest, because its 64-byte loads are the expensive ones and their count per program runs 2 to 8 of 16); the programs with many 64-byte loads (mixA-3, mixA-5, mixB-2) are the slow hours on the 5090 and the fast ones nowhere. (4) The scratch: on the 5090 every RMW share costs 12 to 48 percent against the persistent control, the 32 KiB arena less than the 128 KiB one (the smaller arena, 64 MiB over 2,048 warps, sits inside the 96 MB L2); on the M5 Max the 32 KiB rows are FASTER than the control (+12 and +74 percent at 25 and 50 percent), because the arena (128 MiB over 4,096 warps) lives in the chip's caches and a scratch op is cheaper than a dataset read, so replacing dataset loads raises the rate: the scratch at these sizes is not memory work on Apple and is partly cached on NVIDIA. The chip row for these variants comes from the ca2-soundness branch; what this entry gives is the GPU cost and the share. (5) Latency-bound shares: v2 0.87 to 1.01 on the three cards, w16 0.84 to 1.03, w64 0.58 (5090) and 0.78 (9070 XT); the Mac's shares above 1 are an Apple OpenCL probe under load against a Metal rate. + +Jobs: `run-readwidth-5090-20261005` and `run-readwidth-9070-20261005` (probes; the packs refused for their string seeds, fixed in a9e002c), `run-readwidth-5090-20261005c`, `run-readwidth-9070-20261005c` (benches), `run-readwidth-9070-scratch-20261005d` (the scratch packs after the `__local` fix d0018cf, AMD's compiler requires the exchange buffer at the kernel's outermost scope); read back with `node tools/jobs.mjs --all`. Mac commands and logs: `docs/plans/read-width.md` section 3. The worker exes for the jobs: `proto-cuda/nvrtc/build-windows.sh` on this branch (mingw), sha256 of the CUDA one `6f46336f...defe1`. +## 5 October 2026 (night), the hot table on the M5 Max: a second table sized to GPU cache beside the 1 GiB dataset (Counter ASIC 2.0 layer 5) + +Branch `ca2-cache` (on readwidth 1ea7a52), `docs/plans/hot-table.md`. Apple M5 Max, measure lock held, the Mac's load average 14 to 27 throughout (other agents' CPU work; the lock serialises builds and measurements, not every process), so the ratios inside one session are the result and the absolute rates are not quiet numbers. Packs `proto-cuda/packs-ca2-hot/hot{32,64,96}k4`, `hot64k2`, `hot64k8` from `igneum-pow export --seed igneum-genesis --day 2026-10-03 --class hotk` (the version 2 genesis program with k of its 16 loads redirected to an S MiB table H keyed by `seed_words("igneum-hot/" || seed bytes)`, read at `H[mulhi(src, words)]`). + +**Probe** (`proto-opencl/igneum-bench-cl-igneum-genesis-mh --memprobe --probe-mib S`, Apple OpenCL, wall time, best of 3, 256 dependent steps per lane, work-group 256; the ceiling row is 4,194,304 lanes): + +| MiB | chase at 4,096 lanes | ns per dependent load | chase ceiling, G loads/s | indep x8 ceiling | stream | +|---|---|---|---|---|---| +| 32 | 3.51 G/s | 1,168 | 21.7 | 21.8 | 138.7 GB/s | +| 64 | 3.63 | 1,129 | 12.8 | 13.0 | 199.1 | +| 96 | 3.30 | 1,242 | 12.3 | 12.7 | 242.6 | +| 1024 | 2.22 | 1,844 | 3.50 | 3.50 | 521.5 | + +**Hash rate and bit-exactness** (Metal `proto-metal/packbench --pack --batches 5 --batch-log2 24`, GPU time; Apple OpenCL `--bench-pack --pack --batches 5 --batch-log2 24`, wall; both fill H on the device from the pack's `igneum_hot_fill` and check it; fingerprint = FNV-1a 64 over 2^24 outputs at base 0): + +| Pack | Metal Mhash/s | Apple OpenCL Mhash/s | fingerprint (equal on both) | vectors | hot table head, last line, FNV | g against v2 (Metal) | probe-predicted g | ideal g | +|---|---|---|---|---|---|---|---|---| +| igneum-genesis-mh (v2) | 27.68 | 27.61 | 25f96e7dce90bd4e | 96/96 both | none | 1 | 1 | 1 | +| hot32k4 | 33.90 | 33.92 | d2e6cf3b61d0b9fe | 96/96 both | PASS both | 1.22 | 1.27 | 1.33 | +| hot64k4 | 30.93 | 30.89 | e4c5263ac650cc0d | 96/96 both | PASS both | 1.12 | 1.22 | 1.33 | +| hot96k4 | 29.06 | 28.97 | 5d63439b6e394521 | 96/96 both | PASS both | 1.05 | 1.22 | 1.33 | +| hot64k2 | 27.67 | 27.27 | 352633bdbbb0d2b6 | 96/96 both | PASS both | 1.00 | 1.10 | 1.14 | +| hot64k8 | 47.42 | 46.73 | da54630d7dfaaf85 | 96/96 both | PASS both | 1.71 | 1.57 | 2.0 | + +Hot table fill, Metal GPU time: 0.07 ms (32 MiB), 0.15 (64), 0.22 (96). Hot table FNV-1a 64 of the genesis epoch: c1767ba3ef02719f (32 MiB), 77ca4b9527104530 (64), 79bcf436c4e5bc47 (96); cache 48c4f5bf24166b2e unchanged. + +**CPU verifier** (`igneum-pow bench --seed igneum-genesis --day 2026-10-03 --class --warps 50`, one core, release): + +| Class | hot fill, one core | items per warp | ms per warp | +|---|---|---|---| +| v2 | none | 4,096 | 0.626 | +| hot32k4 | 24.0 ms | 3,072 | 0.489 | +| hot64k4 | 46.4 ms | 3,072 | 0.504 | +| hot96k4 | 73.0 ms | 3,072 | 0.488 | +| hot64k2 | 45.5 ms | 3,584 | 0.560 | +| hot64k8 | 47.7 ms | 2,048 | 0.344 | + +Reading: bit-exact across Metal, Apple OpenCL and the Rust reference on every hot pack, hot table included. On this card the 32 MiB table delivers 92% of the probe's predicted gain with the dataset streaming beside it, 64 MiB about half, 96 MiB a quarter; k = 8 at 64 MiB gives 1.71x against an ideal 2.0x. The verifier gets cheaper with k (a hot load is one table read, a dataset load is an item derivation) and pays 24 to 73 ms per epoch for the fill. Chip model with these g in the plan, section 6.4. The RTX 5090 and RX 9070 XT rows are a prepared PC job (`relay/playbooks/ca2-hot-{5090,9070}-bench.ps1`, zip `~/Desktop/igneum-ca2-hot.zip`), not run. + +Crate: `cargo test --release` 52 pass (39 unit, 13 pack tests: the four pinned v2 packs byte-identical, the five hot packs pinned with their load-form count: exactly 16 - k masked dataset loads and k hot loads per hash kernel). + +**Addendum, the added form** (coordinator's form of 5 October 2026: 16 + k load slots, the k hot ones drawn among them, the 16 dataset loads and the 4,096-item verifier bound unchanged; packs `hot32k4a`, `hot64k4a`, `hot96k4a`; second Mac session 21:03 to 21:19 UTC, load average 7 to 14; same harnesses and commands, branch `ca2-cache` on ca2-v3 464d6e1, the hosts rebuilt on the merged packfile.h): + +| Pack | Metal Mhash/s | Apple OpenCL Mhash/s | fingerprint (equal on both) | vectors | hot table | g against v2 (Metal, v2 27.63 in this session) | probe-predicted g | CPU verify ms/warp (v2 0.602) | hot fill, one core | +|---|---|---|---|---|---|---|---|---|---| +| hot32k4a | 25.76 | 25.72 | 8a3414735db4523c | 96/96 both | PASS both | 0.93 | 0.96 | 0.631 | 21.7 ms | +| hot64k4a | 23.92 | 23.87 | 45668f34105f6307 | 96/96 both | PASS both | 0.87 | 0.94 | 0.609 | 43.3 ms | +| hot96k4a | 22.92 | 22.88 | af763997dfee4c82 | 96/96 both | PASS both | 0.83 | 0.93 | 0.614 | 64.9 ms | + +Reading: the added form costs this card 7, 13 and 17% of its rate at 32, 64 and 96 MiB for four extra loads per iteration, more than the probe predicts as the table grows; the verifier is unchanged (4,096 items, plus 32 table reads) and pays the fill per epoch. Chip arithmetic in the plan, section 6.4. All eight packs load and self-test through the rebuilt OpenCL host (the Windows exe's host.c) on the Mac. + +**Addendum, the PCs** (5 October 2026, 21:29 to 21:35 UTC, PC 1 ae432dc7, app 0.3.9 before and after; fetch `fetch-ca2-hot-20261005` (zip sha256 bd49faa1c9d48024f49c615481faff5c68a4c09f0889dbaf009c208674d67b3f), jobs `run-ca2-hot-5090-20261005` (126 s) and `run-ca2-hot-9070-20261005` (247 s), both exit 0, the card under test switched off in the app through `api/cards` and restored; workers `igneum-worker-cuda.exe` sha256 956c4ab34f42cbcd1d2c1c6fb1a58fd9b3a8c70166df771cafcd0296ca6a27d4 and `igneum-worker-opencl.exe` sha256 32d3d34390aad70485c3524424c354223387137d383b5c5daf01f40073c12703, built from ca2-cache 196db96 on ca2-v3's merged packfile.h d2cd6e1; read back with `node tools/jobs.mjs --all`): + +Probe (`--memprobe --probe-mib S`, dependent 4 B chase ceiling at 4,194,304 lanes, G loads/s; ns per dependent load at 4,096 lanes in brackets): + +| Card | 32 MiB | 64 | 96 | 1024 | stream at 1024 MiB | +|---|---|---|---|---|---| +| RTX 5090 (CUDA, wall) | 112.6 (320) | 112.6 (340) | 112.6 (336) | 17.6 (610) | 1,563 GB/s | +| RX 9070 XT (OpenCL, event) | 9.88 (396) | 9.47 (457) | 8.18 (454) | 2.43 (1,579) | 633 GB/s | + +Rates (5 dispatches of 2^24 after a warm-up; 5090 `--bench --block-warps 1`, 9070 XT `--bench-pack --device 1` work-group 256; every row check=PASS with the Mac's fingerprint; v2 references from the readwidth entry, same night, same workers: 136.1 and 18.15 MH/s): + +| Pack | 5090 MH/s | g | 9070 XT MH/s | g | ideal g | +|---|---|---|---|---|---| +| hot32k4 | 146.6 | 1.08 | 19.79 | 1.09 | 1.33 | +| hot64k4 | 140.8 | 1.03 | 18.73 | 1.03 | 1.33 | +| hot96k4 | 138.5 | 1.02 | 18.33 | 1.01 | 1.33 | +| hot64k2 | 137.5 | 1.01 | 18.17 | 1.00 | 1.14 | +| hot64k8 | 163.6 | 1.20 | 22.32 | 1.23 | 2.0 | +| hot32k4a | 118.7 | 0.87 | 15.27 | 0.84 | 1 | +| hot64k4a | 115.4 | 0.85 | 14.62 | 0.81 | 1 | +| hot96k4a | 114.4 | 0.84 | 14.56 | 0.80 | 1 | + +Reading: the probe promises a full hit rate on the 5090 (every S inside the 96 MiB L2 at one ceiling, 6.4x DRAM) and the hash gets 2 to 8% at k = 4 and 20% at k = 8; the 9070 XT the same shape. The dataset's random lines evict the table from the shared cache on every card. The added form costs 13 to 20% of the rate. Recommendation in `docs/plans/hot-table.md` section 6.4: do not adopt layer 5 in either form on these measurements. +## 5 October 2026 (night), mixer x4 and the cache growth rule: the class v3 dataset construction, with the x8 candidate (Counter ASIC 2.0; branch ca2-mixer on ca2-v3 6c75dad; cryptographer's lane) + +Machine: Apple M5 Max, 64 GiB, Darwin 25.6.0. Write-up `docs/plans/mixer-x4.md`; chip model `docs/analysis/chip-model-v3.md`; code `igneum-pow` (LoadClass mixer_mult and growth, memhard::Shape, the schedule, the three emitters), packs `proto-cuda/packs-ca2-mixer/`, tests `igneum-pow/tests/mixer.rs` and `tests/packs.rs`. Commits 0fc0ad1, 66eeba3, e4c04a7, 7ce8d1e, 504cae4, fe4e193 and this entry's. + +What changed. Under program class v3 (`V3_CLASS = LoadClass::MX4`) every mixer application of the item derivation is `m = 4` applications with round keys `(r m + j + 1) x 0x9E3779B9`, the 8 dependent cache reads per item unchanged; the cache doubles when the dataset doubles (`growth_doublings(d) = floor(log2(1 + d / 1460))`: 2^26 words to day 1,459, 2^27 from day 1,460, 2^28 from day 4,380). Version 2 is byte-identical: fresh exports of igneum-genesis-mh and igneum-devnet-v4-epoch0 `diff -r` IDENTICAL against the checked-in packs, and the crate tests regenerate every pinned file. A v3 program of a seed is the v2 program of that seed instruction for instruction (v2 loads take no width roll); only the dataset words and the hashes change. + +Bit-exactness, `with-lock.sh run`, 22:05 and 21:45 UTC: the two pinned v3 packs (mx4-genesis, mx4-devnet-epoch0: dataset words 0..15 `61ff2180 0d4c7e6c ...` and `afe80d67 b9fbd029 ...`, word MASK `5020180e` and `e6a99c7a`, unit at base 0 lane 0 `63acd2d273f475ba` and `212c6442b51e87ae`) and the two x8 candidate packs on Metal (`packbench`, built from this branch) and Apple OpenCL (`igneum-bench-cl --bench-pack`): 3/3 standalone and 3/3 in batch, 96 of 96 lanes, cache FNV-1a 64 unchanged from v2 (48c4f5bf24166b2e, 448274a57f508cbc), dataset head, word MASK and 64 samples PASS, one 2^24 fingerprint per pack across both harnesses (mx4 6f48d5a2aa0dbe5f and 73caaebb28e808fe; mx8 7c28cfb06c5c65a9 and bbb183f72692f840); hash rate the v2 rate (27.5 to 27.7 MH/s GPU time, the hash kernel is unchanged). Fuzz: 200 class v3 programs (4 units each across the 32-bit range, one in the top 256 nonces) interpreted twice on the CPU, 800 of 800; the same 200 packs on Metal 200 of 200 (`--batch-log2 9 --batch-base 4294967040`, the wrapping unit inside the window), every tenth on Apple OpenCL 20 of 20; x8: 50 of 50 on Metal, 5 of 5 on OpenCL. Stats (8,192 outputs per seed, two seeds): v3 avalanche 49.97 to 49.99 percent, worst bit z 1.92 to 3.09, 0 duplicates (v2 beside it 49.87 to 49.98, z 2.25 to 2.30). Edges: items 0, 1, 2^28 - 1, 2^32 - 1 by hand at m = 1, 2, 4, 8; words 0, 15, 16, 17, MASK - 1, MASK through the fetch path. Determinism: two epochs, every vector and file equal and equal to the pinned pack. The scratch soundness tests of ca2-soundness (cherry-pick 0d8f745) 7 of 7 on this tree. Crate: 44 lib + 12 packs + 4 mixer + 7 scratch tests pass. A first Metal fuzz run reported 200 of 200 FAIL on an empty RESULT line (a packbench built before the `--batch-base` cherry-pick); it was read as a failure, the harness rebuilt, the run repeated. + +Timings, `with-lock.sh measure`, one session 21:40:12 to 21:40:23 UTC, one core, two rounds; the box carried a load average of 5.6 (one minute) and 26 (fifteen minutes) from unlocked processes, so the absolute figures are about 2.2x the quiet readwidth night's 0.604 ms v2 row and the ratios are the measurement: + +| Construction | Verifier ms per 32-lane unit, avg of 50 (two rounds) | Worst cold unit | Against v2 | 256 MiB fill, one core | Metal 1 GiB build, GPU ms | +|---|---|---|---|---|---| +| v2 | 1.361 / 1.310 | 1.579 | 1 | 172 to 173 ms | 29.7 (first touch) / 21.0 | +| x4 (class v3) | 1.956 / 1.923 | 2.043 | 1.45x | 172 to 175 ms | 20.9 / 21.0 | +| x8 (candidate) | 2.785 / 2.790 | 2.942 | 2.09x | 172 ms | 21.9 / 21.9 | + +Reading: the mixer multiplies the verifier's ALU part only (the 8 dependent misses per item are unchanged), hence 1.45x and 2.1x and not 4x and 8x; the Mac's GPU build is latency-bound and does not move with the mixer, so the "under 1 s on every discrete card" half of the x8 rule is the PC job (five packs, `relay/playbooks/mixer-x4-pc1-bench.ps1`, waiting for the go). Verification throughput (C19): a quiet 2026 core serves about 1,100 shares per second at x4 and 800 at x8 (1,660 at v2, re-cutting spec 09's 2,270), a 22,000-member pool at one share per 10 s needs 2 cores at x4 and 3 at x8, IBD over 108,000 headers is 1.6 min at x4 and 2.3 at x8 on that core; the 10 ms gate keeps 8.0 ms (x4) and 7.1 ms (x8) of margin on the loaded core, 6 to 7 ms on a 2019-class laptop core (approximate, unmeasured, O-1.14). + +Chip model (`docs/analysis/chip-model-v3.md`): the on-die-cache recompute chip at 50 T op/s against the 5090's measured 136.1 MH/s: v2 334 MH/s, 2.45x bare, 7.4x with the 3x fixed-function factor; x4 83.5 MH/s, 0.61x bare, 1.84x with the factor, 1.53x with the 128 mm^2 N5 mirror deducted at equal silicon; x8 41.7 MH/s, 0.31x, 0.92x, 0.76x. The claim at x4 is "under 2x" with the margin thin on the equal-budget convention (a 3.3x factor or a 10 percent larger budget reads 2.0x); the hot table in the added form would have raised it to 2.1x to 2.2x at the 5090's g (kept as measured, not adopted). Nothing here is a measurement of a chip. + +**Addendum, 22:15 UTC: the verifier regression, the PC 1 build rows, and x8 into v3.** The era agent measured the same v2 input with readwidth's binary (0.604 ms) and ca2-v3 HEAD's (1.33) in one minute; bisected under the measure lock to this branch's 0fc0ad1 (seam 6c75dad 0.610, 0fc0ad1 1.332; the "loaded box" reading above was wrong by that factor, the load was real but the 2x was the code). Cause: the item loop (`derive_items`) inlined into `MemhardCpu::fetch`; the mask hoisted, the mask constant, and the constant-mask loop inlined all stayed at 1.33, the same loop `#[inline(never)]` read 0.60 to 0.62. Fix: `derive_items_mask`, out of line, one instance per cache size with the line mask a constant. Measured the era agent's way (readwidth's binary beside the fixed one, same input, same minute, 22:07 UTC): v2 0.607 / 0.610 against 0.609 / 0.611; on the fixed binary x4 1.238 / 1.237 (2.0x), x8 2.077 / 2.058 (3.4x), worst cold 2.15 ms; the increments (+0.63, +1.46 ms per unit) equal the slow binary's. Lesson, the class: an inlined item loop costs 2.2x and nothing in the suite sees it; a verifier benchmark with a pinned bound in the crate's CI is filed for the next cut, and until then every change to the item loop is measured against the previous binary on the same input in the same minute. PC 1 (job run-mixer-x4-pc1-20261005, 22:00 to 22:04 UTC, the worker's `cache ... dataset ... ms` wall line): RTX 5090 dataset 23 to 25 ms at v2, x4 and x8; RX 9070 XT (gfx1201) 72 to 77 ms at all three; every fingerprint equal to the Mac's; rates the v2 rate (136.5 to 137.4 and 18.0 to 18.2 MH/s). Decision under the delegated rule (coordinator, 22:05 UTC): x8 enters class v3 (`V3_CLASS = MX8`); pinned packs mx8-genesis (7c28cfb06c5c65a9) and mx8-devnet-epoch0 through the chain path with the era inside (90f794dd556f7a3b, Metal and Apple OpenCL, 22:12 UTC); the x4 packs kept as the candidate's record. Chip headline at x8: 41.7 MH/s, 0.31x bare, 0.92x with the 3x factor, 0.76x at equal silicon (`docs/analysis/chip-model-v3.md`). + ## 5 October 2026 (evening), EVM transaction relay: three nodes in a chain, every transaction sent to one end included by the other two miners (execution and networking engineer) Until this change the node did not relay EVM transactions to its peers, so a transaction sent to one node was only ever included by that node's own templates (this file, "5 October 2026 (afternoon), live devnet: real transactions": 3,794 transfers, all in the Mac's blocks; execution-layer ledger item 9). Fork branch `tx-gossip` (worktree `vendor/igneum-node-txgossip`, from release-0.3.6 a24ab01a, commit e242acd0), main repo branch `tx-gossip`. Design in `docs/design/execution-layer.md` 1.4 "Relay"; the hand-out cooldown of its 10.2 table is gone with it (row "Mempool hold"). @@ -1586,3 +1728,236 @@ The 3-node run (`tools/txgen/relay-net.mjs`, new; this Mac, load 7 to 8 at the e Reading. Every transaction given to A was mined by B or C within 6 s, two thirds of them within 3 s, with no skipped copy: the hold on block-added kept B's and C's parallel blocks from carrying the same transfer. The afternoon run on the live devnet, through one node with the cooldown, had p50 40.7 s and p90 110.8 s with 50-s quiet stretches; here the 1.5 s p50 is one fast-time block plus the relay and the executor's lag. The pool depth matching on all three nodes at every sample is the convergence. Not measured here: a transaction flood above the per-peer rate (the bucket is unit-tested only), a 13 peer in the fleet (the digest check and the version gate are the evidence), and the hold's 30-s expiry on a block that never reaches the chain (not seen in 149 chain blocks). Commands: `IGNEUMD=vendor/igneum-node/target-txgossip/release/igneumd IGNEUM_MINER=vendor/igneum-node/target-txgossip/release/igneum-miner tools/lock/with-lock.sh run node tools/txgen/relay-net.mjs --rate 2 --duration 120 --wallets 16 --fund 2`; the Mac binaries from the fork worktree with `CARGO_TARGET_DIR=vendor/igneum-node/target-txgossip cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` under the build lock (an APFS clone of `target-036`, 2 min 15 s to clone, 5 min 06 s to build); the suites with `node tools/build-job.mjs run --target 1ccfe586 --node vendor/igneum-node-txgossip --targets linux --node-tests "igneum-exec kaspa-p2p-flows" --no-app`. + +## 5 October 2026 (night), the SP1 CPU prover on PC 1 beside the miners, and the backend survey: no zkVM proves on AMD (amd-prove agent) + +Josh, 22:50 BST: "test proving on the amd card?" and "can we test proving on mac?". The analysis with the backend table and the tier consequences: `docs/analysis/amd-proving.md`. The survey (SP1 v6.8.1 and `dev` 318dd530 of 28 Sep 2026, RISC Zero, Jolt, OpenVM, ICICLE, sppark; every claim cites a file or page there): on 5 October 2026 no zkVM proves on an AMD GPU; Apple silicon has RISC Zero's shipped Metal prover and ICICLE's Metal backend; SP1, the prover here, is CPU-only off NVIDIA. + +Machine: PC 1 (machine ae432dc7), Windows 11, WSL2 Ubuntu 24.04 as root, 16 cores and 46,994 MB visible to the VM, the Igneum Miner app 0.3.9 mining on the RTX 5090 and the RX 9070 XT throughout (the 5090 at 89% mean utilisation, 59 to 70% minimum, from a 1-s `nvidia-smi` sampler under the run: the job never touched a card). Signed `run` job `cpu-prove-pc1-small2` (`tools/amd-prove/pc1-cpu-prove.ps1`), 20:49:00Z to 20:59:49Z: the hosted package `igneum-prove-wsl2-pv1b.zip` (sha256 df50dee5...) built WITHOUT the `cuda` feature (6 s warm; the first job `cpu-prove-pc1-small` built it cold in 126 s), `--mode id` the pinned pair (shard `0x2b1a81cb...`, aggregator `0x474678f3...`, pinned 2026-10-05T16:20:38Z), `SP1_PROVER=cpu`, `--mode shard --shard 0` under `/usr/bin/time -v`. Log: `node tools/jobs.mjs cpu-prove-pc1-small2 --all`. + +| Fixture | SP1 cycles | Setup s | Core prove s (bytes, verify s) | Compressed prove s (bytes, verify s) | Wall s | Peak RSS | CPU | +|---|---|---|---|---|---|---|---| +| block-56-transfers-3shards shard 0 (200 pgas, one transfer) | 315,479 | 22.75 (client 19.46, shard keys 1.85, aggregator keys 1.44) | 82.5 (7,310,257, 0.210) VERIFIED | 199.2 (1,272,897, 0.035) VERIFIED | 312.1 | 29,503,652 kB (29.5 GB) | 978% (9.8 of 16 cores), user 2,516 s, system 537 s, load max 11.3 | +| block-78-increment (2 transactions, 1 executed 1 skipped) | 631,127 | 21.75 | 87.0 (7,317,857, 0.209) VERIFIED | 202.3 (1,272,897, 0.034) VERIFIED | 322.3 | 30,517,916 kB (30.5 GB) | 979%, user 2,616 s, system 541 s, load max 13.1 | +| block-338-shard1 (one shard at `S_p`, 60.8 M cycles) | | not run: the PC 1 scheduler kept the machine for the Counter ASIC 2.0 gates (21:05Z). Approximate extrapolation: about 29 SP1 shards of 2^21 cycles at about 80 s each, 40 min of core proof, then hours of recursion; floor from the 5090's ratios (6x core, 4x compressed, block-78 to `S_p`): 9 min core, 13 min compressed | | | | | | + +For comparison (this log): the Apple M5 Max CPU on 4 October, loaded, block-56 shard 0: core 83.1 s, compressed 272.3 s; on 3 October the v0 guest on block-78: core 22.0 s, compressed 55.7 s. The RTX 5090: block-78 core 1.4 s, compressed 2.7 s (4 October, mining paused); a full shard at `S_p` compressed 10.9 s alone and 33.0 s beside the miner; an empty live shard 7.0 to 7.7 s beside the miner (5 October). No fresh Mac run tonight: the measure lock was held from 20:31Z (a read-width `packbench`, three builds, a 1,500-s proving-v1 network under `run`) and did not free inside the 10-minute window set for it. + +Reading, and the consequences (CLAUDE.md, every number). Doubling the cycles added 4.5 s to the core proof and 3.1 s to the compressed proof: about 280 s of a CPU proof is fixed cost in the compressed-proof recursion, so no shard size brings a CPU proof under the launch deadline (20 to 60 s behind the tip) or near the 10-s assignment window; it fits only the v1 unproven deadline (600 s), which pays a CPU prover only when no card has proven the shard in 10 minutes. The 29.5 to 30.5 GB peak RSS means the CPU prover needs 32 GB free: a 64 GB Windows PC (WSL2 takes half the host's RAM by default), a 32 GB Linux machine, a 64 GB Mac; a 16 GB machine cannot run it at all. Per tier: an AMD-only home miner (8, 12 or 16 GB, Windows or Linux) mines and does not prove, and loses the 20% proving-pool share; Apple silicon the same (the M5 Max mines at 26.7 MH/s, this log, 4 October); a mixed rig proves on its NVIDIA cards and the rig installer's `prover_decision` already skips every non-NVIDIA card (`packaging/linux/bin/igneum-rig-lib.sh`, branch `rig-install`), now a stated requirement; the app's `provedefault.rs` already keeps proving off on Apple silicon and off without an NVIDIA card. Decision asked of nobody: no CPU tier (the analysis, section 4a); the public line for the site, litepaper and Proving tile is in section 4c ("Proving needs an NVIDIA card with 16 GB or more today ... AMD and Apple cards mine. A prover for them lands when a zkVM ships one"). The first job proved nothing because an apostrophe inside a single-quoted awk program ended the quote and bash refused the loop while the job reported exit 0; the class fix is `tools/amd-prove/check-job-bash.sh` (`bash -n` on the embedded bash body before publishing) and the same `bash -n` inside the job before the run, both shown to refuse the bad body and pass the fixed one. + +## Counter ASIC 2.0, the numbers + +5 October 2026 (night). The chip-resistance layers measured on the three cards we own (Apple M5 Max, RTX 5090 on PC 1 and PC 2, RX 9070 XT on PC 1's eGPU), the decisions taken under Josh's delegated rules for the devnet, and the chip model before and after. Every number is from an entry above or from the plan documents named; approximate is marked. Levels: `docs/plans/counter-asic-2-public.md`. + +**Program class v3 (the devnet, activation by height switch `program_class_v3_activation_daa`)** = class v2's 128 x 4-byte loads, the era draw of the table layout and the working-set windows (layers 4 and 8), the cache growth rule (layer 6, option C: the cache doubles when the dataset doubles), the mixer at x8 (M16's multiplier), reserve family R1 (integer matrix, switched off) and the epoch length as a signalled reserve parameter (layer 9, 3,600 DAA s until a 90% signal). Not adopted on the measurements: wider reads (layer 1), the per-load width mix (layer 2), the per-warp write scratch (layer 3), the hot table (layer 5). + +| Card | v2 MH/s | v3 MH/s, six eras (spread) | Bytes per hash | Latency-bound share | Daily 1 GiB build, v2 / v3 | +|---|---|---|---|---|---| +| Apple M5 Max, Metal | 27.68 | 27.85 to 27.98 (0.5%) | 512 | 1.06 | 21 / 21 ms | +| RTX 5090, CUDA | 137.2 | 135.90 to 137.70 (1.3%) | 512 | 1.01 | 25 / 23 ms | +| RX 9070 XT, OpenCL | 18.09 | 18.59 to 19.18 (3.1%) | 512 | 0.95 | 74 / 75 ms | + +CPU verifier, one M5 Max core at load average 5.5 (the fixed crate, ca2-mixer 1ab8b21): v2 0.61 ms per warp, v3 (x8) 2.08 ms (3.4x), worst cold 2.15; the 10 ms gate holds 4.8x (4.6x on the worst cold unit). Bit-exact: every v3 pack's fingerprint equal on Metal, Apple OpenCL, CUDA and AMD OpenCL. + +| Layer | Measured | Decision | The number | +|---|---|---|---| +| 1 wider reads | w16 139.8 / 17.90 / 28.26 MH/s (5090 / 9070 XT / M5 Max) against v2 136.1 / 18.15 / 27.74; w64 71.9 on the 5090 (share 0.58, 37% of its stream) | out: keep 4 B | the 9070 XT does 2.4 G dependent reads/s at every width; wider reads make the 5090 bandwidth-bound | +| 2 width mix per load | spread over six programs 18.8 / 7.4 / 11.3% and 22.3 / 5.5 / 8.1% | out | the 5% rule | +| 3 write scratch | GPU cost 12 to 48% at 32 and 128 KB per warp; the on-die-cache chip 2.4x at every share | out (the construct is sound; its tests stay) | the verifier resets the scratch per unit, so a chip keeps it in 80 to 320 B per lane | +| 4 and 8 era layout and windows | six-era spread 1.3 / 3.2 / 0.8% | in | under the 5% rule; the SRAM mirror a chip needs is the whole dataset every hour | +| 5 hot table | added form g 0.87 / 0.85 / 0.84 (5090), 0.84 / 0.81 / 0.80 (9070 XT) at 32 / 64 / 96 MiB | out (a 3.0 option) | no card keeps 32 MiB resident while the dataset streams; the replaced form helps the chip | +| 6 cache schedule | the 256 MiB mirror is 128 mm^2 and $46 at N5 by shipped cache-die density, approximate | option C, in | the cache's job is to stay above GPU L2 (96 MB on the 5090, 128 MB on GB202) | +| 7 integer matrix | dp4a 1.17x a step on the 5090, 1.06x on the 9070 XT, 1.6x emulated on Apple; mm8 native on all three as a tile | reserved R1, off | unlock at era 4 or 90% signal | +| M16 mixer | x4: verifier 1.24 ms, chip 1.84x with the allowance; x8: 2.08 ms, 0.92x; the daily build unmoved on every card | x8 in | the only lever that moves the named chip | +| 9 epoch length | compile-ahead 0.5 s (M5 Max, race off), 1.0 s (5090), 38 s with the race; FPGA compiles 42 to 160 min (PRflow, FPT 2019) | reserved, 600 s to 2 h by signal | at 600 s a per-program bitstream mines 0% of each epoch | + +**The chip model, before and after** (`docs/analysis/chip-model-v3.md`, `docs/analysis/sram-mirror.md`): the strongest chip we can name holds the whole 256 MiB cache on-die (about 128 mm^2 and $46 of silicon at N5, approximate) and computes dataset items on the fly at 50 T integer op/s. Against the RTX 5090's measured 136.1 MH/s: class v2 333 MH/s, 2.4x; class v3 41.7 MH/s, 0.31x bare, 0.92x with a 3x fixed-function allowance (approximate), 0.76x at equal silicon. The claim is "under 2x"; the margin is thin on the allowance (3.3x reads 1.0x) and 9% on the budget. Next levers, named: the mixer at x16 (the verifier at about 4 ms per warp; a 2019-class core unmeasured), a hot table small enough to stay resident beside the streaming dataset. + +**The user tiers.** AMD RDNA 4 sits at about a seventh of a 5090 on this hash (its dependent-read rate: 2.4 G against 17.5 G per second), 2.2x worse per pound at list prices and 4.9x worse per watt (approximate); the card's memory system, not a tuning gap. The integrated tier on the CUDA and OpenCL one-click workers mines v3 with a restart per epoch until per-day dataset reuse lands (0.3.12). Card lifetime under the step schedule: a 4 GB card to year 4, 8 GB to year 12, 12 GB to year 28 with the cache freed after the daily build. + +**The bounty.** A bounty for any chip design beating a GPU by more than 2x on the published model, with a leaderboard by card model, follows the external review (spec O-1.17, January 2027); it is named publicly only once escrowed (`docs/plans/funding.md`, rule 3), which it is not yet. +## 5 October 2026 (evening), proving v1: segment records, the chain rule, the unproven rule; what was measured tonight (proving engineer) + +Branches `proving-v1` (main repository, worktree `igneum-wt-proving-v1`; fork `vendor/igneum-node-pv1` from a24ab01a). Rules: spec 7.8; plan `docs/plans/proving-v1.md`. Every row names its command. The live devnet was in a degraded state the whole evening: from 18:35Z the RTX 5090 workers on PC 1 and PC 2 exited at start on a pack seed mismatch (`the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT`, restart 60+ on PC 2 by 19:05Z, another agent's branch `pack-loop`), the Mac app node was down from 17:45Z, so PC 2 mined 3.4 MH/s from its iGPU and PC 2's prover was the only prover; the coordinator held every PC 2 measurement at 19:00Z until the fleet mines again. + +### Step 1, the prover default and its cost + +| What | Measured | +|---|---| +| The default rule (`app/igneum-app/src/provedefault.rs`) | `cargo test --release -p igneum-app provedefault` on this Mac (the app crate, build lock, 19:05Z): 5 passed (a 5090 with WSL2 on Windows is on; Windows without WSL2 off with the Set up hint; Linux needs no WSL2 and the 12 GB gate holds, a 10 GB 3080 and a 16 GB AMD card stay off; Apple silicon off; the biggest qualifying card is named) | +| Mining alone against mining with the prover, first try (PC 2 job `prover-cost-pc2-pv1`, `tools/proving-v1/pc2-prover-cost.ps1`, published 18:40:48Z, ran 18:41:13Z) | VOID: the job waited 20 min for the 5090 worker to hash and it never did (the pack fault above); "mining alone" was 0 MH/s | +| Mining alone against mining with the prover, the re-run after the coordinator's go (job `prover-cost-pc2-pv1b`, ran 19:20:36Z to 19:51:17Z; the 5090 worker restored at 19:16Z and hashing throughout; prover OFF by `POST /api/prove {"on":false}` 19:40:39Z, back ON 19:46:09Z, left on). The job's own `/api/state` samples stayed empty on PC 2 (`Invoke-RestMethod` returns an object PowerShell 5.1 cannot walk, `cards=0`, the fix is for the next run), so the hash rate is read from the miner's own STATUS lines (`miner-nvidia-1ccfe586-1` uploads, `now=... MH/s wall`, one every 30 s, the intake table `miner_logs`) | prover OFF, 19:41:09 to 19:46:09Z: n 10, mean 124.72 MH/s, p50 124.81, min 124.10, max 125.38. Prover ON, 19:46:39 to 19:51:13Z: n 9, mean 119.74, p50 118.87, min 118.08, max 123.42. The 15 min before the job with the prover on (19:25 to 19:40Z): n 30, mean 119.88, p50 118.79. So the prover costs the 5090 5.0 MH/s, 4.0% of its hash rate, while it proves the devnet's empty shards one after another (1.4 a minute here: the node's paidShards 559 -> 566 over the 5-min phase). A full shard at `S_p` keeps the card busier (the 4 October run proved one in 10.9 s); the cost at that load is the chain job's row | +| GPU memory during proving, first try (phase B of the first job: the prover on for 5 min, 298 one-second `nvidia-smi --query-gpu=memory.used` samples, the 5090 worker dead so the card held nothing else) | memory.used min 1,654 MiB, max 13,816 MiB, utilisation mean 2.7%, power max 190.6 W: the prover alone on empty shards | +| GPU memory with the miner AND the prover on the card (the re-run's phase B, 298 one-second samples, 19:46 to 19:51Z) | memory.used min 3,396 MiB (the miner's dataset and program resident), max 15,590 MiB, utilisation mean 92.9%, power max 328.6 W. So the prover's own peak is about 12.2 GB on an empty shard (15,590 minus the miner's 3,396), and the two together need 15.6 GB: a 16 GB card (5080, 9070 XT class, if it had a CUDA path) sits 0.4 GB under tonight's peak with no room for a full shard, a 24 GB 4090 has 8.4 GB of headroom, a 12 GB card cannot mine and prove at once on this build. The full-shard peak is the chain job's row | +| Shards per minute with the mining worker dead | the node's `paidShards` 510 -> 518 over the 5-min phase: 1.6 shards a minute from one 5090 through the app's loop (export, cut, prove, sign, submit) | +| Host RAM (Windows `Win32_OperatingSystem` and the `vmmem` working set, sampled every 15 s) | host used 25,550 MB of 63,132 MB at the end; the WSL2 VM's working set 7,915 MB (2,334 MB used of 30,914 MB inside the distribution) | +| The SP1 GPU server's compiled targets (`cuobjdump --list-elf /root/.sp1/bin/sp1-gpu-server` inside PC 2's Ubuntu-24.04, CUDA 12.8, driver 610.47) | `sp1-gpu-server` 6.8.1 (251,306,680 bytes, sha256 c2642ad1c42e85d8525159cf0c7cd5200d8766c9be1283f452a1f9bf9fea725c, the asset `sp1_gpu_server_v6.8.1_x86_64.tar.gz` the SDK downloads, `sp1-cuda-6.8.1/src/server.rs`): one ELF each for sm_80, sm_86, sm_89, sm_90, sm_100 and sm_120; `strings` finds compute_120 PTX as well. So sm_89 (Ada: RTX 4090, 4080) is compiled in natively, no JIT; so are Ampere (3090, 3060), Hopper, Blackwell datacentre (sm_100) and consumer (sm_120, the 5090). Nothing for AMD (no HIP path in SP1) | + +### Step 2, aggregated chains + +| What | Measured | +|---|---| +| The new host (`--mode chain`, `aggregate`, `verify-segment`) against every fixture natively | `igneum-prove-host --mode native` on the Mac for the 12 fixtures of `proving/fixtures/` (9 block, 3 fee-switch), host built from this branch 19:06Z: every one MATCHES (the package gate's native half); `--mode id`: shard `0x2b1a81cb...`, aggregator `0x474678f3...`, the 0.3.9 pin, unchanged | +| Eight consecutive live fixtures | `igneum_exportSegments 0x0..0x13cb4` on node 1's exec RPC (127.0.0.1:26790, read-only, 20:06 BST, tip 81,076): 71,042,616 bytes, 81,077 segments, 28 accounts, 0.5 s; `igneum-prove-export export.json block-.json` for 81046..81053: replayed 81,077 segments from genesis in 1.8 s each, every state root equal to the node's; one shard a block, 0 pgas (no transactions on the devnet tonight), `proving/fixtures/chain/` | +| Chain of 2 on the Mac CPU (the known-finished case of `--mode chain` before the GPU; M5 Max under the live nodes, the harness and two builds) | `SP1_PROVER=cpu igneum-prove-host --mode chain --chain block-81046.json,block-81047.json --out results.json` under the run lock, 19:07:48Z to 19:11:28Z: setup 12.2 s; block 81046: shard 0 compressed 55.4 s (1,272,897 bytes, verify 0.036 s), aggregate 52.0 s (1,272,909 bytes, verify 0.031 s), chain_len 1, agg_vk zero; block 81047: shard 41.3 s, aggregate WITH the previous block proof 59.1 s, chain_len 2, agg_vk = the pinned aggregator id; end to end 207.9 s; final proof 1,272,909 bytes, statement 0x232276f4... The recursion over the previous proof cost 7 s more than the first aggregation on this CPU | +| `--mode verify-segment` on that proof (the node's path: SP1 light verifier, pinned aggregator key) | VERIFIED in 0.032 s (0.27 s wall, three runs: 0.033, 0.032, 0.032); known-failed: a wrong statement NOT VERIFIED (0.032 s); the shard verifier (`--mode verify`) on the segment proof NOT VERIFIED, "program id 0x474678f3... IS NOT OURS 0x2b1a81cb..." | +| Chain of 8 on the RTX 5090 (N = 2, 4, 8), job `chain-pc2-pv1b` (`tools/proving-v1/pc2-chain.ps1`; the package `igneum-prove-wsl2-pv1.zip` eb6dccf8..., 1.5 MB, fetched by `fetch-prove-pv1` 19:51Z; the first try `chain-pc2-pv1` died in its own export step, fixed) | Ran 19:58:37Z: the export from PC 2's node (72,901,414 bytes, 1.4 s), the host built in WSL2 against the live build's warm target dir in 6 s and installed to `/opt/igneum-pv1` (the live `/opt/igneum` host untouched, sha 29cc4768...), `--mode id` the pinned pair; eight consecutive fixtures 83346..83353 cut, every one MATCHES natively. The chain on the GPU (SP1_PROVER=cuda, the miner mining on the same card at 119 MH/s): setup 12.7 s; block 83346: shard 7.4 s, aggregate 7.6 s (chain_len 1), 15.1 s; block 83347: shard 7.2 s, aggregate WITH the previous proof 9.5 s (chain_len 2, agg_vk the pinned aggregator id), 16.8 s, cumulative 31.8 s over 2 blocks; block 83348: shard 7.0 s, then at 20:01:09Z the app quit and aborted the job ("quit: stopping the miners, then the node", then "job chain-pc2-pv1b: aborted (the app is quitting)"; NOT an update: nothing of 0.3.10 was published; the log gives the quit no source; 20 s earlier the efficiency sweep's administrator prompt had been cancelled at the keyboard, and 13 s earlier the live prover had failed with "CudaClientError: Connect(PermissionDenied)", the root-owned socket my job had left, below). So N = 2 measured: 31.8 s of GPU time for two empty blocks, the chained aggregation 1.9 s dearer than the first; N = 4 and 8 are the re-run `chain-pc2-pv1c` after the restart. An empty shard's compressed proof on the 5090 is 7.0 to 7.4 s (the 200-pgas shard of 4 October took 2.7 s with the card to itself; tonight the miner held it at 92% utilisation) | +| The chain of 8, the third run `chain-pc2-pv1c` (20:05:21Z to 20:08:33Z, after the app restart; blocks 83616..83623 from PC 2's node at tip 83646, the same script; results `tools/proving-v1/chain-pc2-2026-10-05.json`) | Build 5 s (warm), eight fixtures cut and MATCHING natively, setup 15.7 s, then on the GPU with the miner mining on the same card: shard proofs 7.3 to 7.7 s each (8 x, 59.5 s), aggregations 7.9 s for the first block and 9.6 to 9.7 s for every chained one (75.5 s), every proof VERIFIED, end to end 135.6 s for 8 blocks (17.0 s a block from the second on). Cumulative: N = 2 at 32.6 s, N = 4 at 66.8 s, N = 8 at 135.6 s. The final proof is 1,272,909 bytes whatever N (chain_len 8, agg_vk the pinned aggregator id), the record 586 bytes; `--mode verify-segment` on it: VERIFIED in 0.039, 0.037, 0.040 s after a 0.26-s light-verifier setup, the same three runs each time. GPU memory over the chain (152 one-second samples): max 16,751 MiB with the miner's 3.4 GB resident, so the chained aggregation holds about 13.4 GB, 1.2 GB over the shard-only peak; WSL used 2,456 MB | + +Reading the chain numbers. Aggregation is a fixed cost per block (9.7 s here), not per segment: the recursion verifies one more proof whatever `chain_len`, so the record for N blocks costs N aggregations and the verifier one. Against 4 October with the miner stopped (aggregate 2.2 to 2.5 s, a 200-pgas shard 2.7 s), tonight's 9.7 s and 7.3 s say the miner's 92% utilisation slows the prover about 3 to 4x while the prover slows the miner 4%: the card is shared, and the lottery wins the arbitration. A machine that mines and proves at once delivers one empty block's proof and aggregation in 17 s; one that only proves, about 5 s (approximate, from the 4 October stages). + + +### Step 3, coverage + +| What | Measured | +|---|---| +| A 3-minute window at 18:57Z on node 1 (`node tools/proving-v1/coverage.mjs --minutes 3`, chain blocks 80754..80839, 86 blocks) | 4 blocks with a paid shard (4.7%), 4 fully proven, 4 of 86 shards; on-chain latency (carrier timestamp minus block timestamp) n 4: min 36 s, p50 39 s, max 44 s; 0 content blocks. One prover (PC 2), the Mac verifier node down, PC 2 producing few blocks (3.4 MH/s): the degraded state above, not the fleet's number | +| A 30-minute window, 19:13 to 19:43Z, the degraded fleet (PC 2 the only prover, its 5090 worker restored at 19:16Z, the Mac app node down by decision: the Mac app is attached to node 1) | `node tools/proving-v1/coverage.mjs --minutes 30 --watch` on node 1: chain blocks 81236..82668, 1,433 blocks; 38 with a paid shard (2.7%), all 38 fully proven (one shard a block, 0 content blocks); on-chain latency n 38: min 36, p50 44, p90 52, p99 62, max 65 s. The live page's 10-minute proving object read 0 shards and 0 provers at 19:42Z (it counts what its own node verified; that node is the Mac app node, down), so the chain's own count is the number | +| A 30-minute window with the fleet mining (PC 2 at 119 MH/s from 19:16Z, PC 1 at 128.8 from 19:18Z; PC 2 still the only prover, its prover OFF for the 5 min of the cost job's phase A inside this window; the Mac app node down by decision) | `coverage.mjs --minutes 30 --watch`, 19:21 to 19:51Z on node 1: chain blocks 81644..83069, 1,426 blocks; 34 with a paid shard (2.4%), all fully proven (one shard a block, no content); on-chain latency n 34: min 38, p50 44, p90 51, p99 52, max 53 s. One 5090 through the app's loop as it is covers 2.4 to 2.7% of the blocks; the latency from block to carried record is 44 s at the median, under the litepaper's minute, and would be the same for every block if the fleet were 40 cards (the table below) | + + +### Step 3, the fleet size (arithmetic from measured inputs; every input names its entry) + +Inputs, all RTX 5090 (PC 2), SP1 6.8.1 cuda: a full shard at the provisional `S_p` (6.75 M pgas) compressed in 10.9 s and the four shards of a near-`B_p` block in 10.2 to 10.7 s each (bench-log 4 October 2026, "shard proving on the RTX 5090", runs run-20261004-173115 and run-20261004-r3-shards); one aggregation 2.2 s (two shards) to 2.5 s (four shards), the same entry; tonight's chain of 2 on the Mac CPU shows the recursion over the previous block proof costs the same order as a first aggregation (52.0 s against 59.1 s), so the GPU figure for a chained aggregation is taken as 2.5 s, approximate, until the held PC 2 chain job measures it; the app's live loop tonight: 1.6 shards a minute per card on empty shards (export, cut, key setup, prove, sign, submit: about 37 s a shard, of which the proof is a few seconds), bench-log step 1 above. A 5090 proves one thing at a time. + +| Block content at 1 block/s | Shard proofs a second (fleet) | Card-seconds a second for shards | Aggregations a second | Card-seconds a second for aggregation | 5090-class cards for 100% | Rule | +|---|---|---|---|---|---|---| +| empty blocks (tonight's devnet), the app's loop as it is, the card also mining | 1 | 37 | 1 | 9.7 (measured, `chain-pc2-pv1c`) | 47 | one shard per block, the loop's 37 s each plus a chained aggregation | +| empty blocks, the chain mode's shape (one key setup per process, proofs back to back), the card also mining | 1 | 7.4 (measured) | 1 | 9.7 (measured) | 18 | 17.1 card-seconds a block, `chain-pc2-pv1c` | +| empty blocks, cards that only prove | 1 | 2.7 (4 October, a 200-pgas shard) | 1 | 2.5 (4 October) | 6 (approximate) | the miner's 92% utilisation costs the prover 3 to 4x | +| one full shard a block (`S_p`, 6.75 M pgas), cards that only prove | 1 | 10.9 | 1 | 2.5 | 14 | 4 October's stages | +| one full shard a block, the card also mining | 1 | about 35 (approximate: 10.9 x 3.2, tonight's ratio) | 1 | 9.7 | about 45 (approximate) | the full-shard proof with the miner on the card is not measured | +| blocks at `B_p` (four full shards), cards that only prove | 4 | 42.5 | 1 | 2.5 | 45 | 4 x 10.6 + 2.5 | +| at the adopted v1 budgets (`B_p` 120,000 pgas, `S_p` 30,000, from DAA 210,000 on the devnet): a v1 shard of transfers ran at 213 to 236 cycles per pgas (bench-log 5 October, "the prover carries both fee tables"), 7 M cycles a shard against 60 M for the prototype shard | 4 | under 42.5 (the 5090 time for a 7 M-cycle shard is not measured; scaling 10.9 s by cycles gives about 1.3 s, approximate) | 1 | 2.5 to 9.7 | 8 to 15 (approximate) | measure before the switch lands | + +Reading. The card count is the sum of card-seconds of work per block-second, rounded up, with no slack for the exclusive window, the relay or a card's idle gaps; the devnet's own numbers tonight (one card, 1.4 to 1.6 shards a minute, 2.4 to 4.7% of blocks) are the first row. Two levers, both measured tonight: the loop (a shard's carriage through export, cut and a 12-s key setup is 25 s on top of a 7-s proof; the host's `--mode aggregate` and `--mode chain` hold one key setup per process and the prover loop should do the same, the 0.3.11 item in the plan) and the card's other job (a mining card proves 3 to 4x slower than an idle one, `chain-pc2-pv1c` against 4 October; the prover's cost to mining is 4%). A fleet of 18 mining 5090s, or 6 proving-only ones, covers an empty-block chain at 1 block/s through the chain mode; the mandatory rule waits for the measured share to reach one, not for these rows. + +### The 12 GB requirement (Josh, 20:1xZ: "make sure we can prove on 12gb cards"): the GPU memory peak against SP1's knobs + +Job `memsweep-pc2-pv1` (`tools/proving-v1/pc2-memory-sweep.ps1`), PC 2's RTX 5090 (32,607 MiB), the miners STOPPED by the job and the live prover switched off (its `sp1-gpu-server` would otherwise be the one the client connects to), every row: the server killed first, a 1-s `nvidia-smi memory.used` sampler, one `--mode compressed --shard 0` run of the pv1 host (`/opt/igneum-pv1`, SP1 6.8.1 cuda, `sp1-gpu-server` 6.8.1), 20:19 to 20:25Z. The knobs are the environment the GPU server inherits from the host process (`sp1-core-executor-6.8.1/src/opts.rs`: `SHARD_SIZE`, `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, `MINIMAL_TRACE_CHUNK_THRESHOLD`, `TRACE_CHUNK_SLOTS`; `sp1-prover-6.8.1/src/worker/config.rs`: the `SP1_WORKER_NUM_*` and `*_BUFFER_SIZE` counts, defaults 4 core workers, 8 recursion prover workers). Idle card before the sweep: 1,732 MiB. + +| Config (environment) | Fixture | Cycles | Peak MiB | Compressed prove s | Verified | +|---|---|---|---|---|---| +| baseline (no knob) | block-338-shard1, a full shard at `S_p` (6.75 M pgas) | 60,415,376 | **28,295** | 11.4 | yes | +| baseline | block-83616, an empty live shard | 280,706 | **13,863** | 2.3 | yes | +| ELEMENT_THRESHOLD 2^27 | full shard | 60.4 M | 28,326 | 10.9 | yes | +| ELEMENT_THRESHOLD 2^26, HEIGHT_THRESHOLD 2^21 | full shard | 60.4 M | 28,326 | 10.7 | yes | +| every worker count and buffer 1 | full shard | 60.4 M | 28,326 | 20.8 | yes | +| every worker count and buffer 2 | full shard | 60.4 M | 28,327 | 12.9 | yes | +| workers 1 + ELEMENT 2^27 | full shard | 60.4 M | 28,263 | 20.3 | yes | +| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 | full shard | 60.4 M | 28,326 | 20.6 | yes | +| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 + trace chunks 4 M x 2 slots | full shard | 60.4 M | 28,358 | 22.6 | yes | +| workers 1 + ELEMENT 2^25 + HEIGHT 2^20 | full shard | 60.4 M | 22,919 | 22.2 | yes | +| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 | empty shard | 280,706 | 13,861 | 2.6 | yes | + +Reading. The GPU memory of a compressed shard proof is **13.9 GB for a shard of 280,000 cycles and 28.3 GB for one of 60 M cycles**, and no knob the environment carries moves the floor: the worker counts only slow the proof (11.4 s to 20.8 s), the trace thresholds at 2^26 and 2^27 change nothing, and the smallest trace threshold tried (2^25 elements, 2^20 rows) takes 5.4 GB off the full shard (22.9 GB) at twice the time. The floor sits in the GPU server's own allocation, not in the shard: an empty shard with every knob at its minimum still takes 13.9 GB. So on SP1 6.8.1's `sp1-gpu-server` as shipped, **a 12 GB card cannot prove even an empty shard** (13.9 GB), and the 11.0 GB target of tonight's requirement is out of reach from the environment. The S_p/2 and S_p/4 cuts of block 344 did not run: the package carries no `tools/prove-fixtures/seq.json` (the cut rows need the export; they would sit between the two measured points, and the floor is the binding number anyway). What is left to try, in order: the server's own options (its `--help` and the option names in its strings: the miner-on job prints them), SP1's core-only proof (the node needs the compressed proof, so this changes the protocol), and an SP1 release built for smaller cards (the 6.8.1 release notes are not read here; approximate: the project's documentation names 24 GB as the GPU requirement, `proving/windows-wsl2/setup-wsl.sh` quotes it). + +### The same shard with the miner running (the 16 GB requirement), and the GPU server's own options + +Job `memminer-pc2-pv1` (`tools/proving-v1/pc2-memory-miner-on.ps1`), 20:28 to 20:30Z, the miner at full rate on the card, the live prover off for the run, the same 1-s sampler: the full shard at `S_p` (60.4 M cycles) peaked at **30,039 MiB** and took 33.0 s (28,295 MiB and 11.4 s with the card to itself: the miner costs the prover 2.9x in time and 1.7 GB of memory); the empty shard **15,670 MiB** and 7.7 s (13,863 and 2.3 s alone). So a 32 GB card mines and proves the prototype shard with 2.5 GB to spare; a 24 GB card cannot prove it even alone (28.3 GB); a 16 GB card cannot hold even the empty shard beside the miner (15.7 GB, the display and driver on top). `sp1-gpu-server --help` prints only `--version`: it has no options of its own, and its strings carry no memory setting (`CUDA_OUT_OF_MEMORY` is an error name). The shard SIZE is therefore the only lever left on this build, measured next as the S_p curve. + +The root-socket fault (the class, fixed the same evening). The chain and memory jobs ran the host as root inside WSL2; the first `sp1-gpu-server` they started left `/tmp/sp1-cuda-0.sock` owned by root, and the live prover (the app's own WSL user) then failed every shard with `CudaClientError: Connect(Os { code: 13, kind: PermissionDenied })` (PC 2 app log 1791230456, 20:00:56Z) until the socket was gone. Every pv1 playbook now kills the server and unlinks `/tmp/sp1-cuda-*.sock` at its start and end, `tools/ci/prover-socket-check.sh` fails CI on any playbook that runs a prove mode as root without both lines (shown failing on `pc2-prover-cost.ps1` before its `--mode id`-only exemption, passing after), and the plan carries the rule: a prover job on a shared card runs as the app's user or cleans its socket. It recurred at 21:25Z from another agent's job (agg-cost-pc2-1, the same root-run shape) and survived the 0.3.10 restart at 21:49:41Z; the fix job `socketfix-pc2-pv1` (`tools/proving-v1/pc2-socket-fix.ps1`, 22:01:14 to 22:02:12Z) found `/tmp/sp1-cuda-0.sock` owned by root, removed it, switched the prover off and on, and the app's next shard (block 89011 shard 0) was proven and submitted in 34 s and paid 0.93 IGN at 22:02:24Z. Playbooks that run the host: `tools/proving-v1/pc2-chain.ps1`, `pc2-memory-sweep.ps1`, `pc2-memory-miner-on.ps1`, `pc2-sp-curve.ps1` (all root, all with the cleanup now; the first two chain and sweep runs had none), `pc2-prover-cost.ps1` (`--mode id` only), `relay/playbooks/shard-test.ps1` and `proving/windows-wsl2/prove-shard.sh`, `prove-block.sh` (the app's user, not root), `tools/proving-v0/run.mjs` (the Mac, no server). + +### The S_p curve: peak GPU memory against shard size against time, the card to itself (the first of the two curve jobs) + +Job `spcurve-stopped-pc2-pv1` (`tools/proving-v1/pc2-sp-curve.ps1`, the miners stopped by the job, the live prover off, the server killed and its socket unlinked around every point, a 1-s `nvidia-smi` sampler), 20:33 to 20:37Z, PC 2's RTX 5090, the pv1 host (this run's `--budget` points were ignored by the pv1 host, so its block-344 rows are the fixture's own 6.75 M-pgas shard 0 twice; the pv1b host's re-plans at 2.25 M and 4.5 M pgas are the next job's rows). Idle card 1,743 MiB. + +| Shard | pgas | Witness bytes | SP1 cycles | Peak MiB, card alone | Compressed prove s | Knob | +|---|---|---|---|---|---|---| +| block 83616, an empty live shard | 0 | 13,964 | 280,706 | **13,874** | 2.2 | none | +| block 56, one transfer | 600 | 4,902 | 556,369 | 13,907 | 3.2 | none | +| fees-v1-shards2 shard 0, a shard at the ADOPTED v1 budget (`S_p` 30,000; 4 transactions, 2 shards a block) | 22,172 | 18,390 | 4,717,439 | **20,434** | 4.3 | none | +| the same | 22,172 | 18,390 | 4.7 M | 20,435 | 3.7 | ELEMENT_THRESHOLD 2^25, HEIGHT 2^20 | +| block 338 shard 0, the full PROTOTYPE shard (`S_p` 7.5 M) | 6,751,568 | 21,611 | 60,415,376 | **28,307** | 10.8 | none | +| the same | 6.75 M | 21,611 | 60.4 M | 22,963 | 11.5 | ELEMENT_THRESHOLD 2^25, HEIGHT 2^20 | +| block 344 shard 0 (the fixture's own cut, 6.75 M pgas, modexp) | 6,748,392 | 18,535 | 59,678,420 | 28,275 and 28,307 | 11.5 and 11.0 | none | + +The second job (`spcurve-stopped-pc2-pv1b`, the pv1b host whose `--budget` re-plans a fixture, 20:43 to 20:47Z, the same conditions) repeats the points (empty 13,875 MiB 2.1 s; one transfer 13,907 MiB 3.3 s; the v1 shard 20,435 MiB 4.2 s; the prototype shard 28,275 MiB 11.2 s) and adds the re-plans of block 344 (27 M pgas of modexp): at 2.25 M pgas (one transaction, 16 shards a block, 19,987,938 cycles) **28,371 MiB** and 6.6 s; at 4.5 M pgas (7 shards a block, 40,011,108 cycles) 28,307 MiB and 8.5 s; with the 2^25 trace threshold the 2.25 M shard 22,835 MiB and 6.2 s. So the peak is flat at 28.3 GB from 20 M cycles to 60 M (the server's buffers step up between 4.7 M and 20 M cycles and not after), and cutting the prototype shard smaller buys nothing until the v1 size. + +The third job (`spcurve-miner-pc2-pv1`, the same points WITH THE MINER RUNNING on the card, 20:49Z on, the live prover off): empty shard 15,585 MiB and 7.5 s; one transfer 15,745 MiB and 12.7 s; **the v1 shard 22,210 MiB and 13.2 s** (20,435 and 4.2 s alone: the miner adds 1.8 GB and 3.1x); the 2.25 M shard 30,049 MiB and 17.9 s; the 4.5 M shard 29,954 MiB and 26.3 s; the prototype shard 30,083 MiB and 33.3 s. With the 2^25 trace threshold beside the miner: the 2.25 M shard 24,642 MiB and 21.5 s, the prototype shard 24,739 MiB and 38.8 s (24.7 GB: over a 24 GB card by the display's share, and 3.6x slower than the card alone). So beside the miner the adopted shard needs 22.2 GB: a 24 GB card (24,564 MiB) has 2.3 GB spare for it (the number for a 24 GB card is the 5090's allocation pattern on a 32 GB card, so approximate for the card itself), and the prototype shard needs 30.1 GB, the 32 GB card alone. + +Reading, with the miner-on pairs above (empty shard 15,670 MiB, full prototype shard 30,039 MiB). The witness is never the binding term (4.9 to 21.6 KB a shard); the GPU server's working set is: a floor of 13.9 GB for any shard, 20.4 GB at 4.7 M cycles, 28.3 GB at 60 M cycles (23.0 GB with the smallest trace threshold, at the same time). By card: a **12 GB card proves nothing** on this build (the floor is 13.9 GB alone); a **16 GB card proves only empty and near-empty shards, alone** (13.9 GB; 15.7 GB beside the miner leaves nothing for the display); a **24 GB card proves the adopted v1 shard alone** (20.4 GB) and, at the miner's measured 1.7 GB extra, about 22.1 GB beside it (approximate: not measured on a 24 GB card), and never the prototype shard (28.3 GB); a **32 GB card proves the prototype shard beside the miner with 2.5 GB spare** (30.0 of 32.6 GB). The devnet is on the prototype table until H = 210,000 (6 October, about 19:50Z) and on the adopted v1 table (`S_p` 30,000 pgas) after it, so from H the 24 GB tier joins the provers and the shard that binds the memory is the 4.7 M-cycle one. Shards per block at each size: 1 at the prototype `S_p`, 4 at `B_p`; at the v1 budget 1 to 4 (one a block on tonight's chain, 2 to 3 on the txgen blocks). + +### Step 4, the rule + +| What | Measured | +|---|---| +| Unit tests | `cargo test --release -p kaspa-consensus-core -p igneum-exec --lib -- proving config::params::tests::override_params_carry_the_proving_v1 config::params::tests::consensus_digest` on this Mac (target `vendor/igneum-node/target-pv1`, 19:09Z): consensus core 13 passed (the segment record round trip, signature and the three nested sections; the credit split; the params switch and the digest that moves only once the switch is set), exec 8 passed (the segment grid and the split; the record checks: alignment, block, chain length, the veto naming the field, the deadline, the window; the chain rule both ways; the unproven restart; the shard side at 90%; the pool offering the segment section). The six full node suites go to PC 2 as a build job when the fleet is back | +| The fast-time 3-node harness (`tools/proving-v1/net.mjs`, 29950+, suffix 956, every node in trust mode, three vmine voters, v0 at DAA 60, v1 at DAA 120, 4 blocks a segment, unproven after 60 DAA, a tenth to the aggregator; fork b177718e built on this Mac) | run 2, 19:13:01Z to 19:16:19Z, under the run lock: PASSED, 21 checks in 197.3 s (`tools/proving-v1/report-2026-10-05.json`). v1 start = chain block 119 on all three nodes; the native statement identical on all three. Known-finished: segment 119..122's fresh-chain record submitted to n1 at t=131.1 s, relayed, verified (trust) and PAID on n0 1.0 s later at chain block 129, 253,611,648,000,000,000 wei = a tenth of the four credits, the same on every node, the payout address holding it. Chain rule: segment 123..126's fresh-chain record refused ("does not chain to segment 119..122 ... proven (record paid at chain block 129)"), the continuing one (chain_len 8) accepted and paid. Known-failed: segment 127..130 left without a record: a fresh-chain record for 131..134 refused while 127..130 was pending ("pending until DAA 191"); at DAA 192 the status read unproven, a late record for 127..130 refused ("unproven: carried after the deadline"), the fresh-chain record for 131..134 accepted and paid with chain_len 4; `segmentsInWindow` proven 3, unproven 1. The shard side: a v1 shard's `shardWei` = 90% of its block's credit. Run 1 (19:10Z) failed in its own tooling (the signer's argument order), fixed. Run 3 on the FINAL fork tree (ece42979 on the 0.3.10 commit 21d4c73c, protocol 15, N = 8 both in the params default and `--segment 8`, the fast-time file's four fields), 20:52:41Z to 20:56:45Z: PASSED, 21 checks in 244.4 s (segments of 8: 119..126 paid in 1.0 s after submission, 127..134 refused fresh and paid continuing with chain_len 16, 135..142 left unproven and skipped, 143..150 restarted the chain) | + + +## 5 October 2026 (night), dp4a-class throughput on the M5 Max: the dot4 emulation against the ALU chain (Counter ASIC 2.0 layer 7) + +Apple M5 Max, macOS 26, branch `ca2-analysis` (base `readwidth` 4badcee). The probes are standalone (no pack, no lottery kernel): `proto-metal/dot4-probe.swift` (built `swiftc -O -o dot4-probe dot4-probe.swift -framework Metal` under `with-lock.sh build`), `proto-opencl/dot4-probe.c` (built `cc -std=c99 -O2 -o dot4-probe-cl dot4-probe.c -framework OpenCL`), both run under `with-lock.sh measure` (exclusive; nothing else built or measured on the Mac during the runs). Shape: a dependent chain of one dot4 per step per lane, `acc = dot4(x, y, acc); x = x * 0x9E3779B1 + acc; y = rotl(y, 7) ^ (acc + s)`, 1,048,576 lanes x 4,096 steps, work-group 256, best of 3 with a fresh seed per repetition, device time (Metal: command buffer GPU start to end; OpenCL: event profiling). Beside it the ALU chain of the 9070 XT entry (`x = x * K + rotl(y, 7); y = (y ^ x) + s`, 5 ops per step counted). Every kernel is checked bit for bit against a CPU reference on lanes 0 and 1,048,575 in every repetition ("ok"). Design context: `docs/analysis/int8-matrix-family.md`. + +| API, kernel | What one step is | best ms | G steps/s | ns per dependent step | ok | +|---|---|---|---|---|---| +| Metal, `probe_alu` | mul, add, rotate, xor, add | 4.882 | 879.8 (about 4.4 T int ops/s at 5 per step, approximate) | 1,192 | yes | +| Metal, `probe_dot4s` | signed dot4 emulated: `int4(as_type(a))` x same for b, 4 products summed into a wrapping int, plus the 3-op chain | 22.820 | 188.2 G dot4/s | 5,571 | yes | +| Metal, `probe_dot4u` | unsigned dot4 emulated: `uint4(as_type(a))`, same chain | 7.834 | 548.2 G dot4/s | 1,913 | yes | +| Apple OpenCL 1.2, `alu` | as Metal | 4.928 | 871.5 | 1,203 | yes | +| Apple OpenCL 1.2, `dot4e` | signed dot4 emulated with `convert_int4(as_char4(a))` | 22.797 | 188.4 G dot4/s | 5,566 | yes | +| Apple OpenCL 1.2, `dot4_khr` | `acc + dot(as_char4(x), as_char4(y))` under `#pragma OPENCL EXTENSION cl_khr_integer_dot_product : enable` | 5.076 | 846.2 | 1,239 | NO: mismatched the CPU reference on every lane checked in all 3 repetitions | + +Reading: on this GPU a signed-byte dot4 costs 4.7 ALU-chain steps and an unsigned-byte one 1.6; Metal has no dp4a and no integer simdgroup matrix (MSL 4.1 sections 2.4 and 6.9), so these are the honest Apple costs of a per-lane dot4 family, and an unsigned definition is 3x cheaper for Apple at no cost to NVIDIA or AMD (both carry the unsigned form, PTX `dp4a.u32.u32`, AMD `v_dot4_u32_u8`). Apple's OpenCL does not list `cl_khr_integer_dot_product`; its `dot` on `char4` compiled anyway and returned something other than the integer dot (the mismatch), which is why a family's conformance vectors must gate every vendor path on the feature macro, not on "it compiled". Not run here: NVIDIA and AMD. The PC job is prepared and not published (coordinator's rule): `relay/playbooks/dot4-probe.ps1` with `dot4-probe-cl.exe` (proto-opencl/dot4-probe.c cross-compiled with mingw as `x86_64-w64-mingw32-gcc -std=c99 -O2 -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 -I proto-cuda/nvrtc/redist/include`, sha256 `5adaeb1aceb03dc41135baabe0b53f1ed5fac891a5b3c3849645b03efe4416f4`, 161,863 bytes); it runs the scalar, KHR, AMD `__builtin_amdgcn_sudot4` and NVIDIA inline-PTX `dp4a` variants on every OpenCL GPU of the machine with the mining cards switched off through `/api/cards` and restored after. The CUDA form (`proto-cuda/dot4-probe.cu`, `__dp4a`) needs nvcc on the PC and is the cross-check. + +**PC 1, 5 October 2026 20:29 UTC, the same probe on the RTX 5090 and the RX 9070 XT** (machine ae432dc7, Windows 11; fetch job `fetch-dot4-20261005` placed `dot4-probe-cl.exe` sha256 `5adaeb1a…6416f4`, run job `run-dot4-20261005` ran `relay/playbooks/dot4-probe.ps1`: the app's `nvidia:0` and `amd:1:gfx1201` cards switched off through `POST api/cards`, the probe run on every OpenCL device, the cards restored with their settings (identities 8 and 2, power cap 80% and none); `node tools/jobs.mjs run-dot4-20261005`; 101 s wall, every kernel under 10 ms; device event time, best of 3, same lanes and steps as the Mac rows): + +| Device, platform | alu, G steps/s (ms) | dot4e signed emulation, G dot4/s (ms) | dot4 instruction, G dot4/s (ms) | `cl_khr_integer_dot_product` | ok | +|---|---|---|---|---|---| +| RTX 5090, NVIDIA OpenCL 3.0 CUDA, driver 617.14 | 8,753.5 (0.491) | 1,239.1 (3.466), 7.1x the ALU step | 7,453.6 (0.576) via inline PTX `dp4a.s32.s32`, 1.17x the ALU step | not listed; the `dot(char4,char4)` kernel does not build | yes | +| RX 9070 XT (gfx1201), AMD-APP 3683.0 (PAL,LC), OpenCL 2.0 | 701.4 (6.124) | 480.8 (8.932), 1.46x | 664.3 (6.465) via `__builtin_amdgcn_sudot4`, 1.06x | not listed; same | yes | +| RX 9070 XT, the older 3652.0 platform entry (dup) | 696.2 (6.169) | 501.7 (8.561) | 683.6 (6.283) | not listed | yes | +| gfx1036 (integrated RDNA 2, 2 CUs), 3683.0 | 40.6 (105.9) | 15.8 (272.3), 2.6x | `sudot4` does not build: "needs target feature dot8-insts" | not listed | alu and dot4e yes | + +Reading: one `dp4a` on the 5090 costs about one ALU-chain step (7.45 T dot4/s, 0.85 of the chain's 8.75 T steps/s); one `v_dot4_i32_iu8` on the 9070 XT the same (0.66 T, 0.95 of its chain). Emulating the signed dot4 costs 6.0x the instruction on NVIDIA (the OpenCL compiler does not fold the four sign-extended products into `dp4a`) and 1.38x on AMD. Vendor ratios: the 5090 is 12.5x the 9070 XT on the ALU chain and 11.2x on hardware dot4; against the M5 Max's best (unsigned emulation, 0.55 T) it is 10x on the chain and 13.6x on dot4. The hash itself is bound by DRAM reads, so these per-op numbers bound a family's cost and are not hash rates (`docs/analysis/int8-matrix-family.md` section 4). Adrenalin's OpenCL C accepts the clang builtin and emits the instruction on RDNA 4 (the third-party RDNA 3 report of the same route is now confirmed on this card); no PC platform lists the Khronos integer-dot extension. The 5090 SM clock read 2,505 MHz before and after (nvidia-smi; 2,850 MHz while mining in the telemetry entry), so the card was idle for the probe. + + +## 5 October 2026, layer 3 scratch soundness (Counter ASIC 2.0 step 4; branch ca2-soundness on readwidth b970dda; cryptographer) + +Machine: Apple M5 Max, 64 GiB, Darwin 25.6.0, other agents' builds and the readwidth measurements running beside (the Mac measure lock was free during the GPU runs; nothing here is a hash-rate figure). Write-up `docs/analysis/scratch-soundness.md`; tests `igneum-pow/tests/scratch.rs`; harness `proto-metal/packbench` built from this branch (`--batch-base` added) into the session scratchpad with `swiftc -O -target arm64-apple-macos11 -framework Metal`. + +CPU, `with-lock.sh build nice -n 19 ~/.cargo/bin/cargo test -j4 --test scratch -- --nocapture` (3.6 s): 7 of 7 pass. Stats, 6 classes x 3 seeds x 2^11 units, every read-modify-write traced (3.1 to 12.6 million per class): written-word bias within 6 sigma (worst 3.63); re-hit rate measured against the uniform birthday rate 12.58 vs 10.91 percent (scr2k32), 21.47 vs 20.83 (scr4k32), 36.99 vs 36.50 (scr8k32), 3.84 vs 2.88 (scr2k128), 6.25 vs 5.82 (scr4k128), 11.89 vs 11.37 (scr8k128); slot histogram non-uniform (hottest slot 1.39x to 5.10x the mean: the slot is a register's low bits); deepest chain 5 to 9. Edge: 7 hand-built programs x 2 geometries x 4 bases against an independent hand model, 56 of 56, and 56 of 56 mismatches with the hand model's rewrite words swapped. Static scratch check: 42 of 42 emitted kernels of the 7 scr packs (regenerated byte for byte from program.json first), 6 deliberate breaks caught. Fuzz: 200 generated scratch programs, contract and acceptance on every instruction, 800 units; `IGNEUM_SCRATCH_PACKS_OUT` wrote 214 packs (57 s, three memory-hard caches). The crate's other tests: 33 of 34 lib tests pass; `verify::tests::fold_and_wide_fetch` fails on the readwidth tip itself (`verify.rs:508`, `k as u32 * 0x9E37_79B1` overflows under the test profile; not touched here). + +Metal, `with-lock.sh run