From 0a108a3449fed053494683edaf0b9d8340a14afc Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 19:47:49 +0000 Subject: [PATCH 01/13] V6-07: the floor's limits from FREE memory, the small-card threshold 2^26, two recursion budgets, a pinned memory profile per workload in the host Master review R1 residual V6-07 (the floor patch picked limits from total VRAM, defaulted small cards to 2^27 where the passing rows used 2^26, and returned one recursion constant in both branches). - proving/prover-floor/sp1-gpu-6.8.1-floor.patch (and the fleet's two copies, box-setup.sh re-pinned): gpu_memory_gb() reads the device's FREE memory once per process (OnceLock; SP1_GPU_MEMORY_BUDGET_GB when the host leases it), never the total; the small tier's element threshold is 2^26 (the alone-comp-26-v1 rows of 6 October on the 3060, 3080, 4060, 4060 Ti, 4070, 5070 and the 8 October proof_alone rows on the 3060 at 7,525 MiB and the 4060 at 7,532 MiB); recursion_trace_allocation_for_budget returns upstream's 2^27 on the 24 GB tier and RECURSION_TRACE_ALLOCATION_SMALL = 2^26 + 2^25 under it (a recursion key or shard uses 90,177,536 elements); floor_tests pin the small tier and that the two branches differ; the FLOOR opts line carries free_mib and total_mib. - host/src/memory_profile.rs: the pinned table (full, 24gb, 16gb, small by free MiB; floors per workload shard, aggregate, chain) with tests pinning every value; the row is chosen from the engine's lease (IGNEUM_PROVE_MEM_BUDGET_MB, else IGNEUM_PROVE_MEM_FREE_MB, with IGNEUM_PROVE_DEVICE, IGNEUM_PROVE_WORKLOAD, IGNEUM_PROVE_DEADLINE_S: the app lane's device coordinator interface) or, with no engine, from nvidia-smi memory.free on the device; applied to the floor server by environment before the SP1 client spawns it; a hand override is kept and named; under the floor the host refuses with one line and exit 78 before any setup. - tools/fleet/box-prover.py: the default path names its workload (IGNEUM_PROVE_WORKLOAD=chain, IGNEUM_PROVE_DEVICE) and closes a refused segment as cancelled/memory. Co-Authored-By: Claude Fable 5.1 --- proving/igneum-prove/host/src/main.rs | 10 + .../igneum-prove/host/src/memory_profile.rs | 368 ++++++++++++++++++ .../prover-floor/sp1-gpu-6.8.1-floor.patch | 87 ++++- tools/fleet/box-prover.py | 12 +- tools/fleet/box-setup.sh | 2 +- tools/fleet/floor-v5.patch | 87 ++++- tools/fleet/floor.patch | 87 ++++- 7 files changed, 593 insertions(+), 60 deletions(-) create mode 100644 proving/igneum-prove/host/src/memory_profile.rs diff --git a/proving/igneum-prove/host/src/main.rs b/proving/igneum-prove/host/src/main.rs index 98e45f3fe..ae2cec3ce 100644 --- a/proving/igneum-prove/host/src/main.rs +++ b/proving/igneum-prove/host/src/main.rs @@ -20,6 +20,7 @@ //! ids with no setup. `--mode verify` uses SP1's light verifier and the pinned verifying key: no prover client, //! no key generation (the 114 s to 138 s the Mac's node spent per proof on 5 October). +mod memory_profile; mod pinned; mod proof_system; @@ -69,6 +70,15 @@ fn run() -> Result<()> { let args: Vec = std::env::args().collect(); let arg = |name: &str| args.iter().position(|a| a == name).and_then(|i| args.get(i + 1)).cloned(); let mode = arg("--mode").unwrap_or_else(|| "all".into()); + // V6-07 (8 October 2026): on the GPU prover the memory profile for this process's workload is chosen here, once, + // from the card's FREE memory (the engine's lease, else nvidia-smi) and handed to the floor server by environment + // before the SP1 client spawns it; a budget under the workload's floor is refused with exit 78 before any setup. + if std::env::var("SP1_PROVER").map(|p| p == "cuda").unwrap_or(false) { + if let Err(refusal) = memory_profile::apply(&mode) { + println!("RESULT memory_profile refused: {refusal}"); + std::process::exit(memory_profile::EXIT_REFUSED); + } + } let pinned = pinned::Pinned::load()?; if mode == "id" { println!("RESULT id: {}", pinned.describe()); diff --git a/proving/igneum-prove/host/src/memory_profile.rs b/proving/igneum-prove/host/src/memory_profile.rs new file mode 100644 index 000000000..f6f1c82eb --- /dev/null +++ b/proving/igneum-prove/host/src/memory_profile.rs @@ -0,0 +1,368 @@ +//! The pinned memory profile per workload (master review R1 residual V6-07, 8 October 2026): the GPU server's two +//! device buffers (the core element threshold and the recursion trace allocation) are chosen HERE, once, from the +//! memory that is FREE on the card when the host starts, never from the card's total. The table below is the +//! profile; `choose` picks a row for the workload this process is admitted for; `apply` hands the row to the server +//! through the floor patch's environment (`SP1_GPU_ELEMENT_THRESHOLD`, `SP1_GPU_RECURSION_TRACE_ALLOCATION`, +//! `SP1_GPU_MEMORY_BUDGET_GB`) before the SP1 client spawns it. An explicit `SP1_GPU_ELEMENT_THRESHOLD` or +//! `SP1_GPU_RECURSION_TRACE_ALLOCATION` already in the environment is a hand override and wins, named in the line. +//! +//! Where the free memory comes from, in order (the app lane's device coordinator interface, the shipper's answer of +//! 20:4x UK 8 October 2026; the engine owns admission and the host never re-reads a card the engine leased): +//! 1. `IGNEUM_PROVE_MEM_BUDGET_MB`: the lease's grant, the hard ceiling for this process (the engine's coordinator, +//! app/igneum-app `src/device.rs`, passes it with `IGNEUM_PROVE_DEVICE`, `IGNEUM_PROVE_WORKLOAD`, +//! `IGNEUM_PROVE_MEM_FREE_MB` and `IGNEUM_PROVE_DEADLINE_S` on every spawn); +//! 2. `IGNEUM_PROVE_MEM_FREE_MB`: the engine's own read at admission, when no grant is given; +//! 3. `nvidia-smi --query-gpu=memory.free` on the device (`IGNEUM_PROVE_DEVICE`, else `IGNEUM_CUDA_DEVICE`, else the +//! first of `CUDA_VISIBLE_DEVICES`, else 0): the fleet's `tools/fleet/box-prover.py` path and a hand run, where no +//! engine admits the job. +//! When none answers (no nvidia-smi on the path), no row is applied and the server's own free-memory rule decides. +//! +//! A budget under the workload's floor is refused before the server starts: one line, exit code 78 (the engine +//! records the refusal on the lease and the window reads "proving needs N GB free"). +//! +//! The rows cited for the small tier's 2^26: `docs/analysis/prover-tiers-real-cards.md` (6 October 2026, +//! `alone-comp-26-v1`: RTX 3060 7.4 GB 14.4 s, RTX 3080 8.0 GB 7.1 s, RTX 4060 7.4 GB 18.4 s, RTX 4060 Ti 8 GB 7.6 GB +//! 9.6 s, RTX 4060 Ti 16 GB 7.8 GB 11.6 s, RTX 4070 7.6 GB 12.1 s, RTX 5070 7.6 GB 4.8 s, all verified) and +//! `docs/analysis/class-v6/coexist-rows.md` (8 October 2026, `proof_alone` at `SP1_GPU_ELEMENT_THRESHOLD=67108864`: +//! RTX 3060 peak 7,525 MiB 13.2 s verified, RTX 4060 peak 7,532 MiB 8.2 s verified). No small-card row ever passed +//! at the patch's earlier default of 2^27 on this host; the 10 GB tier's 2^27 rows (8,642 MiB, 7 October 2026) ran +//! alone on the segment host and are not this path. + +use std::fmt; + +/// Upstream's core element threshold (sp1-core-executor 6.8.1 `opts.rs` 12): 2^28 + 2^27. +pub const ELEMENT_THRESHOLD_FULL: u64 = (1 << 28) + (1 << 27); +/// Upstream's 24 GB tier (sp1-gpu `builder.rs` 44): the full threshold less 2^26 + 2^25 + 2^24. +pub const ELEMENT_THRESHOLD_24GB: u64 = ELEMENT_THRESHOLD_FULL - (1 << 26) - (1 << 25) - (1 << 24); +/// The floor patch's 16 GB tier: 2^27 + 2^26 (unmeasured on this fixture; the patch's own figure). +pub const ELEMENT_THRESHOLD_16GB: u64 = (1 << 27) + (1 << 26); +/// The small-card threshold: 2^26, the value every passing small-card row used (module note). +pub const ELEMENT_THRESHOLD_SMALL: u64 = 1 << 26; +/// Upstream's recursion trace allocation (sp1-gpu `builder.rs` 15): 2^27 elements, 0.75 GiB each. +pub const RECURSION_TRACE_ALLOCATION: usize = 1 << 27; +/// The small-card recursion trace allocation: the recursion keys and shards use 90,177,536 elements each (35.6 M +/// preprocessed, 54.5 M main; `docs/analysis/prover-floor.md`, sweep 1), so 2^26 + 2^25 = 100,663,296 holds them +/// with the stacking slack the patch's `floor_capacity` adds. The two constants differ (`recursion_branches_differ`). +pub const RECURSION_TRACE_ALLOCATION_SMALL: usize = (1 << 26) + (1 << 25); +/// What a recursion key or shard was measured to use (elements); the small allocation must clear it with slack. +pub const RECURSION_TRACE_USED: usize = 90_177_536; + +/// The one job this process is admitted for. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Workload { + /// One shard's compressed (or core) proof. + Shard, + /// The aggregator guest over shard proofs (and the previous segment's proof). + Aggregate, + /// A whole segment in one process: every shard, then the aggregation. + Chain, +} + +impl Workload { + pub fn name(self) -> &'static str { + match self { + Workload::Shard => "shard", + Workload::Aggregate => "aggregate", + Workload::Chain => "chain", + } + } + /// `IGNEUM_PROVE_WORKLOAD` when the engine names it, else the host's own mode. + pub fn from_env_or_mode(mode: &str) -> Option { + if let Ok(w) = std::env::var("IGNEUM_PROVE_WORKLOAD") { + return match w.as_str() { + "shard" => Some(Workload::Shard), + "aggregate" => Some(Workload::Aggregate), + "chain" => Some(Workload::Chain), + _ => None, + }; + } + Workload::from_mode(mode) + } + /// The workload a host mode runs on the GPU (modes that never prove return None). + pub fn from_mode(mode: &str) -> Option { + match mode { + "shard" | "compressed" | "core" => Some(Workload::Shard), + "aggregate" => Some(Workload::Aggregate), + "chain" | "block" | "all" => Some(Workload::Chain), + _ => None, + } + } + /// The least free memory (MiB) a workload runs in. Shard: the small tier's measured peak (7,525 to 7,532 MiB + /// on the 3060 and 4060 at 2^26) plus headroom for the driver's own context. Aggregate and chain: the same + /// floor until the 8 October 2026 rows on the default path land their aggregation peak (the analysis document + /// `docs/analysis/floor-memory-profile-2026-10-08.md` carries the measured figure beside this constant). + pub fn floor_mib(self) -> u64 { + match self { + Workload::Shard => 7_700, + Workload::Aggregate => 7_700, + Workload::Chain => 7_700, + } + } +} + +/// One tier of the profile table: the least free memory it needs and the two buffers it sets. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Tier { + pub name: &'static str, + pub min_free_mib: u64, + pub element_threshold: u64, + pub recursion_trace_allocation: usize, +} + +/// The pinned table, largest first. The free-memory lines are the card tiers as the floor patch reads them (free +/// GiB, ceiling, +4 as upstream computed it from the total): 26 GiB free reads over 30 (a 32 GB card alone), 20 GiB +/// reads 24 (a 24 GB card alone), 14 GiB reads 18 (a 16 GB card alone), under that the small tier (12 GB and 8 GB +/// cards alone; any card beside a miner's resident set). +pub const TIERS: [Tier; 4] = [ + Tier { name: "full", min_free_mib: 26 * 1024, element_threshold: ELEMENT_THRESHOLD_FULL, recursion_trace_allocation: RECURSION_TRACE_ALLOCATION }, + Tier { name: "24gb", min_free_mib: 20 * 1024, element_threshold: ELEMENT_THRESHOLD_24GB, recursion_trace_allocation: RECURSION_TRACE_ALLOCATION }, + Tier { name: "16gb", min_free_mib: 14 * 1024, element_threshold: ELEMENT_THRESHOLD_16GB, recursion_trace_allocation: RECURSION_TRACE_ALLOCATION_SMALL }, + Tier { name: "small", min_free_mib: 0, element_threshold: ELEMENT_THRESHOLD_SMALL, recursion_trace_allocation: RECURSION_TRACE_ALLOCATION_SMALL }, +]; + +/// The row chosen for a process. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Profile { + pub workload: Workload, + pub tier: Tier, + pub free_mib: u64, +} + +/// Why a process is refused before the server starts. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Refusal { + pub workload: Workload, + pub free_mib: u64, + pub floor_mib: u64, +} + +impl fmt::Display for Refusal { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + f, + "proving needs {} GB free on the card for the {} workload ({} MiB floor); {} MiB free", + (self.floor_mib + 1023) / 1024, + self.workload.name(), + self.floor_mib, + self.free_mib + ) + } +} + +/// The exit code of a refusal (the engine reads it on the lease). +pub const EXIT_REFUSED: i32 = 78; + +/// The tier for a free-memory reading (pure; the table's first row whose line the reading clears). +pub fn tier_for(free_mib: u64) -> Tier { + *TIERS.iter().find(|t| free_mib >= t.min_free_mib).expect("the small tier has no floor") +} + +/// The row for a workload at a free-memory reading, or the refusal. +pub fn choose(workload: Workload, free_mib: u64) -> Result { + let floor_mib = workload.floor_mib(); + if free_mib < floor_mib { + return Err(Refusal { workload, free_mib, floor_mib }); + } + Ok(Profile { workload, tier: tier_for(free_mib), free_mib }) +} + +/// Where a free-memory reading came from, for the line. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Source { + /// `IGNEUM_PROVE_MEM_BUDGET_MB`, the engine's lease. + Lease, + /// `IGNEUM_PROVE_MEM_FREE_MB`, the engine's read without a grant. + EngineRead, + /// `nvidia-smi --query-gpu=memory.free` on the named device. + NvidiaSmi(String), +} + +impl fmt::Display for Source { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Source::Lease => write!(f, "lease IGNEUM_PROVE_MEM_BUDGET_MB"), + Source::EngineRead => write!(f, "engine IGNEUM_PROVE_MEM_FREE_MB"), + Source::NvidiaSmi(d) => write!(f, "nvidia-smi device {d}"), + } + } +} + +fn env_u64(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.trim().parse::().ok()) +} + +/// The device ordinal the host runs on, as the module note orders it. +pub fn device_ordinal() -> String { + if let Ok(d) = std::env::var("IGNEUM_PROVE_DEVICE") { + return d; + } + if let Ok(d) = std::env::var("IGNEUM_CUDA_DEVICE") { + return d; + } + if let Ok(v) = std::env::var("CUDA_VISIBLE_DEVICES") { + if let Some(first) = v.split(',').next() { + if !first.trim().is_empty() { + return first.trim().to_string(); + } + } + } + "0".into() +} + +/// The free memory on the card at this call, from the sources in the module note's order. +pub fn read_free_mib() -> Option<(u64, Source)> { + if let Some(b) = env_u64("IGNEUM_PROVE_MEM_BUDGET_MB") { + return Some((b, Source::Lease)); + } + if let Some(b) = env_u64("IGNEUM_PROVE_MEM_FREE_MB") { + return Some((b, Source::EngineRead)); + } + let dev = device_ordinal(); + let out = std::process::Command::new("nvidia-smi") + .args(["--query-gpu=memory.free", "--format=csv,noheader,nounits", "-i", &dev]) + .output() + .ok()?; + if !out.status.success() { + return None; + } + let text = String::from_utf8_lossy(&out.stdout); + let first = text.lines().next()?.trim(); + first.parse::().ok().map(|v| (v, Source::NvidiaSmi(dev))) +} + +/// The environment the row sets for the floor server (what `apply` writes), as (name, value) pairs; an explicit +/// hand override already present keeps its value and is reported. +pub fn env_for(p: &Profile) -> Vec<(&'static str, String)> { + vec![ + ("SP1_GPU_ELEMENT_THRESHOLD", p.tier.element_threshold.to_string()), + ("SP1_GPU_RECURSION_TRACE_ALLOCATION", p.tier.recursion_trace_allocation.to_string()), + ("SP1_GPU_MEMORY_BUDGET_GB", format!("{:.1}", p.free_mib as f64 / 1024.0)), + ] +} + +/// Chooses and applies the row for this process: prints one `RESULT memory_profile` line and returns Ok(Some) with +/// the profile, Ok(None) when no reading was possible (the server's own rule decides) or the workload never proves, +/// and Err with the refusal line (the caller exits `EXIT_REFUSED`). +pub fn apply(mode: &str) -> Result, Refusal> { + let Some(workload) = Workload::from_env_or_mode(mode) else { return Ok(None) }; + let Some((free_mib, source)) = read_free_mib() else { + println!("RESULT memory_profile: no free-memory reading (no lease, no nvidia-smi on device {}); the server's own free-memory rule decides for workload {}", device_ordinal(), workload.name()); + return Ok(None); + }; + let profile = choose(workload, free_mib)?; + let mut set = Vec::new(); + let mut kept = Vec::new(); + for (k, v) in env_for(&profile) { + match std::env::var(k) { + Ok(have) if k != "SP1_GPU_MEMORY_BUDGET_GB" => kept.push(format!("{k}={have} (hand override kept, the row said {v})")), + _ => { + std::env::set_var(k, &v); + set.push(format!("{k}={v}")); + } + } + } + let deadline = std::env::var("IGNEUM_PROVE_DEADLINE_S").ok().map(|d| format!(", deadline {d} s")).unwrap_or_default(); + println!( + "RESULT memory_profile: workload {} tier {} from {} MiB free ({}){}: element_threshold {} recursion_trace_allocation {}; set {}{}", + workload.name(), + profile.tier.name, + free_mib, + source, + deadline, + profile.tier.element_threshold, + profile.tier.recursion_trace_allocation, + set.join(" "), + if kept.is_empty() { String::new() } else { format!("; {}", kept.join("; ")) } + ); + Ok(Some(profile)) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The table's values are pinned: a change here is a change of the profile and needs its rows. + #[test] + fn table_is_pinned() { + assert_eq!(ELEMENT_THRESHOLD_FULL, 402_653_184); + assert_eq!(ELEMENT_THRESHOLD_24GB, 285_212_672); + assert_eq!(ELEMENT_THRESHOLD_16GB, 201_326_592); + assert_eq!(ELEMENT_THRESHOLD_SMALL, 67_108_864); + assert_eq!(RECURSION_TRACE_ALLOCATION, 134_217_728); + assert_eq!(RECURSION_TRACE_ALLOCATION_SMALL, 100_663_296); + let names: Vec<&str> = TIERS.iter().map(|t| t.name).collect(); + assert_eq!(names, ["full", "24gb", "16gb", "small"]); + assert_eq!(TIERS[0].min_free_mib, 26_624); + assert_eq!(TIERS[1].min_free_mib, 20_480); + assert_eq!(TIERS[2].min_free_mib, 14_336); + assert_eq!(TIERS[3].min_free_mib, 0); + assert_eq!(TIERS[3].element_threshold, ELEMENT_THRESHOLD_SMALL); + assert_eq!(TIERS[3].recursion_trace_allocation, RECURSION_TRACE_ALLOCATION_SMALL); + assert_eq!(TIERS[0].recursion_trace_allocation, RECURSION_TRACE_ALLOCATION); + assert_eq!(Workload::Shard.floor_mib(), 7_700); + assert_eq!(Workload::Aggregate.floor_mib(), 7_700); + assert_eq!(Workload::Chain.floor_mib(), 7_700); + assert_eq!(EXIT_REFUSED, 78); + } + + /// The two recursion branches are two values (the review found one constant in both), and the small one + /// clears what a recursion key or shard was measured to use, with a stacking height of slack. + #[test] + fn recursion_branches_differ() { + assert_ne!(RECURSION_TRACE_ALLOCATION, RECURSION_TRACE_ALLOCATION_SMALL); + assert!(RECURSION_TRACE_ALLOCATION_SMALL < RECURSION_TRACE_ALLOCATION); + assert!(RECURSION_TRACE_ALLOCATION_SMALL >= RECURSION_TRACE_USED + (1 << 22)); + assert_ne!(tier_for(24 * 1024).recursion_trace_allocation, tier_for(12 * 1024).recursion_trace_allocation); + } + + /// The small tier's threshold is 2^26, the passing rows' value; the card tiers alone read their own rows. + #[test] + fn tiers_from_free_memory() { + assert_eq!(tier_for(8_186).name, "small"); + assert_eq!(tier_for(8_186).element_threshold, 1 << 26); + assert_eq!(tier_for(12_100).name, "small"); + assert_eq!(tier_for(12_100).element_threshold, 1 << 26); + assert_eq!(tier_for(16_100).name, "16gb"); + assert_eq!(tier_for(24_200).name, "24gb"); + assert_eq!(tier_for(32_300).name, "full"); + // a 12 GB card beside the 5.5 GiB miner (6,129 MiB resident) reads the small tier by what is FREE + assert_eq!(tier_for(12_288 - 6_129).name, "small"); + } + + /// The refusal: under the floor, before the server starts; at the floor, the small row. + #[test] + fn refuses_under_the_floor() { + let r = choose(Workload::Shard, 12_288 - 6_129).unwrap_err(); + assert_eq!(r.floor_mib, 7_700); + assert_eq!(r.free_mib, 6_159); + assert_eq!(r.to_string(), "proving needs 8 GB free on the card for the shard workload (7,700 MiB floor); 6,159 MiB free".replace(",", "")); + let p = choose(Workload::Shard, 7_700).unwrap(); + assert_eq!(p.tier.name, "small"); + let p = choose(Workload::Chain, 8_186).unwrap(); + assert_eq!(p.tier.element_threshold, ELEMENT_THRESHOLD_SMALL); + assert_eq!(p.tier.recursion_trace_allocation, RECURSION_TRACE_ALLOCATION_SMALL); + } + + /// The host's modes map to the three workloads; modes that never prove map to none. + #[test] + fn workload_from_mode() { + assert_eq!(Workload::from_mode("compressed"), Some(Workload::Shard)); + assert_eq!(Workload::from_mode("core"), Some(Workload::Shard)); + assert_eq!(Workload::from_mode("aggregate"), Some(Workload::Aggregate)); + assert_eq!(Workload::from_mode("chain"), Some(Workload::Chain)); + assert_eq!(Workload::from_mode("all"), Some(Workload::Chain)); + assert_eq!(Workload::from_mode("verify"), None); + assert_eq!(Workload::from_mode("id"), None); + assert_eq!(Workload::from_mode("native"), None); + } + + /// The environment a row hands the floor server. + #[test] + fn env_for_the_server() { + let p = choose(Workload::Shard, 8_186).unwrap(); + let env = env_for(&p); + assert_eq!(env[0], ("SP1_GPU_ELEMENT_THRESHOLD", "67108864".to_string())); + assert_eq!(env[1], ("SP1_GPU_RECURSION_TRACE_ALLOCATION", "100663296".to_string())); + assert_eq!(env[2], ("SP1_GPU_MEMORY_BUDGET_GB", "8.0".to_string())); + } +} diff --git a/proving/prover-floor/sp1-gpu-6.8.1-floor.patch b/proving/prover-floor/sp1-gpu-6.8.1-floor.patch index c66939337..dcc6fbad4 100644 --- a/proving/prover-floor/sp1-gpu-6.8.1-floor.patch +++ b/proving/prover-floor/sp1-gpu-6.8.1-floor.patch @@ -104,15 +104,16 @@ diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/pr index 5dccd9d..574d4fa 100644 --- a/sp1-gpu/crates/prover_components/src/builder.rs +++ b/sp1-gpu/crates/prover_components/src/builder.rs -@@ -23,28 +23,75 @@ use crate::{ +@@ -23,28 +23,124 @@ use crate::{ SP1CudaProverComponents, }; -+/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger -+/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a -+/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. -+/// The proof format, the verifier and the program ids do not change: the element threshold only decides where -+/// the executor splits shards, as upstream's own 24 GB tier already does. ++/// Igneum prover-floor patch (5 October 2026; the memory rules of V6-07, 8 October 2026). Upstream sizes every ++/// device buffer for a 24 GB card or larger and panics below that, whatever the shard. Here the card's FREE memory ++/// at start (or `SP1_GPU_MEMORY_BUDGET_GB`, the host's lease) picks a tier, read once for the process, and ++/// `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. The proof ++/// format, the verifier and the program ids do not change: the element threshold only decides where the executor ++/// splits shards, as upstream's own 24 GB tier already does. +fn env_usize(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} @@ -121,8 +122,17 @@ index 5dccd9d..574d4fa 100644 + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} + -+/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it -+/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16). ++/// The recursion trace allocation under the 24 GB tier (V6-07): a recursion key or shard uses 90,177,536 elements ++/// (35.6 M preprocessed and 54.5 M main; the floor sweep of 5 October 2026), so 2^26 + 2^25 = 100,663,296 holds it ++/// with the stacking slack `floor_capacity` adds; the pinned host copies (four per prover) shrink with it. Upstream's ++/// 2^27 stays on the 24 GB tier and above. Two distinct values: `floor_tests::recursion_branches_differ`. ++pub const RECURSION_TRACE_ALLOCATION_SMALL: usize = (1 << 26) + (1 << 25); ++ ++/// The core element threshold for a memory budget in GB (the budget is the FREE memory, +4, as upstream computed ++/// its tiers from the total: a 32 GB card alone reads 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16, an ++/// 8 GB card 12). Under the 18 tier the threshold is 2^26, the value every passing small-card row used (the ++/// alone-comp-26-v1 rows of 6 October 2026 on the 3060, 3080, 4060, 4060 Ti, 4070 and 5070; the 8 October 2026 ++/// proof_alone rows on the 3060 at 7,525 MiB and the 4060 at 7,532 MiB); no small-card row passed at 2^27 here. +pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 { + if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) { + ELEMENT_THRESHOLD @@ -131,31 +141,69 @@ index 5dccd9d..574d4fa 100644 + } else if gpu_memory_gb >= 18 { + (1 << 27) + (1 << 26) + } else { -+ 1 << 27 ++ 1 << 26 + } +} + -+/// The recursion trace allocation (elements) for a memory budget. ++/// The recursion trace allocation (elements) for a memory budget: upstream's on the 24 GB tier and above, the ++/// small constant under it. +pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize { + if gpu_memory_gb >= 24 { + RECURSION_TRACE_ALLOCATION + } else { -+ RECURSION_TRACE_ALLOCATION ++ RECURSION_TRACE_ALLOCATION_SMALL + } +} + ++/// The memory budget in GB, read ONCE for the process (the core opts and the recursion prover are built at ++/// different moments, after allocations that lower the free figure; one reading keeps both on one tier): ++/// `SP1_GPU_MEMORY_BUDGET_GB` when set (the host's lease), else the device's FREE memory as the driver reports it ++/// at the first call, never the total (a card beside a miner has the miner's resident set gone), +4 as upstream ++/// computed its tiers. +pub fn gpu_memory_gb() -> usize { -+ let gb = 1024.0 * 1024.0 * 1024.0; -+ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { -+ Some(b) => (b.ceil() as usize) + 4, -+ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4, -+ } ++ static BUDGET: std::sync::OnceLock = std::sync::OnceLock::new(); ++ *BUDGET.get_or_init(|| { ++ let gb = 1024.0 * 1024.0 * 1024.0; ++ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { ++ Some(b) => (b.ceil() as usize) + 4, ++ None => { ++ let (free, _total) = cuda_memory_info().unwrap(); ++ (((free as f64) / gb).ceil() as usize) + 4 ++ } ++ } ++ }) +} + +pub fn recursion_trace_allocation() -> usize { + env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION") + .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb())) +} ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::*; ++ ++ #[test] ++ fn small_tier_is_two_to_the_26() { ++ assert_eq!(element_threshold_for_budget(12, false), 1 << 26, "an 8 GB card reads 12"); ++ assert_eq!(element_threshold_for_budget(16, false), 1 << 26, "a 12 GB card reads 16"); ++ assert_eq!(element_threshold_for_budget(17, false), 1 << 26); ++ assert_eq!(element_threshold_for_budget(20, false), (1 << 27) + (1 << 26), "a 16 GB card reads 20"); ++ assert_eq!(element_threshold_for_budget(28, false), ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)); ++ assert_eq!(element_threshold_for_budget(28, true), ELEMENT_THRESHOLD); ++ assert_eq!(element_threshold_for_budget(36, false), ELEMENT_THRESHOLD); ++ } ++ ++ #[test] ++ fn recursion_branches_differ() { ++ assert_ne!(recursion_trace_allocation_for_budget(28), recursion_trace_allocation_for_budget(16)); ++ assert_eq!(recursion_trace_allocation_for_budget(28), RECURSION_TRACE_ALLOCATION); ++ assert_eq!(recursion_trace_allocation_for_budget(16), RECURSION_TRACE_ALLOCATION_SMALL); ++ assert_eq!(RECURSION_TRACE_ALLOCATION_SMALL, 100_663_296); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL < RECURSION_TRACE_ALLOCATION); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL >= 90_177_536 + (1 << 22)); ++ } ++} + pub fn local_gpu_opts() -> SP1CoreOpts { let mut opts = SP1CoreOpts::default(); @@ -171,7 +219,7 @@ index 5dccd9d..574d4fa 100644 - if gpu_memory_gb < 24 { - panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"); - } -+ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given. ++ // The card's FREE memory plus 4, as upstream computed its tiers from the total, or the budget given. + let gpu_memory_gb = gpu_memory_gb(); - let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 { @@ -183,17 +231,18 @@ index 5dccd9d..574d4fa 100644 + None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards), }; + let height_threshold = opts.sharding_threshold.height_threshold; ++ let (free_mib, total_mib) = cuda_memory_info().map(|(f, t)| (f >> 20, t >> 20)).unwrap_or((0, 0)); - tracing::debug!("Shard threshold: {shard_threshold}"); + eprintln!( -+ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", ++ "FLOOR opts gpu_memory_gb={gpu_memory_gb} free_mib={free_mib} total_mib={total_mib} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", + recursion_trace_allocation(), + opts.full_size_shards + ); opts.sharding_threshold.element_threshold = shard_threshold; opts.global_dependencies_opt = true; -@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier( +@@ -92,7 +188,7 @@ pub async fn recursion_prover_and_verifier( ) { let recursion_verifier = SP1CudaProverComponents::compress_verifier(); ( diff --git a/tools/fleet/box-prover.py b/tools/fleet/box-prover.py index 521a1afc9..48c4ea968 100644 --- a/tools/fleet/box-prover.py +++ b/tools/fleet/box-prover.py @@ -11,7 +11,8 @@ proving-v1, 272b025) and tools/proving-v1/pc2-segments.ps1 around the four binar offered again every pass until the segment's deadline (the 272b025 behaviour). 3. one export (igneum_exportSegments 0..last), one fixture per block (igneum-prove-export), one host run (igneum-prove-host --mode chain --chain ... --save-shards [--prev]) on the patched server (HOME=/opt/igneum-floor/home, - SP1_GPU_ELEMENT_THRESHOLD from THRESHOLD), the miner paused for the run when MINER=pause (prove-alone cards). + IGNEUM_PROVE_WORKLOAD=chain and no threshold by default, the host's memory profile from the card's free memory; + SP1_GPU_ELEMENT_THRESHOLD from THRESHOLD as a hand override), the miner paused for the run when MINER=pause (prove-alone cards). 4. every shard record signed (igneum-miner sign-record) and submitted (igneum_submitProofRecord); the segment record (sign-segment-record, igneum_submitSegmentRecord) once every shard is accepted and the statement equals the node's. 5. the paid state of every submitted segment polled each pass (igneum_getSegmentRecords); a state file for the @@ -255,7 +256,10 @@ while (time.time() - t_run0) / 3600 < RUN_HOURS: # chain if MINER == "pause": miner_stop() kill_server() - env = dict(os.environ, HOME=f"{FLOOR}/home", SP1_PROVER="cuda", RUST_LOG="off") + # the default job path (V6-07, 8 October 2026): no threshold override; the host picks its memory profile for the chain + # workload from the card's FREE memory (host/src/memory_profile.rs) and hands it to the floor server; THRESHOLD, when + # set, is a hand override the host keeps and names in its RESULT memory_profile line + env = dict(os.environ, HOME=f"{FLOOR}/home", SP1_PROVER="cuda", RUST_LOG="off", IGNEUM_PROVE_WORKLOAD="chain", IGNEUM_PROVE_DEVICE=str(DEV)) if THRESHOLD: env["SP1_GPU_ELEMENT_THRESHOLD"] = THRESHOLD args = [HOST, "--mode", "chain", "--chain", ",".join(fixtures), "--prover", WALLET, "--save-shards", "--out", f"{d}/chain-results.json"] if prev_file: args += ["--prev", prev_file] @@ -269,6 +273,10 @@ while (time.time() - t_run0) / 3600 < RUN_HOURS: except Exception: peak = 0 seg["peak_mib"] = peak open(f"{d}/chain.log", "w").write((rr.stdout if rr else "") + "\n" + (rr.stderr if rr else "TIMEOUT")) + if rr and rr.returncode == 78: + # the host refused the card's free memory for the chain workload before any setup (exit 78, one line) + line = [l for l in rr.stdout.split("\n") if "memory_profile refused" in l] + say(f"RESULT seg {first} chain REFUSED {stamp()} {line[-1].strip() if line else 'memory profile refused'}"); seg["failed"] = "memory"; close(first, "cancelled", "memory"); time.sleep(60); continue if not rr or rr.returncode != 0 or not os.path.exists(f"{d}/chain-results.json"): say(f"RESULT seg {first} chain FAILED {stamp()} rc={rr.returncode if rr else 'timeout'} wall={seg['chain_s']} s: {((rr.stderr if rr else '') or '')[-200:].strip()}"); seg["failed"] = "chain"; close(first, "cancelled", "chain" if rr else "timeout"); continue res = json.load(open(f"{d}/chain-results.json")) diff --git a/tools/fleet/box-setup.sh b/tools/fleet/box-setup.sh index 634c34238..30e8e308e 100755 --- a/tools/fleet/box-setup.sh +++ b/tools/fleet/box-setup.sh @@ -22,7 +22,7 @@ fail() { echo "RESULT setup_failed $1 $(stamp)"; exit 2; } ARCHS="${ARCHS:-86,89,120}"; LABEL="${LABEL:-box}"; WALLET="${WALLET:-}" PKG_URL=https://dl.igneum.network/dl/public/igneum-hive-0.3.12.tar.gz PKG_SHA=7972af92e7cd9a032303eca4d95b533f53e0e68d1b9cae5bfe406a5b7c30a454 -PATCH_SHA=e81cb0d03b291f9fd4bf0a109d6da2d7c897795c9ffd7f797c0ddce723eee2b1 +PATCH_SHA=f53871d92e27b743b82545c03bfa6ef21cf799f43fae6138ad4136637f3ce2c2 SEED=188.245.5.161:26611 echo "RESULT start $(stamp) label=$LABEL archs=$ARCHS host=$(hostname) nproc=$(nproc) ram_gb=$(( $(awk '/MemTotal/{print $2}' /proc/meminfo) / 1048576 )) disk_avail=$(df -BG /root | awk 'NR==2{print $4}')" echo "RESULT gpu $(nvidia-smi --query-gpu=name,memory.total,driver_version,pci.bus_id,power.limit,power.min_limit,power.max_limit,clocks.max.sm --format=csv,noheader 2>&1 | tr '\n' ';')" diff --git a/tools/fleet/floor-v5.patch b/tools/fleet/floor-v5.patch index 8b28f2aad..c3a8ec532 100644 --- a/tools/fleet/floor-v5.patch +++ b/tools/fleet/floor-v5.patch @@ -199,15 +199,16 @@ diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/pr index 5dccd9d..574d4fa 100644 --- a/sp1-gpu/crates/prover_components/src/builder.rs +++ b/sp1-gpu/crates/prover_components/src/builder.rs -@@ -23,28 +23,75 @@ use crate::{ +@@ -23,28 +23,124 @@ use crate::{ SP1CudaProverComponents, }; -+/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger -+/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a -+/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. -+/// The proof format, the verifier and the program ids do not change: the element threshold only decides where -+/// the executor splits shards, as upstream's own 24 GB tier already does. ++/// Igneum prover-floor patch (5 October 2026; the memory rules of V6-07, 8 October 2026). Upstream sizes every ++/// device buffer for a 24 GB card or larger and panics below that, whatever the shard. Here the card's FREE memory ++/// at start (or `SP1_GPU_MEMORY_BUDGET_GB`, the host's lease) picks a tier, read once for the process, and ++/// `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. The proof ++/// format, the verifier and the program ids do not change: the element threshold only decides where the executor ++/// splits shards, as upstream's own 24 GB tier already does. +fn env_usize(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} @@ -216,8 +217,17 @@ index 5dccd9d..574d4fa 100644 + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} + -+/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it -+/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16). ++/// The recursion trace allocation under the 24 GB tier (V6-07): a recursion key or shard uses 90,177,536 elements ++/// (35.6 M preprocessed and 54.5 M main; the floor sweep of 5 October 2026), so 2^26 + 2^25 = 100,663,296 holds it ++/// with the stacking slack `floor_capacity` adds; the pinned host copies (four per prover) shrink with it. Upstream's ++/// 2^27 stays on the 24 GB tier and above. Two distinct values: `floor_tests::recursion_branches_differ`. ++pub const RECURSION_TRACE_ALLOCATION_SMALL: usize = (1 << 26) + (1 << 25); ++ ++/// The core element threshold for a memory budget in GB (the budget is the FREE memory, +4, as upstream computed ++/// its tiers from the total: a 32 GB card alone reads 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16, an ++/// 8 GB card 12). Under the 18 tier the threshold is 2^26, the value every passing small-card row used (the ++/// alone-comp-26-v1 rows of 6 October 2026 on the 3060, 3080, 4060, 4060 Ti, 4070 and 5070; the 8 October 2026 ++/// proof_alone rows on the 3060 at 7,525 MiB and the 4060 at 7,532 MiB); no small-card row passed at 2^27 here. +pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 { + if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) { + ELEMENT_THRESHOLD @@ -226,31 +236,69 @@ index 5dccd9d..574d4fa 100644 + } else if gpu_memory_gb >= 18 { + (1 << 27) + (1 << 26) + } else { -+ 1 << 27 ++ 1 << 26 + } +} + -+/// The recursion trace allocation (elements) for a memory budget. ++/// The recursion trace allocation (elements) for a memory budget: upstream's on the 24 GB tier and above, the ++/// small constant under it. +pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize { + if gpu_memory_gb >= 24 { + RECURSION_TRACE_ALLOCATION + } else { -+ RECURSION_TRACE_ALLOCATION ++ RECURSION_TRACE_ALLOCATION_SMALL + } +} + ++/// The memory budget in GB, read ONCE for the process (the core opts and the recursion prover are built at ++/// different moments, after allocations that lower the free figure; one reading keeps both on one tier): ++/// `SP1_GPU_MEMORY_BUDGET_GB` when set (the host's lease), else the device's FREE memory as the driver reports it ++/// at the first call, never the total (a card beside a miner has the miner's resident set gone), +4 as upstream ++/// computed its tiers. +pub fn gpu_memory_gb() -> usize { -+ let gb = 1024.0 * 1024.0 * 1024.0; -+ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { -+ Some(b) => (b.ceil() as usize) + 4, -+ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4, -+ } ++ static BUDGET: std::sync::OnceLock = std::sync::OnceLock::new(); ++ *BUDGET.get_or_init(|| { ++ let gb = 1024.0 * 1024.0 * 1024.0; ++ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { ++ Some(b) => (b.ceil() as usize) + 4, ++ None => { ++ let (free, _total) = cuda_memory_info().unwrap(); ++ (((free as f64) / gb).ceil() as usize) + 4 ++ } ++ } ++ }) +} + +pub fn recursion_trace_allocation() -> usize { + env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION") + .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb())) +} ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::*; ++ ++ #[test] ++ fn small_tier_is_two_to_the_26() { ++ assert_eq!(element_threshold_for_budget(12, false), 1 << 26, "an 8 GB card reads 12"); ++ assert_eq!(element_threshold_for_budget(16, false), 1 << 26, "a 12 GB card reads 16"); ++ assert_eq!(element_threshold_for_budget(17, false), 1 << 26); ++ assert_eq!(element_threshold_for_budget(20, false), (1 << 27) + (1 << 26), "a 16 GB card reads 20"); ++ assert_eq!(element_threshold_for_budget(28, false), ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)); ++ assert_eq!(element_threshold_for_budget(28, true), ELEMENT_THRESHOLD); ++ assert_eq!(element_threshold_for_budget(36, false), ELEMENT_THRESHOLD); ++ } ++ ++ #[test] ++ fn recursion_branches_differ() { ++ assert_ne!(recursion_trace_allocation_for_budget(28), recursion_trace_allocation_for_budget(16)); ++ assert_eq!(recursion_trace_allocation_for_budget(28), RECURSION_TRACE_ALLOCATION); ++ assert_eq!(recursion_trace_allocation_for_budget(16), RECURSION_TRACE_ALLOCATION_SMALL); ++ assert_eq!(RECURSION_TRACE_ALLOCATION_SMALL, 100_663_296); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL < RECURSION_TRACE_ALLOCATION); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL >= 90_177_536 + (1 << 22)); ++ } ++} + pub fn local_gpu_opts() -> SP1CoreOpts { let mut opts = SP1CoreOpts::default(); @@ -266,7 +314,7 @@ index 5dccd9d..574d4fa 100644 - if gpu_memory_gb < 24 { - panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"); - } -+ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given. ++ // The card's FREE memory plus 4, as upstream computed its tiers from the total, or the budget given. + let gpu_memory_gb = gpu_memory_gb(); - let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 { @@ -278,17 +326,18 @@ index 5dccd9d..574d4fa 100644 + None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards), }; + let height_threshold = opts.sharding_threshold.height_threshold; ++ let (free_mib, total_mib) = cuda_memory_info().map(|(f, t)| (f >> 20, t >> 20)).unwrap_or((0, 0)); - tracing::debug!("Shard threshold: {shard_threshold}"); + eprintln!( -+ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", ++ "FLOOR opts gpu_memory_gb={gpu_memory_gb} free_mib={free_mib} total_mib={total_mib} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", + recursion_trace_allocation(), + opts.full_size_shards + ); opts.sharding_threshold.element_threshold = shard_threshold; opts.global_dependencies_opt = true; -@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier( +@@ -92,7 +188,7 @@ pub async fn recursion_prover_and_verifier( ) { let recursion_verifier = SP1CudaProverComponents::compress_verifier(); ( diff --git a/tools/fleet/floor.patch b/tools/fleet/floor.patch index 71a8d88e4..3d01b9ee5 100644 --- a/tools/fleet/floor.patch +++ b/tools/fleet/floor.patch @@ -199,15 +199,16 @@ diff --git a/sp1-gpu/crates/prover_components/src/builder.rs b/sp1-gpu/crates/pr index 5dccd9d..574d4fa 100644 --- a/sp1-gpu/crates/prover_components/src/builder.rs +++ b/sp1-gpu/crates/prover_components/src/builder.rs -@@ -23,28 +23,75 @@ use crate::{ +@@ -23,28 +23,124 @@ use crate::{ SP1CudaProverComponents, }; -+/// Igneum prover-floor patch (5 October 2026). Upstream sizes every device buffer for a 24 GB card or larger -+/// and panics below that, whatever the shard. Here the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks a -+/// tier, and `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. -+/// The proof format, the verifier and the program ids do not change: the element threshold only decides where -+/// the executor splits shards, as upstream's own 24 GB tier already does. ++/// Igneum prover-floor patch (5 October 2026; the memory rules of V6-07, 8 October 2026). Upstream sizes every ++/// device buffer for a 24 GB card or larger and panics below that, whatever the shard. Here the card's FREE memory ++/// at start (or `SP1_GPU_MEMORY_BUDGET_GB`, the host's lease) picks a tier, read once for the process, and ++/// `SP1_GPU_ELEMENT_THRESHOLD` / `SP1_GPU_RECURSION_TRACE_ALLOCATION` set the two buffers directly. The proof ++/// format, the verifier and the program ids do not change: the element threshold only decides where the executor ++/// splits shards, as upstream's own 24 GB tier already does. +fn env_usize(name: &str) -> Option { + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} @@ -216,8 +217,17 @@ index 5dccd9d..574d4fa 100644 + std::env::var(name).ok().and_then(|s| s.parse::().ok()) +} + -+/// The core element threshold for a memory budget in GB (upstream's own figure for the budget, +4, as it -+/// computed it: a 32 GB card is 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16). ++/// The recursion trace allocation under the 24 GB tier (V6-07): a recursion key or shard uses 90,177,536 elements ++/// (35.6 M preprocessed and 54.5 M main; the floor sweep of 5 October 2026), so 2^26 + 2^25 = 100,663,296 holds it ++/// with the stacking slack `floor_capacity` adds; the pinned host copies (four per prover) shrink with it. Upstream's ++/// 2^27 stays on the 24 GB tier and above. Two distinct values: `floor_tests::recursion_branches_differ`. ++pub const RECURSION_TRACE_ALLOCATION_SMALL: usize = (1 << 26) + (1 << 25); ++ ++/// The core element threshold for a memory budget in GB (the budget is the FREE memory, +4, as upstream computed ++/// its tiers from the total: a 32 GB card alone reads 36, a 24 GB card 28, a 16 GB card 20, a 12 GB card 16, an ++/// 8 GB card 12). Under the 18 tier the threshold is 2^26, the value every passing small-card row used (the ++/// alone-comp-26-v1 rows of 6 October 2026 on the 3060, 3080, 4060, 4060 Ti, 4070 and 5070; the 8 October 2026 ++/// proof_alone rows on the 3060 at 7,525 MiB and the 4060 at 7,532 MiB); no small-card row passed at 2^27 here. +pub fn element_threshold_for_budget(gpu_memory_gb: usize, full_size_shards: bool) -> u64 { + if gpu_memory_gb > 30 || (full_size_shards && gpu_memory_gb >= 24) { + ELEMENT_THRESHOLD @@ -226,31 +236,69 @@ index 5dccd9d..574d4fa 100644 + } else if gpu_memory_gb >= 18 { + (1 << 27) + (1 << 26) + } else { -+ 1 << 27 ++ 1 << 26 + } +} + -+/// The recursion trace allocation (elements) for a memory budget. ++/// The recursion trace allocation (elements) for a memory budget: upstream's on the 24 GB tier and above, the ++/// small constant under it. +pub fn recursion_trace_allocation_for_budget(gpu_memory_gb: usize) -> usize { + if gpu_memory_gb >= 24 { + RECURSION_TRACE_ALLOCATION + } else { -+ RECURSION_TRACE_ALLOCATION ++ RECURSION_TRACE_ALLOCATION_SMALL + } +} + ++/// The memory budget in GB, read ONCE for the process (the core opts and the recursion prover are built at ++/// different moments, after allocations that lower the free figure; one reading keeps both on one tier): ++/// `SP1_GPU_MEMORY_BUDGET_GB` when set (the host's lease), else the device's FREE memory as the driver reports it ++/// at the first call, never the total (a card beside a miner has the miner's resident set gone), +4 as upstream ++/// computed its tiers. +pub fn gpu_memory_gb() -> usize { -+ let gb = 1024.0 * 1024.0 * 1024.0; -+ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { -+ Some(b) => (b.ceil() as usize) + 4, -+ None => (((cuda_memory_info().unwrap().1 as f64) / gb).ceil() as usize) + 4, -+ } ++ static BUDGET: std::sync::OnceLock = std::sync::OnceLock::new(); ++ *BUDGET.get_or_init(|| { ++ let gb = 1024.0 * 1024.0 * 1024.0; ++ match env_f64("SP1_GPU_MEMORY_BUDGET_GB") { ++ Some(b) => (b.ceil() as usize) + 4, ++ None => { ++ let (free, _total) = cuda_memory_info().unwrap(); ++ (((free as f64) / gb).ceil() as usize) + 4 ++ } ++ } ++ }) +} + +pub fn recursion_trace_allocation() -> usize { + env_usize("SP1_GPU_RECURSION_TRACE_ALLOCATION") + .unwrap_or_else(|| recursion_trace_allocation_for_budget(gpu_memory_gb())) +} ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::*; ++ ++ #[test] ++ fn small_tier_is_two_to_the_26() { ++ assert_eq!(element_threshold_for_budget(12, false), 1 << 26, "an 8 GB card reads 12"); ++ assert_eq!(element_threshold_for_budget(16, false), 1 << 26, "a 12 GB card reads 16"); ++ assert_eq!(element_threshold_for_budget(17, false), 1 << 26); ++ assert_eq!(element_threshold_for_budget(20, false), (1 << 27) + (1 << 26), "a 16 GB card reads 20"); ++ assert_eq!(element_threshold_for_budget(28, false), ELEMENT_THRESHOLD - (1 << 26) - (1 << 25) - (1 << 24)); ++ assert_eq!(element_threshold_for_budget(28, true), ELEMENT_THRESHOLD); ++ assert_eq!(element_threshold_for_budget(36, false), ELEMENT_THRESHOLD); ++ } ++ ++ #[test] ++ fn recursion_branches_differ() { ++ assert_ne!(recursion_trace_allocation_for_budget(28), recursion_trace_allocation_for_budget(16)); ++ assert_eq!(recursion_trace_allocation_for_budget(28), RECURSION_TRACE_ALLOCATION); ++ assert_eq!(recursion_trace_allocation_for_budget(16), RECURSION_TRACE_ALLOCATION_SMALL); ++ assert_eq!(RECURSION_TRACE_ALLOCATION_SMALL, 100_663_296); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL < RECURSION_TRACE_ALLOCATION); ++ assert!(RECURSION_TRACE_ALLOCATION_SMALL >= 90_177_536 + (1 << 22)); ++ } ++} + pub fn local_gpu_opts() -> SP1CoreOpts { let mut opts = SP1CoreOpts::default(); @@ -266,7 +314,7 @@ index 5dccd9d..574d4fa 100644 - if gpu_memory_gb < 24 { - panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"); - } -+ // The card's memory plus 4, as upstream computed it (a 32 GB card reads 36), or the budget given. ++ // The card's FREE memory plus 4, as upstream computed its tiers from the total, or the budget given. + let gpu_memory_gb = gpu_memory_gb(); - let shard_threshold = if !opts.full_size_shards && gpu_memory_gb <= 30 { @@ -278,17 +326,18 @@ index 5dccd9d..574d4fa 100644 + None => element_threshold_for_budget(gpu_memory_gb, opts.full_size_shards), }; + let height_threshold = opts.sharding_threshold.height_threshold; ++ let (free_mib, total_mib) = cuda_memory_info().map(|(f, t)| (f >> 20, t >> 20)).unwrap_or((0, 0)); - tracing::debug!("Shard threshold: {shard_threshold}"); + eprintln!( -+ "FLOOR opts gpu_memory_gb={gpu_memory_gb} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", ++ "FLOOR opts gpu_memory_gb={gpu_memory_gb} free_mib={free_mib} total_mib={total_mib} element_threshold={shard_threshold} height_threshold={height_threshold} recursion_trace_allocation={} full_size_shards={}", + recursion_trace_allocation(), + opts.full_size_shards + ); opts.sharding_threshold.element_threshold = shard_threshold; opts.global_dependencies_opt = true; -@@ -92,7 +139,7 @@ pub async fn recursion_prover_and_verifier( +@@ -92,7 +188,7 @@ pub async fn recursion_prover_and_verifier( ) { let recursion_verifier = SP1CudaProverComponents::compress_verifier(); ( From 0d208582a51b8099c9ba54ce34a604f424df6688 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 19:54:26 +0000 Subject: [PATCH 02/13] V6-07: the host's row from the lesser of the engine's grant and its free reading; prover-floor.md names the free-memory rule The window lane's engine half (reviewb-202 af5ade43) grants a fixed proof budget beside the miner (7,532 MiB plus 10 percent) while the card's free figure can be under it (a 12 GB card beside the 6.1 GiB miner: 6,159 MiB free), so memory_profile::lease_reading takes the lesser of IGNEUM_PROVE_MEM_BUDGET_MB and IGNEUM_PROVE_MEM_FREE_MB when both are set; test lease_takes_the_lesser_of_grant_and_free. Suite on box 3: 7 passed. Co-Authored-By: Claude Fable 5.1 --- docs/analysis/prover-floor.md | 12 ++++-- .../igneum-prove/host/src/memory_profile.rs | 40 +++++++++++++++---- 2 files changed, 41 insertions(+), 11 deletions(-) diff --git a/docs/analysis/prover-floor.md b/docs/analysis/prover-floor.md index 8c15af8c2..fe227bff1 100644 --- a/docs/analysis/prover-floor.md +++ b/docs/analysis/prover-floor.md @@ -48,10 +48,14 @@ from 20 M cycles is the threshold's padded area reached. The witness (5 to 22 KB ## What the patch does (`proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, three files) -1. `builder.rs`: the panic is gone; the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks the element threshold - from a tier table (`element_threshold_for_budget`: over 30 as read, the full 402.6 M; 24 to 30, upstream's 24 GB - figure; 18 to 24 (a 16 GB card), 2^27 + 2^26 = 201.3 M; under 18 (a 12 GB card), 2^27 = 134.2 M); - `SP1_GPU_ELEMENT_THRESHOLD` sets it directly and `SP1_GPU_RECURSION_TRACE_ALLOCATION` the recursion buffer. +1. `builder.rs`: the panic is gone; the card's FREE memory at start (read once per process; or + `SP1_GPU_MEMORY_BUDGET_GB`, the host's lease), never its total, picks the element threshold from a tier table + (`element_threshold_for_budget`: over 30 as read, the full 402.6 M; 24 to 30, upstream's 24 GB figure; 18 to 24 + (a 16 GB card alone), 2^27 + 2^26 = 201.3 M; under 18 (a 12 GB or 8 GB card, any card beside a miner), 2^26 = + 67.1 M, the value the passing small-card rows used) and the recursion trace allocation (upstream's 2^27 on the + 24 GB tier and above, 2^26 + 2^25 under it: V6-07, 8 October 2026, `docs/analysis/floor-memory-profile-2026-10-08.md`); + `SP1_GPU_ELEMENT_THRESHOLD` sets it directly and `SP1_GPU_RECURSION_TRACE_ALLOCATION` the recursion buffer. The + host's own profile table (`host/src/memory_profile.rs`) sets both by environment before the server starts. The chosen numbers are printed as a `FLOOR opts` line. Every other option is as upstream. 2. `jagged_tracegen/src/lib.rs`: with `SP1_GPU_FLOOR_LOG` set, every trace allocation prints its capacity and, after the shard's traces are in, the elements actually used and the device memory in use. diff --git a/proving/igneum-prove/host/src/memory_profile.rs b/proving/igneum-prove/host/src/memory_profile.rs index f6f1c82eb..61a6fb52a 100644 --- a/proving/igneum-prove/host/src/memory_profile.rs +++ b/proving/igneum-prove/host/src/memory_profile.rs @@ -10,8 +10,9 @@ //! 20:4x UK 8 October 2026; the engine owns admission and the host never re-reads a card the engine leased): //! 1. `IGNEUM_PROVE_MEM_BUDGET_MB`: the lease's grant, the hard ceiling for this process (the engine's coordinator, //! app/igneum-app `src/device.rs`, passes it with `IGNEUM_PROVE_DEVICE`, `IGNEUM_PROVE_WORKLOAD`, -//! `IGNEUM_PROVE_MEM_FREE_MB` and `IGNEUM_PROVE_DEADLINE_S` on every spawn); -//! 2. `IGNEUM_PROVE_MEM_FREE_MB`: the engine's own read at admission, when no grant is given; +//! `IGNEUM_PROVE_MEM_FREE_MB` and `IGNEUM_PROVE_DEADLINE_S` on every spawn); with `IGNEUM_PROVE_MEM_FREE_MB` +//! beside it the LESSER of the two decides (`lease_reading`); +//! 2. `IGNEUM_PROVE_MEM_FREE_MB` alone: the engine's own read at admission, when no grant is given; //! 3. `nvidia-smi --query-gpu=memory.free` on the device (`IGNEUM_PROVE_DEVICE`, else `IGNEUM_CUDA_DEVICE`, else the //! first of `CUDA_VISIBLE_DEVICES`, else 0): the fleet's `tools/fleet/box-prover.py` path and a hand run, where no //! engine admits the job. @@ -187,6 +188,20 @@ impl fmt::Display for Source { } } +/// The reading from the engine's two numbers: the grant is the ceiling and the engine's free reading is a fact, so +/// the row is chosen from the LESSER of the two when both are given (the window lane's engine half, 20:5x UK +/// 8 October 2026, grants a fixed proof budget beside the miner, 7,532 MiB plus 10 percent, while a 12 GB card +/// beside the 6.1 GiB miner has 6,159 MiB free: the grant alone would pick a row the card cannot hold, and the +/// refusal must come from the free figure). One number alone is used as it is. +pub fn lease_reading(budget: Option, free: Option) -> Option<(u64, Source)> { + match (budget, free) { + (Some(b), Some(f)) => Some((b.min(f), Source::Lease)), + (Some(b), None) => Some((b, Source::Lease)), + (None, Some(f)) => Some((f, Source::EngineRead)), + (None, None) => None, + } +} + fn env_u64(name: &str) -> Option { std::env::var(name).ok().and_then(|s| s.trim().parse::().ok()) } @@ -211,11 +226,10 @@ pub fn device_ordinal() -> String { /// The free memory on the card at this call, from the sources in the module note's order. pub fn read_free_mib() -> Option<(u64, Source)> { - if let Some(b) = env_u64("IGNEUM_PROVE_MEM_BUDGET_MB") { - return Some((b, Source::Lease)); - } - if let Some(b) = env_u64("IGNEUM_PROVE_MEM_FREE_MB") { - return Some((b, Source::EngineRead)); + let budget = env_u64("IGNEUM_PROVE_MEM_BUDGET_MB"); + let free = env_u64("IGNEUM_PROVE_MEM_FREE_MB"); + if let Some(choice) = lease_reading(budget, free) { + return Some(choice); } let dev = device_ordinal(); let out = std::process::Command::new("nvidia-smi") @@ -343,6 +357,18 @@ mod tests { assert_eq!(p.tier.recursion_trace_allocation, RECURSION_TRACE_ALLOCATION_SMALL); } + /// The engine's grant beside its free reading: the lesser decides; one alone is used as it is. + #[test] + fn lease_takes_the_lesser_of_grant_and_free() { + assert_eq!(lease_reading(Some(8_285), Some(6_159)), Some((6_159, Source::Lease))); + assert_eq!(lease_reading(Some(8_285), Some(12_287)), Some((8_285, Source::Lease))); + assert_eq!(lease_reading(Some(8_285), None), Some((8_285, Source::Lease))); + assert_eq!(lease_reading(None, Some(12_287)), Some((12_287, Source::EngineRead))); + assert_eq!(lease_reading(None, None), None); + // a 12 GB card beside the miner under a fixed grant is refused by its free figure + assert!(choose(Workload::Shard, lease_reading(Some(8_285), Some(6_159)).unwrap().0).is_err()); + } + /// The host's modes map to the three workloads; modes that never prove map to none. #[test] fn workload_from_mode() { From 7641dd4f9cef37a39afabd5bb5bb26c6a41e2702 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 20:39:40 +0000 Subject: [PATCH 03/13] Fast-time two-node cache-history case (review B F01, Phase 1 item c): a proof's bytes refused once for context pay later as their own record from the cached facts with no second verify; invalid bytes are refused again from the cache Three local nodes at fast time; the real proof is one of the chain's own blocks, exported from H1, cut by igneum-prove-export and proven compressed by igneum-prove-host on the box's CPU inside the run, so its statement is H1's native one. Verdict: paid once; H1's verifies.run grew by exactly 2 (the real proof, the invalid bytes); cacheAnswers by at least 2; the context refusal read; the invalid bytes refused again with no new verify. Needs the node's verifies counters (verdict-cache-fix-node, the status JSON) and node and host on the same pair. Co-Authored-By: Claude Fable 5.1 --- igneum-pow/src/accept.rs | 296 +------------ igneum-pow/src/emit.rs | 452 +++----------------- igneum-pow/src/generator.rs | 483 +++------------------- igneum-pow/src/main.rs | 59 +-- igneum-pow/src/memhard.rs | 93 +---- igneum-pow/src/verify.rs | 279 +------------ igneum-pow/tests/packs.rs | 226 ---------- infra/fast-time/proving-cache-history.mjs | 253 ++++++++++++ 8 files changed, 424 insertions(+), 1717 deletions(-) create mode 100644 infra/fast-time/proving-cache-history.mjs diff --git a/igneum-pow/src/accept.rs b/igneum-pow/src/accept.rs index e02c92efe..9937dba06 100644 --- a/igneum-pow/src/accept.rs +++ b/igneum-pow/src/accept.rs @@ -111,13 +111,6 @@ pub enum Reject { HotItemSite { site: u8, distinct: u32, evaluations: u32, ratio_milli: u32, top_index: u32, top_count: u32 }, /// (c): output bit `bit` was set in `ones` of 2,048 hashes. OutputBias { bit: u8, ones: u32 }, - /// The reg64 window's liveness rule (the coordinator's spec of 8 October 2026, `check_window_liveness`, a research - /// class, not wired into the acceptance): xoring register `reg` of lane `lane` with a probe word at the start of iteration 0 - /// left the iteration's first load address unchanged (`address_changed` false) or the final hash unchanged - /// (`result_changed` false). A window whose address fold reads a subset of the registers is refused here. - DeadWindowRegister { reg: u8, lane: u8, address_changed: bool, result_changed: bool }, - /// `check_window_liveness` on a program without the reg64 window. - NotAWindow, /// (c): the distinct-address sum was `sum`. DistinctAddresses { sum: u64 }, } @@ -140,13 +133,6 @@ impl std::fmt::Display for Reject { Reject::LowEntropySite { site, distinct, evaluations, ratio_milli } => write!(f, "(c'') load site {site} read {distinct} distinct word indices over {evaluations} evaluations, {}.{:03} of a uniform source on its window (floor {MIN_DISTINCT_RATIO_V4} at 2^20)", ratio_milli / 1000, ratio_milli % 1000), Reject::SaturatedSource { site, count } => write!(f, "(c') load site {site} read a saturated source value in {count} of 16384 evaluations (limit 163)"), Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"), - Reject::DeadWindowRegister { reg, lane, address_changed, result_changed } => write!( - f, - "(reg64 liveness) xoring register {reg} of lane {lane} with a probe word at the start of iteration 0 left the first load address {} and the result {}", - if *address_changed { "changed" } else { "unchanged" }, - if *result_changed { "changed" } else { "unchanged" } - ), - Reject::NotAWindow => write!(f, "(reg64 liveness) not a reg64 window class"), Reject::DistinctAddresses { sum } => { write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0) } @@ -227,116 +213,6 @@ pub fn check_distinct_indices_v4(p: &Program) -> Result<(), Reject> { distinct_ratio_pass(p, ACCEPT_UNITS_DISTINCT_V4, MIN_DISTINCT_RATIO_V4).map(|_| ()) } -/// The index-bit bias read of class v6 lane 1 (`docs/analysis/class-v6/family-gate.md` section 3.2, the value-level -/// test of adv-cache-2; `docs/design/class-v6-rotating-family.md` section 5.1): a research-class check beside (c''), -/// NOT wired into the acceptance (as a refusal it would redraw about 40 percent of epochs on the plain `load_index`; -/// with the index fold in it is the guard that the fold holds). Per load site, in load order, the one-count of every -/// address bit over the site's `units x 32 x 8` word indices on the rule's closed form (`2^28` words) against n / 2, -/// in sigma (`sigma = sqrt(n) / 2`, 512 at the (c'') sample of 2^20). The free bits of a site are those at or above -/// the width's alignment and below the window's cut (`28 - k`, `k = min(win, 2)`); a bit the window fixes reads 0 or -/// n by construction and is reported as 0 sigma. `Err` when the run hits a lane-constant site. The known-failed set is -/// lane D's: on the plain address, F8's p4, p8, p10, p15, p34 (class v4) and the gate's p212, p225 (class v5) each -/// carry one site over 6 sigma at address bit R or R + 1 (the era's rotation); with the fold every one reads under 6. -pub fn index_bit_sigma(p: &Program, units: usize) -> Result, Reject> { - const BITS: usize = ACCEPT_DATASET_LOG2 as usize; - let indices = site_indices(p, units)?; - let n = (units * LANES * ITERATIONS) as f64; - let sigma = n.sqrt() / 2.0; - let mut out = Vec::with_capacity(indices.len()); - let loads: Vec<&Instr> = p.instrs.iter().filter(|i| i.op.is_load()).collect(); - for (site, ix) in indices.iter().enumerate() { - let ins = loads[site]; - let k = (ins.win as u32).min(ACCEPT_DATASET_LOG2.saturating_sub(26)); - let low = (ins.width as u32).trailing_zeros(); - let high = ACCEPT_DATASET_LOG2 - k; - let mut ones = [0u64; BITS]; - for &x in ix { - for (b, o) in ones.iter_mut().enumerate() { - *o += ((x >> b) & 1) as u64; - } - } - let mut z = [0.0f64; BITS]; - for b in 0..BITS { - if (b as u32) >= low && (b as u32) < high { - z[b] = (ones[b] as f64 - n / 2.0) / sigma; - } - } - out.push(z); - } - Ok(out) -} - -/// The largest `|sigma|` of [`index_bit_sigma`] with its site and bit: `(sigma, site, bit)`, signed as read. -pub fn index_bit_sigma_max(p: &Program, units: usize) -> Result<(f64, usize, usize), Reject> { - let z = index_bit_sigma(p, units)?; - let mut best = (0.0f64, 0usize, 0usize); - for (site, row) in z.iter().enumerate() { - for (bit, &v) in row.iter().enumerate() { - if v.abs() > best.0.abs() { - best = (v, site, bit); - } - } - } - Ok(best) -} - -/// The reg64 window's liveness rule (the coordinator's spec, 8 October 2026; a research class, beside the census, -/// NOT wired into the acceptance): the window's state must stay live across the dependent memory chain, not only -/// inside an arithmetic block. For each of the 64 registers in turn, the register is xored with each of two -/// seed-derived pseudo-random words ([`REG64_PROBE_WORDS`]) in every lane of unit 0 (the seed's first acceptance base nonce) at the start of iteration 0, on the closed-form dataset of -/// the acceptance; the program passes when, in every lane, the iteration's first load reads a different word index -/// and the final hash differs. A window whose load addresses read a subset of the registers (the arithmetic-only -/// form, where a load's address is its own window's register) is refused at the first register outside that -/// subset; the full-chain form (every address source `src ^ m`, `m` the rotate-xor chain over the 63 other -/// registers) passes, both patterns moving the address and the result in every lane. Two pseudo-random words, -/// not the complement and not one bit: the chain is linear (xor and rotate), so a pattern that enters it twice -/// through a copy made by the program (the pinned draw's instruction 1, `r15 ^= r8`) cancels when the two copies -/// agree after their rotations, which the complement always does (all ones under any rotation) and a single bit -/// does whenever the rotations agree mod 32; a one-bit flip is also lost through a carry before the first load. -/// A dead register fails both words always; a live one fails both with probability about 2^-60. A program without the window is `Reject::NotAWindow`. -pub const REG64_PROBE_WORDS: usize = 2; - -/// The two probe words of register `reg` under the program's seed: splitmix32 of the seed's first word, the -/// register and the word index, never 0, never all ones, never a single bit (the patterns a linear fold can lose). -pub fn reg64_probe_words(seed: &[u32; 8], reg: usize) -> [u32; REG64_PROBE_WORDS] { - let mut out = [0u32; REG64_PROBE_WORDS]; - for (j, w) in out.iter_mut().enumerate() { - let mut x = splitmix32(seed[0] ^ (reg as u32).wrapping_mul(0x9e3779b9) ^ (j as u32 + 1).wrapping_mul(0x85ebca6b)); - while x == 0 || x == u32::MAX || x.is_power_of_two() { - x = splitmix32(x.wrapping_add(0x6c62272e)); - } - *w = x; - } - out -} - -pub fn check_window_liveness(p: &Program) -> Result<(), Reject> { - use crate::verify::{interpret_warp_probe, Probe}; - if !p.class.reg64 { - return Err(Reject::NotAWindow); - } - let shape = crate::memhard::Shape::for_class(&p.class); - let ds = crate::verify::DatasetSource::from_key_shape(p.seed, crate::verify::DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2, shape); - let base = accept_base_nonces_n(&p.seed, 1)[0]; - let mut plain = Probe::default(); - let plain_out = interpret_warp_probe(p, &p.seed, base, &ds, &mut plain); - assert!(plain.seen_first_load, "a program has a load in every iteration"); - for reg in 0..p.registers() { - for word in reg64_probe_words(&p.seed, reg) { - let mut pr = Probe { flip: Some((reg, word)), ..Default::default() }; - let out = interpret_warp_probe(p, &p.seed, base, &ds, &mut pr); - for lane in 0..LANES { - let address_changed = pr.first_load_idx[lane] != plain.first_load_idx[lane]; - let result_changed = out.hashes[lane] != plain_out.hashes[lane]; - if !address_changed || !result_changed { - return Err(Reject::DeadWindowRegister { reg: reg as u8, lane: lane as u8, address_changed, result_changed }); - } - } - } - } - Ok(()) -} - /// One ratio pass over `units`: every load site's distinct word indices against the uniform expectation on its /// window (`N - N^2 / 2W`, the window `2^28 >> min(win, 2)` words of the closed-form dataset), `Err` at the first /// site under `floor`, else the minimum ratio and its site. @@ -368,10 +244,10 @@ pub fn site_window_words(ins: &Instr) -> u64 { (1u64 << ACCEPT_DATASET_LOG2) >> (ins.win as u64).min(2) } -/// One interpreter run over `units` units of the seed's acceptance stream with every load site's word indices kept, -/// in evaluation order: one vector per site in load order (the ratio pass sorts them; the index-bit read of class v6 -/// lane 1 counts bits over them). -pub fn site_indices(p: &Program, units: usize) -> Result>, Reject> { +/// One interpreter run over `units` units with every load site's word indices kept, then per site the sorted +/// run lengths: [`SiteIndexStats`] per site in load order. The ratio pass (c'') and the hot-item rule (c''') read +/// the same run, so class v5 pays the sort once. +pub fn site_index_stats(p: &Program, units: usize) -> Result, Reject> { let loads = p.loads_per_hash(); let sites = loads / ITERATIONS; let mut acc = Acc { @@ -388,16 +264,8 @@ pub fn site_indices(p: &Program, units: usize) -> Result>, Reject> for (unit, &base) in accept_base_nonces_n(&p.seed, units).iter().enumerate() { run_unit(p, unit, base, &mut acc, &mut lane_addrs)?; } - Ok(acc.indices.take().unwrap()) -} - -/// One interpreter run over `units` units with every load site's word indices kept, then per site the sorted -/// run lengths: [`SiteIndexStats`] per site in load order. The ratio pass (c'') and the hot-item rule (c''') read -/// the same run, so class v5 pays the sort once. -pub fn site_index_stats(p: &Program, units: usize) -> Result, Reject> { - let mut indices = site_indices(p, units)?; - let mut out = Vec::with_capacity(indices.len()); - for ix in indices.iter_mut() { + let mut out = Vec::with_capacity(sites); + for ix in acc.indices.take().unwrap().iter_mut() { ix.sort_unstable(); let mut st = SiteIndexStats { distinct: 0, pairs: 0, top_index: 0, top_count: 0 }; let mut i = 0; @@ -486,10 +354,8 @@ pub fn check_indices_v5(p: &Program) -> Result<(), Reject> { /// the era set aside): the shape the sub-version 2 rules (a') and (c') apply to, on every draw path. pub fn is_class_v4_shape(class: &LoadClass) -> bool { matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. })) - // class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside; - // class v6 lane 1's index fold and re-weight table are set aside too (the address path and the op table are - // not the shape) - && LoadClass { era: None, shadow: None, state: false, fold: false, rw: 0, ..*class } == LoadClass { shadow: None, ..V4_CLASS } + // class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside + && LoadClass { era: None, shadow: None, state: false, ..*class } == LoadClass { shadow: None, ..V4_CLASS } } /// One pass of the dataflow freshness over the base program then the shadow block (the order of one iteration), @@ -1017,152 +883,6 @@ mod tests { assert_eq!(v5_under, 0); } - /// The (c''') reading on a listed set of chain-shaped programs (run by hand on a box through the lease): the file - /// `IGNEUM_V5_PROGRAMS` holds one program per line, ` [tag...]` - /// (a hex field of 64 characters is the 32 bytes; anything else is a label whose `seed_words_from_bytes` words are - /// the bytes, as the f8 and adv label spaces derive theirs). Per line: the class v4 program at that attempt (its id, - /// minimum site and ratio at the 2^20 sample), the class v5 verdict at the same attempt, and the class v5 draw's - /// attempt. The coordinator's question of 7 October 2026, 23:3x UK: whether the drawn-era programs adv-cache-2 read - /// as biased (R from 3 to 22) sit under the 0.995 floor. - #[test] - #[ignore] - fn v5_floor_on_listed_programs() { - use crate::generator::{V5_CLASS, V3_ALLOWED}; - use crate::seed::seed_words_from_bytes; - let path = std::env::var("IGNEUM_V5_PROGRAMS").expect("IGNEUM_V5_PROGRAMS="); - let text = std::fs::read_to_string(&path).expect("the programs file"); - let bytes_of = |f: &str| -> Vec { - if f.len() == 64 && f.chars().all(|c| c.is_ascii_hexdigit()) { - crate::bind::unhex(f).unwrap() - } else { - seed_words_from_bytes(f.as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect() - } - }; - let mut refused = 0; - let mut rows = 0; - for line in text.lines() { - let line = line.trim(); - if line.is_empty() || line.starts_with('#') { - continue; - } - let f: Vec<&str> = line.split_whitespace().collect(); - if f.len() < 3 { - println!("skip (needs epoch era attempt): {line}"); - continue; - } - let (epoch, era) = (bytes_of(f[0]), bytes_of(f[1])); - // "draw" = the class v4 chain draw's own attempt for this seed (the attempt adv-cache-2's table shows) - let attempt = if f[2] == "draw" { f8_draw(&epoch, &era, V4_CLASS).attempt } else { f[2].parse::().expect("attempt") }; - let tag = f[3..].join(" "); - let label = format!("igneum-epoch/{}", epoch.iter().map(|b| format!("{b:02x}")).collect::()); - let v4 = candidate_class(&label, &epoch, attempt, LoadClass::era(V4_CLASS, &era, &V3_ALLOWED)); - let st = site_index_stats(&v4, ACCEPT_UNITS_DISTINCT_V4).unwrap(); - let (min, site) = distinct_ratio_on(&v4, &st, ACCEPT_UNITS_DISTINCT_V4, 0.0).unwrap(); - let v4_verdict = check(&v4).is_ok(); - let v5 = candidate_class(&label, &epoch, attempt, LoadClass::era(V5_CLASS, &era, &V3_ALLOWED)); - let v5_verdict = check(&v5); - let drawn = f8_draw(&epoch, &era, V5_CLASS); - let era_params = v4.class.era.as_ref().map(|e| e.stride_rot).unwrap_or(0); - rows += 1; - if v5_verdict.is_err() { - refused += 1; - } - println!( - "listed: id {:016x} attempt {attempt} R {era_params} | v4 {} min site {site} ratio {min:.4} (most read {:#x} x{}) | v5 at this attempt {} | v5 draw attempt {} id {:016x} | {tag}", - crate::generator::program_id(crate::generator::GENERATOR_VERSION_V4, &v4.seed, attempt), - if v4_verdict { "accepted" } else { "REJECTED" }, - st[site].top_index, - st[site].top_count, - match &v5_verdict { Ok(_) => "accepted".to_string(), Err(r) => format!("REFUSED {r}") }, - drawn.attempt, - drawn.program_id() - ); - } - println!("listed: {refused} of {rows} refused under class v5 at the listed attempt"); - } - - /// Class v5's attempts census (run by hand on a box through the lease: `IGNEUM_V5_CENSUS_SEEDS` seeds of the f8 label - /// space from `IGNEUM_V5_CENSUS_FROM`, `IGNEUM_V5_CENSUS_THREADS` threads): every candidate of the class v5 chain draw - /// with the FIRST failing part of the rule, so spec 1.4.7's census paragraph states class v5's own per-candidate - /// rejection and its split by part ((a), (b), (a'), (c), (c'), (c''), (c''')), the exhaustions (must be 0) and - /// P(256 consecutive rejections) = r^256, in the form the crypto lane's attempts census gives for sub-version 3 - /// (per-candidate rejection 0.68, (a') 83.3 percent of rejections). - #[test] - #[ignore] - fn v5_attempts_census() { - use crate::generator::{V5_CLASS, V3_ALLOWED}; - use std::sync::atomic::{AtomicU32, Ordering}; - let seeds: u32 = std::env::var("IGNEUM_V5_CENSUS_SEEDS").ok().and_then(|v| v.parse().ok()).unwrap_or(1000); - let from: u32 = std::env::var("IGNEUM_V5_CENSUS_FROM").ok().and_then(|v| v.parse().ok()).unwrap_or(0); - let threads: usize = std::env::var("IGNEUM_V5_CENSUS_THREADS").ok().and_then(|v| v.parse().ok()).unwrap_or(48); - let next = AtomicU32::new(from); - let t0 = std::time::Instant::now(); - let part_of = |r: &Reject| -> &'static str { - match r { - Reject::StaleLoadSource { .. } => "(a) stale load", - Reject::NoInjectingWrite { .. } => "(b) no injecting write", - Reject::UnfreshLoadSource { .. } => "(a') unfresh source", - Reject::ConstantBit { .. } => "(c) constant bit", - Reject::LaneConstantSite { .. } => "(c) lane-constant site", - Reject::Saturated { .. } => "(c) saturated", - Reject::OutputBias { .. } => "(c) output bias", - Reject::DeadWindowRegister { .. } => "(reg64 liveness) dead window register", - Reject::NotAWindow => "(reg64 liveness) not a window", - Reject::DistinctAddresses { .. } => "(c) distinct addresses", - Reject::SaturatedSource { .. } => "(c') saturated source", - Reject::RepeatedSource { .. } => "(c'') repeated source", - Reject::LowEntropySite { .. } => "(c'') low-entropy site", - Reject::HotItemSite { .. } => "(c''') hot-item site", - } - }; - // per seed: (k, accepted attempt or None, the parts of every rejected candidate) - let rows: Vec<(u32, Option, Vec<&'static str>)> = std::thread::scope(|sc| { - let hs: Vec<_> = (0..threads).map(|_| sc.spawn(|| { - let mut out = Vec::new(); - loop { - let k = next.fetch_add(1, Ordering::Relaxed); - if k >= from + seeds { - break out; - } - let (epoch, era) = f8_seed(k); - let class = LoadClass::era(V5_CLASS, &era, &V3_ALLOWED); - let label = f8_label(&epoch); - let mut parts = Vec::new(); - let mut accepted = None; - for attempt in 0..crate::generator::MAX_ATTEMPTS_V4 { - let c = candidate_class(&label, &epoch, attempt, class); - match check(&c) { - Ok(_) => { accepted = Some(attempt); break; } - Err(r) => parts.push(part_of(&r)), - } - } - out.push((k, accepted, parts)); - } - })).collect(); - let mut rows: Vec<_> = hs.into_iter().flat_map(|h| h.join().unwrap()).collect(); - rows.sort_by_key(|r| r.0); - rows - }); - let rejected: usize = rows.iter().map(|r| r.2.len()).sum(); - let candidates = rejected + rows.iter().filter(|r| r.1.is_some()).count(); - let exhausted = rows.iter().filter(|r| r.1.is_none()).count(); - let r = rejected as f64 / candidates as f64; - let mut by_part = std::collections::BTreeMap::new(); - for row in &rows { - for p in &row.2 { - *by_part.entry(*p).or_insert(0usize) += 1; - } - } - println!("v5_attempts_census: {} seeds {from}..{} in {:.0} s on {threads} threads", rows.len(), from + seeds, t0.elapsed().as_secs_f64()); - println!("candidates {candidates}, rejected {rejected}, per-candidate rejection {r:.4}, exhaustions {exhausted}, P(256 consecutive) {:.3e}", r.powf(256.0)); - for (p, n) in &by_part { - println!(" first failing part {p}: {n} ({:.1} percent of rejections, {:.1} percent of candidates)", *n as f64 * 100.0 / rejected as f64, *n as f64 * 100.0 / candidates as f64); - } - let mean = rows.iter().filter_map(|r| r.1).map(|a| a as f64).sum::() / rows.iter().filter(|r| r.1.is_some()).count() as f64; - println!("accepted attempt mean {mean:.3}"); - assert_eq!(exhausted, 0, "no class v5 seed exhausts its attempts"); - } - #[test] fn distinct_bound_scales_with_the_load_count() { assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM); diff --git a/igneum-pow/src/emit.rs b/igneum-pow/src/emit.rs index 25f14e983..90105ba01 100644 --- a/igneum-pow/src/emit.rs +++ b/igneum-pow/src/emit.rs @@ -16,112 +16,25 @@ use crate::memhard::{ CHACHA_ROUNDS, CHACHA_SIGMA, HOT_TAG, ITEM_ROUNDS, }; use crate::seed::SplitMix64; -use crate::verify::{window, window32, DatasetGeom, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT, INDEX_FOLD_SHIFT}; +use crate::verify::{window, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT}; /// The index expression of a dataset load (era layout, `docs/plans/era-layout.md` section 1.3). For every class /// without an era it is the lottery hash's `rN & MASK`; for an era program it is the one form /// `((rotl_imm(rN * M, R) & WM) | OFF) & MASK` with the site's window constants at the pack's dataset size. -/// Under the multiply-shift geometry (research class ds55, `--dataset-words`) the AND is the dialect's high -/// multiply by the word count: `mulhi(rN, N)` and `mulhi(((rotl_imm(rN * M, R) & WM32) | OFF32), N)` with the -/// window in the source space (`verify::window32`). -/// -/// Class v6 lane 1 (`EraParams::fold`, `verify::stride`): the rotation's argument becomes `fold16_(rN * M)`, the -/// kernel's `y ^ (y >> 16)` helper, in every dialect; the text of every era without the fold is unchanged. -fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, geom: DatasetGeom) -> String { +fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, dataset_log2: u32) -> String { let mask_name = match dialect { CoreDialect::Metal => "MASK", _ => "mask", }; - if geom.mulshift { - let (mulhi, words) = mulhi_name(dialect); - return match era { - None => format!("{mulhi}({a}, {words})"), - Some(e) => { - let (wm, off) = window32(ins, geom.log2); - format!("{mulhi}(((rotl_imm({}, {}u) & {}) | {}), {words})", stride_product_expr(e, a), e.stride_rot, hex(wm), hex(off)) - } - }; - } - let dataset_log2 = geom.log2; match era { None => format!("{a} & {mask_name}"), Some(e) => { let (wm, off) = window(ins, mask_for(dataset_log2), dataset_log2); - format!("((rotl_imm({}, {}u) & {}) | {}) & {mask_name}", stride_product_expr(e, a), e.stride_rot, hex(wm), hex(off)) + format!("((rotl_imm({a} * {}, {}u) & {}) | {}) & {mask_name}", hex(e.stride_mul), e.stride_rot, hex(wm), hex(off)) } } } -/// The product the era's rotation takes: `rN * M`, or `fold16_(rN * M)` under the index fold of class v6 lane 1. -fn stride_product_expr(e: &EraParams, a: &str) -> String { - if e.fold { - format!("fold16_({a} * {})", hex(e.stride_mul)) - } else { - format!("{a} * {}", hex(e.stride_mul)) - } -} - -/// The `fold16_` helper of a kernel under the index fold (empty for every other program, so every pinned pack keeps -/// its text): `y ^ (y >> 16)`, one xor and one shift on the address path, the same in the three dialects. -fn fold_helper_lines(dialect: CoreDialect, p: &Program) -> String { - if !p.class.era.map(|e| e.fold).unwrap_or(false) { - return String::new(); - } - let (qual, u) = match dialect { - CoreDialect::Metal => ("inline", "uint"), - CoreDialect::Cuda => ("__device__ __forceinline__", "uint32_t"), - CoreDialect::OpenCl => ("static inline", "uint"), - }; - format!( - "// Class v6 lane 1, the index fold (8 October 2026, docs/design/class-v6-rotating-family.md section 2; a research class, NOT the lottery\n\ - // hash): every era load address rotates fold16_(src * STRIDE_MUL) instead of the bare product, y ^ (y >> {}), so no era's rotation lands\n\ - // a biased product bit (bit 0 of an odd product is bit 0 of src; bit 1 is set at 3/8) on an address bit.\n\ - {qual} {u} fold16_({u} y) {{ return y ^ (y >> {}u); }}\n", - INDEX_FOLD_SHIFT, INDEX_FOLD_SHIFT - ) -} - -/// The dialect's high 32-bit multiply and the name of the word-count constant (the multiply-shift geometry). -fn mulhi_name(dialect: CoreDialect) -> (&'static str, &'static str) { - match dialect { - CoreDialect::Metal => ("mulhi", "DS_WORDS"), - CoreDialect::Cuda => ("__umulhi", "IGNEUM_DS_WORDS"), - CoreDialect::OpenCl => ("mul_hi", "IGNEUM_DS_WORDS"), - } -} - -/// The word-count lines of a kernel under the multiply-shift geometry (empty under the mask path, so every pinned -/// pack keeps its text): the define the load expressions read, and the comment that says what moved. -fn ds_words_lines(dialect: CoreDialect, geom: DatasetGeom) -> String { - if !geom.mulshift { - return String::new(); - } - let (mulhi, words) = mulhi_name(dialect); - let mut s = String::new(); - s.push_str(&format!( - "// Research class ds55 (8 October 2026, NOT the lottery hash): the dataset holds {} words ({} items, {} bytes),\n", - geom.words, - geom.items(), - geom.bytes() - )); - s.push_str("// not a power of two. Every load address is the multiply-shift range reduction of spec 01 section 1.13.3,\n"); - s.push_str(&format!("// idx = (src * {words}) >> 32 in 64 bits ({mulhi}), in place of src & mask; the mask argument is not read by a load.\n")); - s.push_str(&format!("#define {words} {}\n", hex(geom.words as u32))); - s -} - -/// The warp-coalesced load's base expression (lever b, `wload`): lane 0's register range-reduced and aligned -/// down to 32 words. -fn wload_base_expr(dialect: CoreDialect, bcast: &str, geom: DatasetGeom) -> String { - if geom.mulshift { - let (mulhi, words) = mulhi_name(dialect); - format!("({mulhi}({bcast}, {words}) & ~31u) + lane") - } else { - let wmask = if dialect == CoreDialect::Metal { "WMASK" } else { "wmask" }; - format!("({bcast} & {wmask}) + lane") - } -} - /// The era lines of program.h (empty without an era). fn era_header_lines(p: &Program) -> String { let Some(e) = p.class.era else { return String::new() }; @@ -138,10 +51,6 @@ fn era_header_lines(p: &Program) -> String { s.push_str(&format!("#define IGNEUM_ERA_STRIDE_ROT {}\n", e.stride_rot)); s.push_str(&format!("#define IGNEUM_ERA_INTERLEAVE {{ {}, {}, {}, {} }}\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3])); s.push_str(&format!("#define IGNEUM_ERA_WINDOWS {}\n", jstr(&era_windows(p)))); - if e.fold { - s.push_str(&format!("// Class v6 lane 1 (research): the product is folded before the rotation, y = src * STRIDE_MUL; y ^= y >> {INDEX_FOLD_SHIFT}; y = rotl(y, STRIDE_ROT).\n")); - s.push_str(&format!("#define IGNEUM_ERA_INDEX_FOLD {INDEX_FOLD_SHIFT}\n")); - } s } @@ -202,7 +111,7 @@ enum WideSource { /// same shape in the three dialects: the vector loads differ (`uint4` pointer on Metal and CUDA, `vload4` on /// OpenCL C 1.2). For `width == 1` the caller emits the lottery hash's one-word form instead. fn wide_load_stmt(dialect: CoreDialect, d: &str, idx: &str, width: u8, src: WideSource, closed: Option<(u32, u32)>) -> String { - debug_assert!(width == 4 || width == 8 || width == 16); + debug_assert!(width == 4 || width == 16); let (u, base_ptr) = match dialect { CoreDialect::Metal => ("uint", "dataset"), CoreDialect::Cuda => ("uint32_t", "ds"), @@ -315,14 +224,6 @@ fn class_header_lines(p: &Program) -> String { } s.push_str(&format!("#define IGNEUM_LOAD_CLASS {} ", jstr(&p.class.name()))); - if p.class.rw != 0 { - // class v6 lane 1: the op roll's table (program.json "op_weights" has the weights) - let (w, sum) = p.class.nonload_weights(); - s.push_str(&format!("// Class v6 lane 1 (research): the op roll draws from re-weight table rw{} (sum {sum}) in place of the plain table (sum 75): {}. -", p.class.rw, w.iter().map(|(o, n)| format!("{} {n}", o.name())).collect::>().join(", "))); - s.push_str(&format!("#define IGNEUM_OP_WEIGHTS_TABLE {} -", p.class.rw)); - } if p.class.mixer_mult != 1 || p.class.growth { s.push_str(&format!("#define IGNEUM_CLASS_MIXER_MULT {} ", p.class.mixer_mult)); @@ -337,13 +238,6 @@ fn class_header_lines(p: &Program) -> String { s.push_str(&format!("#define IGNEUM_CLASS_DERIVE_LEN {} ", p.class.derive_len)); } - if p.class.reg64 { - s.push_str("// reg64 (the hash lane's measurement, 8 October 2026, a research class): 64 live registers per lane; r8..r63 = r[k & 7] * 0x9e3779b9 + k;\n"); - s.push_str("// each drawn instruction i runs on window 0 (register field + 8 * (i % 4)) then on window 1 (+32); both windows fold into r0..r7 by xor before the hash fold.\n"); - s.push_str("#define IGNEUM_REG64 1\n"); - s.push_str(&format!("#define IGNEUM_REGISTERS {}\n", p.registers())); - s.push_str(&format!("#define IGNEUM_REG64_ADDRESS_MIX {} // 1: every load's address source is src ^ m, m the rotate-xor chain (m = first; m = rotl(m, 1) ^ next) over the 63 registers other than src, in index order (the full chain)\n", p.address_mix() as u8)); - } s.push_str(&format!("#define IGNEUM_LOAD_SLOTS {} ", p.class.load_slots)); s.push_str(&format!("#define IGNEUM_LOAD_MIX {{ {}, {}, {} }} @@ -967,34 +861,23 @@ const DS_ELEM_BODY: &str = " x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n /// The Metal hash kernel (`generateMSL`, program.metal). pub fn metal_program(p: &Program, dataset_log2: u32, source: LoadSource) -> String { - metal_program_impl(p, DatasetGeom::pow2(dataset_log2), source, false) -} - -/// [`metal_program`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`). -pub fn metal_program_geom(p: &Program, geom: DatasetGeom, source: LoadSource) -> String { - metal_program_impl(p, geom, source, false) + metal_program_impl(p, dataset_log2, source, false) } /// The header-bound Metal kernel (`program_bound.metal`, serve mode of proto-metal): `igneum_hash_bound` reads its /// init words `I` from `constant uint* initw [[buffer(3)]]` (`bind::block_init_words`) instead of `SEEDW`. Same /// instruction text as `igneum_hash`. Stored dataset only. pub fn metal_program_bound(p: &Program, dataset_log2: u32) -> String { - metal_program_impl(p, DatasetGeom::pow2(dataset_log2), LoadSource::Stored, true) + metal_program_impl(p, dataset_log2, LoadSource::Stored, true) } -/// [`metal_program_bound`] at a dataset geometry. -pub fn metal_program_bound_geom(p: &Program, geom: DatasetGeom) -> String { - metal_program_impl(p, geom, LoadSource::Stored, true) -} - -fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: bool) -> String { - let mask = geom.mask(); +fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: bool) -> String { + let mask = mask_for(dataset_log2); let mut s = String::with_capacity(5000); s.push_str("#include \n"); s.push_str("using namespace metal;\n"); s.push('\n'); s.push_str(&format!("#define MASK {}\n", hex(mask))); - s.push_str(&ds_words_lines(CoreDialect::Metal, geom)); s.push_str(&hot_define(p)); s.push_str(&format!("constant uint SEEDW[8] = {{ {} }};\n", join_hex(&p.seed))); s.push('\n'); @@ -1005,13 +888,12 @@ fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: s.push_str(" return x;\n"); s.push_str("}\n"); s.push_str("inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31\n"); - s.push_str(&fold_helper_lines(CoreDialect::Metal, p)); s.push_str("inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); s.push_str("inline uint ds_elem(uint i, uint d0, uint d1) {\n"); s.push_str(" uint x = i ^ d0;\n"); s.push_str(DS_ELEM_BODY); s.push('\n'); - if p.has_wide() && !geom.mulshift { + if p.has_wide() { s.push_str("#define WMASK (MASK & ~31u)\n\n"); } let mut buffer0 = "device const uint* dataset [[buffer(0)]]"; @@ -1047,7 +929,7 @@ fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); } s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(®_decl(p, "uint")); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); if p.has_wide() { s.push_str(" uint lane = gid & 31u;\n"); } @@ -1059,14 +941,13 @@ fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: (i + 1) & 7 )); } - s.push_str(®64_init(p)); s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); let era = p.class.era; let word_index = |a: &str, wide: bool, ins: &Instr| -> String { if wide { - wload_base_expr(CoreDialect::Metal, &format!("simd_broadcast({a}, 0)"), geom) + format!("(simd_broadcast({a}, 0) & WMASK) + lane") } else { - load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, geom) + load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, dataset_log2) } }; let fetch = |idx: String| -> String { @@ -1076,17 +957,10 @@ fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: LoadSource::InlineMemhard(_) => format!("mh_word(cache, {idx})"), } }; - // the drawn program, or its two-window interleaving under reg64 (Program::scheduled, the CUDA text's list); - // reg64 full chain: every load's address source is (rS ^ m), m mixed just before the load (the same chain) - let address_mix = p.address_mix(); - let addr_src = |a: &str| -> String { if address_mix { format!("({a} ^ m)") } else { a.to_string() } }; - for (k, ins) in p.scheduled().iter().enumerate() { + for (k, ins) in p.instrs.iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); let b = format!("r{}", ins.src2); - if address_mix && ins.op == Op::Load { - s.push_str(&format!(" {}\n", reg64_mix_line(p, ins.src))); - } let line = match ins.op { Op::Add => format!( "{d} = {d} + {a} + select({}, {}, ((sel >> {}u) & 1u) != 0u);", @@ -1109,9 +983,9 @@ fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: LoadSource::InlineClosed(d0, d1) => (WideSource::InlineClosed, Some((*d0, *d1))), LoadSource::InlineMemhard(_) => (WideSource::InlineMemhard, None), }; - wide_load_stmt(CoreDialect::Metal, &d, &word_index(&addr_src(&a), false, ins), ins.width, src, closed) + wide_load_stmt(CoreDialect::Metal, &d, &word_index(&a, false, ins), ins.width, src, closed) } - Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&addr_src(&a), false, ins))), + Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false, ins))), Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true, ins))), Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()), Op::Hot => hot_stmt(CoreDialect::Metal, &d, &a), @@ -1120,7 +994,6 @@ fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: } s.push_str(&shadow_block(p, CoreDialect::Metal)); s.push_str(" }\n"); - s.push_str(®64_fold(p)); s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); @@ -1154,78 +1027,14 @@ fn init_line(p: &Program, u: &str, i: usize) -> String { ) } -/// The per-lane register declaration of the kernels (`ty` the dialect's 32-bit word: `uint32_t` in CUDA, `uint` in -/// OpenCL and Metal): r0..r7, or r0..r63 under reg64, with `m` for the full chain. -fn reg_decl(p: &Program, ty: &str) -> String { - if !p.class.reg64 { - return format!(" {ty} r0, r1, r2, r3, r4, r5, r6, r7;\n"); - } - let names: Vec = (0..p.registers()).map(|k| format!("r{k}")).collect(); - let m = if p.address_mix() { format!("\n {ty} m; // reg64 full chain: the address mix of all 64 registers before every load") } else { String::new() }; - format!(" {ty} {}; // reg64: 64 live registers per lane (two 32-register windows){m}\n", names.join(", ")) -} - -/// reg64 full chain: the mix statement before a load, `m = r0; m = rotl_imm(m, 1u) ^ r1; ... ^ r63;` over the 63 -/// registers other than the load's source `src` (the verifier's `addr_src`, the same chain), one line; the same -/// text in the three dialects (`rotl_imm` is defined in each). -fn reg64_mix_line(p: &Program, src: u8) -> String { - let mut s = String::new(); - for k in 0..p.registers() { - if k == src as usize { - continue; - } - if s.is_empty() { - s.push_str(&format!("m = r{k};")); - } else { - s.push_str(&format!(" m = rotl_imm(m, 1u) ^ r{k};")); - } - } - s -} - -/// reg64: r8..r63 from the eight seeded registers, `r[k] = r[k & 7] * 0x9E3779B9 + k` (the verifier's text), the -/// same statements in the three dialects. Empty otherwise. -fn reg64_init(p: &Program) -> String { - if !p.class.reg64 { - return String::new(); - } - let mut s = String::from(" // reg64: the derived registers of both windows\n"); - for k in 8..p.registers() { - s.push_str(&format!(" r{k} = r{} * 0x9e3779b9u + {k}u;\n", k & 7)); - } - s -} - -/// reg64: both windows fold into r0..r7 by xor before the hash fold, the same statements in the three dialects. -/// Empty otherwise. -fn reg64_fold(p: &Program) -> String { - if !p.class.reg64 { - return String::new(); - } - let mut s = String::from(" // reg64: fold the 64 registers into the eight output registers\n"); - for k in 0..8 { - let terms: Vec = (k + 8..p.registers()).step_by(8).map(|j| format!("r{j}")).collect(); - s.push_str(&format!(" r{k} = r{k} ^ {};\n", terms.join(" ^ "))); - } - s -} - /// The instruction lines of the CUDA hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`). -fn cuda_instr_lines(p: &Program, geom: DatasetGeom) -> String { +fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String { let mut s = String::with_capacity(6000); let era = p.class.era; - // the drawn program, or its two-window interleaving under reg64 (Program::scheduled: the verifier reads the same list) - // reg64 full chain: every load's address source is (rS ^ m), m the rotate-xor mix of all 64 registers computed - // just before the load (the verifier's addr_src, the same chain) - let address_mix = p.address_mix(); - let addr_src = |a: &str| -> String { if address_mix { format!("({a} ^ m)") } else { a.to_string() } }; - for (k, ins) in p.scheduled().iter().enumerate() { + for (k, ins) in p.instrs.iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); let b = format!("r{}", ins.src2); - if address_mix && ins.op == Op::Load { - s.push_str(&format!(" {}\n", reg64_mix_line(p, ins.src))); - } let line = match ins.op { // Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well. Op::Add => format!( @@ -1244,10 +1053,10 @@ fn cuda_instr_lines(p: &Program, geom: DatasetGeom) -> String { Op::Mad => format!("{d} = {a} * {b} + {d};"), Op::Shfl => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask), Op::Load if load_width(ins) > 1 => { - wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &addr_src(&a), geom), ins.width, WideSource::Stored, None) + wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None) } - Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &addr_src(&a), geom)), - Op::WLoad => format!("{d} = {d} ^ ds[{}];", wload_base_expr(CoreDialect::Cuda, &format!("__shfl_sync(0xffffffffu, {a}, 0)"), geom)), + Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2)), + Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"), Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()), Op::Hot => hot_stmt(CoreDialect::Cuda, &d, &a), }; @@ -1264,11 +1073,6 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { /// [`cuda_kernel`] at a dataset size (an era program's window constants are literals of the pack's size; every /// other class ignores it). pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { - cuda_kernel_geom(p, memhard, DatasetGeom::pow2(dataset_log2)) -} - -/// [`cuda_kernel_at`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`). -pub fn cuda_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String { let layout = p.class.layout(); let mut s = String::with_capacity(9000); s.push_str(&generated_by(&p.seed_string)); @@ -1283,7 +1087,6 @@ pub fn cuda_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetG s.push_str("#include \"memhard.h\"\n"); } s.push('\n'); - s.push_str(&ds_words_lines(CoreDialect::Cuda, geom)); s.push_str(&hot_define(p)); s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n"); s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); @@ -1293,7 +1096,6 @@ pub fn cuda_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetG s.push_str("}\n"); s.push_str("// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.\n"); s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n"); - s.push_str(&fold_helper_lines(CoreDialect::Cuda, p)); s.push_str("// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.\n"); s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); s.push_str("__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {\n"); @@ -1360,22 +1162,17 @@ pub fn cuda_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetG s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); } s.push_str(" uint32_t nonce = baseNonce + gid;\n"); - s.push_str(®_decl(p, "uint32_t")); + s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n"); if p.has_wide() { - s.push_str(" uint32_t lane = threadIdx.x & 31u;\n"); - if !geom.mulshift { - s.push_str(" uint32_t wmask = mask & ~31u;\n"); - } + s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n"); } for i in 0..8 { s.push_str(&init_line(p, "uint32_t", i)); } - s.push_str(®64_init(p)); s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n")); - s.push_str(&cuda_instr_lines(p, geom)); + s.push_str(&cuda_instr_lines(p, dataset_log2)); s.push_str(&shadow_block(p, CoreDialect::Cuda)); s.push_str(" }\n"); - s.push_str(®64_fold(p)); s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n"); @@ -1475,11 +1272,6 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { /// [`cuda_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]). pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { - cuda_kernel_bound_geom(p, memhard, DatasetGeom::pow2(dataset_log2)) -} - -/// [`cuda_kernel_bound_at`] at a dataset geometry. -pub fn cuda_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String { let mut s = String::with_capacity(9000); s.push_str(&generated_by(&p.seed_string)); s.push_str( @@ -1496,7 +1288,6 @@ pub fn cuda_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: Da s.push_str("#include \n"); s.push_str("#include \"program.h\"\n"); s.push('\n'); - s.push_str(&ds_words_lines(CoreDialect::Cuda, geom)); s.push_str("struct IgneumInitWords { uint32_t w[8]; };\n"); s.push('\n'); s.push_str(&hot_define(p)); @@ -1507,7 +1298,6 @@ pub fn cuda_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: Da s.push_str(" return x;\n"); s.push_str("}\n"); s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n"); - s.push_str(&fold_helper_lines(CoreDialect::Cuda, p)); s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); s.push('\n'); let _ = memhard; // the bound kernel reads the stored dataset in both constructions @@ -1521,12 +1311,9 @@ pub fn cuda_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: Da s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); } s.push_str(" uint32_t nonce = baseNonce + gid;\n"); - s.push_str(®_decl(p, "uint32_t")); + s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n"); if p.has_wide() { - s.push_str(" uint32_t lane = threadIdx.x & 31u;\n"); - if !geom.mulshift { - s.push_str(" uint32_t wmask = mask & ~31u;\n"); - } + s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n"); } for i in 0..8 { s.push_str(&format!( @@ -1535,12 +1322,10 @@ pub fn cuda_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: Da (i + 1) & 7 )); } - s.push_str(®64_init(p)); s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n")); - s.push_str(&cuda_instr_lines(p, geom)); + s.push_str(&cuda_instr_lines(p, dataset_log2)); s.push_str(&shadow_block(p, CoreDialect::Cuda)); s.push_str(" }\n"); - s.push_str(®64_fold(p)); s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n"); @@ -1584,20 +1369,13 @@ pub fn cuda_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: Da } /// The instruction lines of the OpenCL hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`). -fn opencl_instr_lines(p: &Program, geom: DatasetGeom) -> String { +fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String { let mut s = String::with_capacity(6000); let era = p.class.era; - // the drawn program, or its two-window interleaving under reg64 (Program::scheduled, the CUDA text's list); - // reg64 full chain: every load's address source is (rS ^ m), m mixed just before the load (the same chain) - let address_mix = p.address_mix(); - let addr_src = |a: &str| -> String { if address_mix { format!("({a} ^ m)") } else { a.to_string() } }; - for (k, ins) in p.scheduled().iter().enumerate() { + for (k, ins) in p.instrs.iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); let b = format!("r{}", ins.src2); - if address_mix && ins.op == Op::Load { - s.push_str(&format!(" {}\n", reg64_mix_line(p, ins.src))); - } let line = match ins.op { Op::Add => format!( "{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});", @@ -1615,10 +1393,10 @@ fn opencl_instr_lines(p: &Program, geom: DatasetGeom) -> String { Op::Mad => format!("{d} = {a} * {b} + {d};"), Op::Shfl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask), Op::Load if load_width(ins) > 1 => { - wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &addr_src(&a), geom), ins.width, WideSource::Stored, None) + wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None) } - Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &addr_src(&a), geom)), - Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[{}]; }}", wload_base_expr(CoreDialect::OpenCl, "t_", geom)), + Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2)), + Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"), Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()), Op::Hot => hot_stmt(CoreDialect::OpenCl, &d, &a), }; @@ -1636,12 +1414,7 @@ pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { /// [`opencl_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]). pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { - opencl_kernel_bound_geom(p, memhard, DatasetGeom::pow2(dataset_log2)) -} - -/// [`opencl_kernel_bound_at`] at a dataset geometry. -pub fn opencl_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String { - let mut s = opencl_kernel_geom(p, memhard, geom); + let mut s = opencl_kernel_at(p, memhard, dataset_log2); s.push('\n'); s.push_str( "// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.\n", @@ -1666,11 +1439,11 @@ pub fn opencl_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: s.push_str("#endif\n"); s.push_str(&unit_loop); s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(®_decl(p, "uint")); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); } else { s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(®_decl(p, "uint")); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); s.push_str("#if IGNEUM_EXCHANGE == 0\n"); s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); @@ -1680,10 +1453,7 @@ pub fn opencl_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: s.push_str("#endif\n"); } if p.has_wide() { - s.push_str(" uint lane = lid & 31u;\n"); - if !geom.mulshift { - s.push_str(" uint wmask = mask & ~31u;\n"); - } + s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n"); } for i in 0..8 { s.push_str(&format!( @@ -1692,12 +1462,10 @@ pub fn opencl_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: (i + 1) & 7 )); } - s.push_str(®64_init(p)); s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); - s.push_str(&opencl_instr_lines(p, geom)); + s.push_str(&opencl_instr_lines(p, dataset_log2)); s.push_str(&shadow_block(p, CoreDialect::OpenCl)); s.push_str(" }\n"); - s.push_str(®64_fold(p)); s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); @@ -1715,11 +1483,6 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { /// [`opencl_kernel`] at a dataset size (see [`cuda_kernel_at`]). pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { - opencl_kernel_geom(p, memhard, DatasetGeom::pow2(dataset_log2)) -} - -/// [`opencl_kernel_at`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`). -pub fn opencl_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String { let layout = p.class.layout(); let mut s = String::with_capacity(14000); s.push_str(&generated_by(&p.seed_string)); @@ -1735,7 +1498,6 @@ pub fn opencl_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: Datase s.push_str("// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.\n"); s.push_str("#ifndef IGNEUM_GROUP\n#define IGNEUM_GROUP 32\n#endif\n"); s.push_str("#ifndef IGNEUM_EXCHANGE\n#define IGNEUM_EXCHANGE 0\n#endif\n"); - s.push_str(&ds_words_lines(CoreDialect::OpenCl, geom)); s.push_str("#ifdef __OPENCL_VERSION__\n"); s.push_str("#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))\n"); s.push_str("#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]\n"); @@ -1776,7 +1538,6 @@ pub fn opencl_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: Datase s.push_str("}\n"); s.push_str("// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.\n"); s.push_str("static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }\n"); - s.push_str(&fold_helper_lines(CoreDialect::OpenCl, p)); s.push_str("// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.\n"); s.push_str("static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }\n"); s.push_str("static inline uint ds_elem(uint i, uint d0, uint d1) {\n"); @@ -1851,10 +1612,10 @@ pub fn opencl_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: Datase s.push_str("#endif\n"); s.push_str(&unit_loop); s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(®_decl(p, "uint")); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); } else { s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(®_decl(p, "uint")); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); s.push_str("#if IGNEUM_EXCHANGE == 0\n"); s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); s.push_str(" uint xk = 0u;\n"); @@ -1863,20 +1624,15 @@ pub fn opencl_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: Datase s.push_str("#endif\n"); } if p.has_wide() { - s.push_str(" uint lane = lid & 31u;\n"); - if !geom.mulshift { - s.push_str(" uint wmask = mask & ~31u;\n"); - } + s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n"); } for i in 0..8 { s.push_str(&init_line(p, "uint", i)); } - s.push_str(®64_init(p)); s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); - s.push_str(&opencl_instr_lines(p, geom)); + s.push_str(&opencl_instr_lines(p, dataset_log2)); s.push_str(&shadow_block(p, CoreDialect::OpenCl)); s.push_str(" }\n"); - s.push_str(®64_fold(p)); s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); @@ -1911,8 +1667,7 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { let key = &ds.key; let dataset_log2 = ds.log2_words; let memhard = ds.memhard().map(|m| &m.params); - let geom = ds.geom; - let mask = geom.mask(); + let mask = mask_for(dataset_log2); let mut s = String::with_capacity(2600); s.push_str(&generated_by(&p.seed_string)); s.push_str("// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.\n"); @@ -1930,20 +1685,8 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!("#define IGNEUM_DAY_BYTES_HEX {}\n", jstr(&hex_bytes(&ds.key_bytes)))); s.push_str(&format!("#define IGNEUM_DAY0 {}\n", hex(key[0]))); s.push_str(&format!("#define IGNEUM_DAY1 {}\n", hex(key[1]))); - if geom.mulshift { - s.push_str(&format!("// Research class ds55 (8 October 2026): a dataset of {} words, not a power of two. IGNEUM_DATASET_LOG2 is floor(log2(words));\n", geom.words)); - s.push_str("// the host allocates IGNEUM_DATASET_WORDS words and builds IGNEUM_DATASET_ITEMS items; IGNEUM_MASK is the last word index (the\n"); - s.push_str("// self-test reads dataset[IGNEUM_MASK]) and is never ANDed: every load is idx = (src * IGNEUM_DATASET_WORDS) >> 32 (spec 01 section 1.13.3).\n"); - s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n")); - s.push_str(&format!("#define IGNEUM_DATASET_WORDS {}u\n", geom.words)); - s.push_str(&format!("#define IGNEUM_DATASET_ITEMS {}u\n", geom.items())); - s.push_str(&format!("#define IGNEUM_DATASET_BYTES {}ull\n", geom.bytes())); - s.push_str("#define IGNEUM_DATASET_MULSHIFT 1\n"); - s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask))); - } else { - s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n")); - s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask))); - } + s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n")); + s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask))); s.push_str("#define IGNEUM_LANES 32\n"); s.push_str(&format!("#define IGNEUM_ITERATIONS {ITERATIONS}\n")); s.push_str(&format!("#define IGNEUM_INSTR_COUNT {INSTR_COUNT}\n")); @@ -2077,13 +1820,6 @@ pub fn sample_indices(mask: u32) -> Vec { (0..64).map(|_| (sr.next() as u32) & mask).collect() } -/// [`sample_indices`] at a dataset geometry: the same 64 draws through the geometry's range reduction, so the -/// mask path is [`sample_indices`] exactly and the multiply-shift path samples by the mapping the loads use. -pub fn sample_indices_geom(geom: DatasetGeom) -> Vec { - let mut sr = SplitMix64::new(0x6d68_7361_6d70_6c65); - (0..64).map(|_| geom.reduce(sr.next() as u32)).collect() -} - /// vectors.h (`generateVectorsHeader`). pub fn vectors_header( p: &Program, @@ -2167,15 +1903,14 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { let key = &ds.key; let dataset_log2 = ds.log2_words; let memhard = ds.memhard().map(|m| &m.params); - let geom = ds.geom; - let mask = geom.mask(); + let mask = mask_for(dataset_log2); let mut s = String::with_capacity(14000); s.push_str("{\n"); s.push_str(" \"format\": \"igneum-program-pack-3\",\n"); s.push_str(&format!(" \"generator\": {},\n", p.generator)); s.push_str(&format!(" \"attempt\": {},\n", p.attempt)); s.push_str(&format!(" \"program_id\": {},\n", jhex64(p.program_id()))); - s.push_str(&format!(" \"program_id_derivation\": {},\n", jstr(&p.program_id_derivation()))); + s.push_str(" \"program_id_derivation\": \"FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32\",\n"); s.push_str(&format!( " \"dataset_mode\": {},\n", jstr(if memhard.is_some() { "memory-hard" } else { "closed-form" }) @@ -2215,18 +1950,6 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { if !p.class.is_v2() { let c = p.width_counts(); s.push_str(&format!(" \"load_class\": {},\n", jstr(&p.class.name()))); - if p.class.rw != 0 { - // class v6 lane 1: the re-weight table the op roll draws from, in draw order, and its sum (the roll's range) - let (w, sum) = p.class.nonload_weights(); - s.push_str(&format!( - " \"op_weights\": {{\"table\": \"rw{}\", \"sum\": {sum}, \"draw_order\": [{}], \"rule\": \"class v6 lane 1 (8 October 2026, a research class): the table replaces the plain ten-family table (sum 75) for the op roll of every base and shadow instruction, roll = below(sum) walked over the weights in draw order; a family at weight 0 is never drawn; shuffle stays at 4 under rw1; the program id carries 'rw/' || table_u8\"}},\n", - p.class.rw, - w.iter().map(|(o, n)| format!("[{}, {n}]", jstr(o.name()))).collect::>().join(", ") - )); - } - if p.class.reg64 { - s.push_str(&format!(" \"reg64\": {{\"registers\": {}, \"variant\": {}, \"address_mix\": {}, \"liveness\": {}, \"statements_per_iteration\": {}, \"loads_per_hash\": {}, \"rule\": \"the hash lane's 64-register window (8 October 2026, a research class, NOT the lottery hash): r0..r7 seeded as today, r[k] = r[k & 7] * 0x9E3779B9 + k for k in 8..63; instruction i of the drawn program runs on window 0 with every register field + 8 * (i % 4), then on window 1 with +32 more; after the 8 iterations r[k] ^= r[k + 8] ^ ... ^ r[k + 56] for k in 0..7, then the hash fold; the draw, the acceptance rule and the dataset are the class's without the flag; address_mix 1 (the full chain): every load's address source is src ^ m, m the rotate-xor chain (m = first; m = rotl(m, 1) ^ next) over the 63 registers other than src in index order, computed before the load; src stays out of the chain so no register's term can cancel its own direct term (liveness rule: xoring any register with either of two seed-derived probe words at the start of an iteration moves the iteration's first load address and the final hash)\"}},\n", p.registers(), jstr(if p.address_mix() { "window, full chain" } else { "window, arithmetic-only" }), p.address_mix() as u8, jstr(&p.reg64_liveness()), p.scheduled().len(), p.scheduled().iter().filter(|i| i.op.is_load()).count() * ITERATIONS)); - } if p.class.derive_len != 0 { s.push_str(&format!(" \"derive_len\": {},\n", p.class.derive_len)); s.push_str(" \"derive\": \"Counter ASIC 3.0 item 2 (6 October 2026, docs/plans/counter-asic-3-derivation.md; a prototype, not class v3): the nine mixer slots of the item derivation each run a straight-line program of derive_len instructions drawn from the day key stream after the 40 mixer draws, four draws per instruction (op roll below(100), destination roll below(15), third-register roll below(14), the immediate next()); every instruction reads the register the previous one wrote (s[0] first) and writes another; twelve forms, each a bijection on the state; the 8 dependent cache reads per item unchanged; the acceptance test of derive.rs (every register written per round program, 8 distinct rotations, the x8 mixer's operation and multiply counts as floors) rejects a draw and the next attempt continues the stream\",\n"); @@ -2256,15 +1979,7 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!(" \"stride_mul\": {},\n", jhex(e.stride_mul))); s.push_str(&format!(" \"stride_rot\": {},\n", e.stride_rot)); s.push_str(&format!(" \"interleave\": [{}, {}, {}, {}],\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3])); - let y = if e.fold { format!("y = src * stride_mul; y ^= y >> {INDEX_FOLD_SHIFT}; y = rotl(y, stride_rot)") } else { "y = rotl(src * stride_mul, stride_rot)".to_string() }; - if geom.mulshift { - s.push_str(&format!(" \"address\": \"{y}; D = floor(log2(words)); k = min(win, D - 26); v = (y & (0xffffffff >> k)) | ((off & (2^k - 1)) << (32 - k)); idx = (v * words) >> 32 in 64 bits (the window in the source space, then the multiply-shift of spec 01 section 1.13.3); a wide load aligns idx down to W words\",\n")); - } else { - s.push_str(&format!(" \"address\": \"{y}; k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n")); - } - if e.fold { - s.push_str(&format!(" \"index_fold\": \"class v6 lane 1 (8 October 2026, docs/design/class-v6-rotating-family.md section 2, a research class): the product's low bits are folded before the stride rotation, y ^= y >> {INDEX_FOLD_SHIFT}, so no era's rotation lands a biased product bit (bit 0 of an odd product is bit 0 of src; bit 1 is set at 3/8; lane D's 7 of 16 eras at 130 to 511 sigma) on an address bit; the design names the fold and not its form, so this form is the lane's: one xor and one shift on the address path, the same in verify::stride, the three kernel texts (fold16_) and these vectors; the program id carries 'fold/'\",\n")); - } + s.push_str(" \"address\": \"y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n"); s.push_str(" \"windows\": \"per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)\",\n"); s.push_str(" \"dataset_word\": \"dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed\",\n"); s.push_str(" \"program_id_suffix\": \"'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]\"\n"); @@ -2309,34 +2024,17 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str( " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n", ); - if geom.mulshift { - s.push_str(" \"load\": \"dst = dst ^ dataset[(src * dataset.words) >> 32]\",\n"); - s.push_str(" \"wload\": \"base = ((src of lane 0 * dataset.words) >> 32) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); - } else { - s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"); - s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); - } + s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"); + s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); if p.has_hot() { s.push_str(",\n \"hot\": \"dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)\""); } s.push('\n'); s.push_str(" },\n"); s.push_str(" \"dataset\": {\n"); - if geom.mulshift { - s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); - s.push_str(" \"log2_words_note\": \"floor(log2(words)): the dataset is not a power of two (research class ds55, 8 October 2026); allocate words, build items\",\n"); - s.push_str(&format!(" \"words\": {},\n", geom.words)); - s.push_str(&format!(" \"bytes\": {},\n", geom.bytes())); - s.push_str(&format!(" \"items\": {},\n", geom.items())); - s.push_str(" \"mapping\": \"mulshift\",\n"); - s.push_str(" \"index\": \"idx = (src * words) >> 32 computed in 64 bits (spec 01 section 1.13.3, the multiply-shift range reduction; uniform to within 2^-32, branch-free, integer only), in place of src & mask; the item index is idx >> 4 under the linear layout and t(idx) under an era layout, below items\",\n"); - s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); - s.push_str(" \"mask_note\": \"the last word index, words - 1; never ANDed under the multiply-shift\",\n"); - } else { - s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); - s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2))); - s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); - } + s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); + s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2))); + s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); s.push_str(&format!(" \"day\": {},\n", jstr(day))); s.push_str(&format!(" \"day_bytes\": {},\n", jstr(&hex_bytes(&ds.key_bytes)))); s.push_str(" \"day_words_from\": \"seed_words_from_bytes(day_bytes)\",\n"); @@ -2480,36 +2178,12 @@ pub fn vectors_json( source: &str, memhard: bool, ) -> String { - let geom = DatasetGeom::pow2(dataset_log2); - debug_assert_eq!(mask, geom.mask()); - vectors_json_geom(p, day, geom, bases, outs, v, source, memhard) -} - -/// [`vectors_json`] at a dataset geometry: under the multiply-shift the file also carries `dataset_words` and -/// `dataset_mapping`, and `dataset_last_index` is `words - 1`. -#[allow(clippy::too_many_arguments)] -pub fn vectors_json_geom( - p: &Program, - day: &str, - geom: DatasetGeom, - bases: &[u32], - outs: &[[u64; 32]], - v: &PackVectors, - source: &str, - memhard: bool, -) -> String { - let dataset_log2 = geom.log2; - let mask = geom.mask(); let mut s = String::with_capacity(6500); s.push_str("{\n"); s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string))); s.push_str(&format!(" \"day\": {},\n", jstr(day))); s.push_str(&format!(" \"dataset_mode\": {},\n", jstr(if memhard { "memory-hard" } else { "closed-form" }))); s.push_str(&format!(" \"dataset_log2_words\": {dataset_log2},\n")); - if geom.mulshift { - s.push_str(&format!(" \"dataset_words\": {},\n", geom.words)); - s.push_str(" \"dataset_mapping\": \"mulshift\",\n"); - } s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); s.push_str(" \"lanes\": 32,\n"); s.push_str(&format!(" \"source\": {},\n", jstr(source))); @@ -2581,17 +2255,15 @@ impl Pack { pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { let p = &epoch.program; let ds: &DatasetSource = &epoch.dataset; - let geom = ds.geom; - let mask = geom.mask(); + let mask = ds.mask; let memhard = ds.memhard().map(|m| &m.params); let bases = PACK_VECTOR_BASES.to_vec(); let outs: Vec<[u64; 32]> = bases.iter().map(|&b| epoch.hash_warp(b)).collect(); - // the self-test words under the program's layout (era layout; linear for every other class) and the - // geometry's range reduction (the sampled indices go through the mapping the loads use) + // the self-test words under the program's layout (era layout; linear for every other class) let mut v = PackVectors { head: (0..16).map(|i| epoch.dataset_word(i)).collect(), - last: epoch.dataset_word(geom.last_index()), - sample_idx: sample_indices_geom(geom), + last: epoch.dataset_word(mask), + sample_idx: sample_indices(mask), ..Default::default() }; v.sample_val = v.sample_idx.iter().map(|&i| epoch.dataset_word(i)).collect(); @@ -2612,16 +2284,16 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { let is_mh = memhard.is_some(); let mut files = vec![ ("program.json".to_string(), program_json(p, day, ds)), - ("vectors.json".to_string(), vectors_json_geom(p, day, geom, &bases, &outs, &v, source, is_mh)), - ("kernel.cu".to_string(), cuda_kernel_geom(p, memhard, geom)), - ("kernel.cl".to_string(), opencl_kernel_geom(p, memhard, geom)), + ("vectors.json".to_string(), vectors_json(p, day, ds.log2_words, &bases, &outs, &v, mask, source, is_mh)), + ("kernel.cu".to_string(), cuda_kernel_at(p, memhard, ds.log2_words)), + ("kernel.cl".to_string(), opencl_kernel_at(p, memhard, ds.log2_words)), ("program.h".to_string(), program_header(p, day, ds)), ("vectors.h".to_string(), vectors_header(p, &bases, &outs, &v, mask, source, is_mh)), - ("program.metal".to_string(), metal_program_geom(p, geom, LoadSource::Stored)), + ("program.metal".to_string(), metal_program(p, ds.log2_words, LoadSource::Stored)), // Header-bound kernels (3 October 2026, bind.rs): new files, the seven above are unchanged. - ("program_bound.metal".to_string(), metal_program_bound_geom(p, geom)), - ("kernel_bound.cu".to_string(), cuda_kernel_bound_geom(p, memhard, geom)), - ("kernel_bound.cl".to_string(), opencl_kernel_bound_geom(p, memhard, geom)), + ("program_bound.metal".to_string(), metal_program_bound(p, ds.log2_words)), + ("kernel_bound.cu".to_string(), cuda_kernel_bound_at(p, memhard, ds.log2_words)), + ("kernel_bound.cl".to_string(), opencl_kernel_bound_at(p, memhard, ds.log2_words)), ]; if let Some(mp) = memhard { files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp))); diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index 61fe4874b..b3667c8c3 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -242,39 +242,6 @@ pub struct LoadClass { /// the item derivation XORs the window's state leaf into every item before the first mixer (`crate::state`, /// `memhard::derive_items_leaves`). The program draw does not read it. `false` for every other class. pub state: bool, - /// W = 8 (the floor programme, 8 October 2026, a research class behind `w32`): the 4-word slot of the width set - /// reads 8 words (one 32-byte sector) instead of 4; the mix, the slots and every other rule stand. `false` for - /// every other class. See [`LoadClass::widths`]. - pub wide8: bool, - /// Class v6 lane 1 (`docs/design/class-v6-rotating-family.md` section 2, the index fold; a research class, - /// 8 October 2026): `true` folds the product's low bits in every era load address before the stride rotation - /// (`verify::load_index`: `y = x * M; y ^= y >> 16; y = rotl(y, R)`), so no era's R lands a biased product bit - /// on an address bit. The class's era carries the same bit ([`EraParams::fold`]). `false` for every other class. - pub fold: bool, - /// Class v6 lane 1, the op-mix re-weight behind the fold: 0 is the plain table [`NONLOAD_WEIGHTS`]; 1 the k lane's - /// optimiser split [`NONLOAD_WEIGHTS_RW`] (sum 83, `or` never drawn); 2 the census lane's neighbouring table - /// [`NONLOAD_WEIGHTS_RW2`] (sum 75). The draw rolls against the table's own sum ([`LoadClass::nonload_weights`]). - pub rw: u8, - /// The 64-register window (the hash lane's reg64 measurement, 8 October 2026, a research class behind `+reg64` - /// and `--reg64`): each lane holds 64 live 32-bit registers. r0..r7 are seeded as today, r8..r63 derived from them - /// (`r[k] = r[k & 7] * 0x9E3779B9 + k`); the 64 drawn instructions run twice per iteration in an interleaved - /// order, instruction i on window 0 (register field + 8 * (i % 4), registers 0..31) then the same instruction on - /// window 1 (+32, registers 32..63); the windows fold into r0..r7 by xor before the hash fold - /// ([`Program::scheduled`], [`crate::verify`], the emitters). The draw, the acceptance rule and the dataset are the - /// class's without the flag. `false` for every other class. - pub reg64: bool, - /// reg64, the full chain (the coordinator's amendment of 8 October 2026, 15:1x UK, class suffix `+reg64c`, - /// `--reg64-chain`): the address of every load consumes all 64 registers: address source `src ^ m`, `m` the - /// rotate-xor chain (`m = first; m = rotl(m, 1) ^ next`) over the 63 registers other than `src` in index order - /// (the same text in the verifier and the emitters), so an in-flight hash holds 64 independently necessary - /// values for the length of the dependent memory chain. The source stays out of the chain: inside it, r31 and - /// r63 land at rotation 0 mod 32 and cancel their own direct term (found by the liveness rule on the pinned draw). - /// Init rule: r0..r7 from the seed words as every class, `r[k] = r[k & 7] * 0x9E3779B9 + k` for k in 8..63. - /// Output rule: `r[k] ^= r[k + 8] ^ r[k + 16] ^ ... ^ r[k + 56]` for k in 0..7, then the class's hash fold. - /// Liveness rule (`accept::check_window_liveness`): xoring any one register with either of two seed-derived - /// probe words at the start of an iteration moves that iteration's first load address and the final hash, in - /// every lane (the complement and a single bit are the patterns a linear fold loses). Requires `reg64`. - pub reg64_chain: bool, } /// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of @@ -296,9 +263,6 @@ pub struct EraParams { /// Layer 4, interleave: the four ascending bit positions (0..15) of the word-within-item bits in the word /// index; the first `log2(width_words)` are `0..`, so one aligned load stays inside one item. pub pos: [u8; 4], - /// The index fold of class v6 lane 1 ([`LoadClass::fold`]): the product's low bits folded before the rotation. - /// Not drawn: set from the class the era is composed over. `false` for every era of every other class. - pub fold: bool, } /// Domain tag of the era stream seed. @@ -358,7 +322,7 @@ impl EraParams { pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams { assert!(!allowed.is_empty() && allowed.len() <= 3, "the allowed width set has 1 to 3 entries"); for (i, &w) in allowed.iter().enumerate() { - assert!(WIDTH_WORDS.contains(&w) || w == 8, "allowed width {w} is not 1, 4, 8 or 16 words"); + assert!(WIDTH_WORDS.contains(&w), "allowed width {w} is not 1, 4 or 16 words"); assert!(i == 0 || allowed[i - 1] < w, "the allowed width set is ascending"); } let words = EraParams::stream_words(era_bytes); @@ -394,7 +358,7 @@ pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams { } let mut al = [0u8; 3]; al[..allowed.len()].copy_from_slice(allowed); - EraParams { words, allowed: al, width_words, stride_mul, stride_rot, pos, fold: false } + EraParams { words, allowed: al, width_words, stride_mul, stride_rot, pos } } /// The hot table of a class: `mb` MiB (32, 64 or 96 in the experiment) and `k` hot slots. Two forms: `replaced` @@ -455,13 +419,13 @@ impl LoadClass { impl LoadClass { /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash. pub const V2: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false, wide8: false, fold: false, rw: 0, reg64: false, reg64_chain: false }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false }; /// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): /// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the /// mixer applied 4 times per round, and the cache growth rule. Name "mx4". pub const MX4: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false, wide8: false, fold: false, rw: 0, reg64: false, reg64_chain: false }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false }; /// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed` /// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base @@ -475,14 +439,12 @@ impl LoadClass { c.mix = [0, 0, 0]; c.mix[i] = 100; } else { - let widest = (0..3).rev().find(|&i| c.mix[i] > 0).map(|i| c.widths()[i]).unwrap_or(1); + let widest = (0..3).rev().find(|&i| c.mix[i] > 0).map(|i| WIDTH_WORDS[i]).unwrap_or(1); if widest != e.width_words { // the interleave must keep the widest load inside one item: redraw the positions for that width e = era_draw(era_bytes, &[widest]); } } - // class v6 lane 1: the era's address path carries the class's index fold - e.fold = c.fold; c.era = Some(e); c } @@ -524,16 +486,6 @@ impl LoadClass { LoadClass { mix, load_slots, ..LoadClass::V2 } } - /// The width set this class draws from, in words: [`WIDTH_WORDS`], or `[1, 8, 16]` under `wide8` (W = 8). - pub fn widths(&self) -> [u8; 3] { - if self.wide8 { [1, 8, 16] } else { WIDTH_WORDS } - } - - /// W = 8: every load reads 8 words (a 32-byte sector), `load_slots` loads per program (the class `w32`). - pub fn fixed8(load_slots: u8) -> LoadClass { - LoadClass { mix: [0, 100, 0], load_slots, wide8: true, ..LoadClass::V2 } - } - /// Per-load width drawn from `mix` (percent for 4, 16, 64 bytes), 16 loads per program. pub fn mixed(mix: [u8; 3]) -> LoadClass { assert_eq!(mix.iter().map(|&m| m as u32).sum::(), 100, "the mix must sum to 100"); @@ -601,19 +553,7 @@ impl LoadClass { /// This class with another class's era draw (tests: a rung's class composed with the chain's era). pub fn with_era_of(self, other: &LoadClass) -> LoadClass { - // the era's fold bit follows THIS class's flag, not the other's - LoadClass { era: other.era.map(|mut e| { e.fold = self.fold; e }), ..self } - } - - /// The class with the 64-register window per lane ("mx8+reg64", a research class). - pub fn with_reg64(self) -> LoadClass { - LoadClass { reg64: true, ..self } - } - - /// The reg64 class with the full-chain address mix ("mx8+reg64c"). - pub fn with_reg64_chain(self) -> LoadClass { - assert!(self.reg64, "the full-chain address mix is a reg64 variant"); - LoadClass { reg64_chain: true, ..self } + LoadClass { era: other.era, ..self } } /// The class with the state leaves of class v5 folded into every item ("mx8+sh256x27+state"). @@ -621,29 +561,6 @@ impl LoadClass { LoadClass { state: true, ..self } } - /// Class v6 lane 1: the class with the index fold on every era load address ("...+fold"). An era already - /// composed takes the bit too, so the flag and the era never disagree. - pub fn with_fold(self) -> LoadClass { - LoadClass { fold: true, era: self.era.map(|mut e| { e.fold = true; e }), ..self } - } - - /// Class v6 lane 1: the class drawing its non-load ops from re-weight table `rw` (1: the k lane's split, "+rw"; - /// 2: the census lane's table, "+rw2"; 0: the plain table). - pub fn with_rw(self, rw: u8) -> LoadClass { - assert!(rw <= 2, "re-weight table must be 0, 1 or 2"); - LoadClass { rw, ..self } - } - - /// The non-load op table this class draws from, in draw order, and its sum (the roll's range). The plain table - /// for every class without the re-weight flag, so their streams are byte for byte what they were. - pub fn nonload_weights(&self) -> (&'static [(Op, u64); 10], u64) { - match self.rw { - 1 => (&NONLOAD_WEIGHTS_RW, NONLOAD_WEIGHTS_RW_SUM), - 2 => (&NONLOAD_WEIGHTS_RW2, NONLOAD_WEIGHTS_RW2_SUM), - _ => (&NONLOAD_WEIGHTS, 75), - } - } - /// Shadow instructions per hash (0 without a shadow). pub fn shadow_instrs_per_hash(&self) -> usize { self.shadow.map(|s| s.instrs_per_hash()).unwrap_or(0) @@ -675,25 +592,6 @@ impl LoadClass { /// "mx4": the v3 construction; a trailing "m" and "g" set the mixer multiplier and the growth rule on any /// load class, "w16m4g" for example). pub fn parse(s: &str) -> Option { - // class v6 lane 1: "+fold" (the index fold) and "+rw" / "+rw2" (the re-weight table) over - // any class, in any order, outermost of all - if let Some(base) = s.strip_suffix("+fold") { - return Some(LoadClass::parse(base)?.with_fold()); - } - if let Some(base) = s.strip_suffix("+rw2") { - return Some(LoadClass::parse(base)?.with_rw(2)); - } - if let Some(base) = s.strip_suffix("+rw") { - return Some(LoadClass::parse(base)?.with_rw(1)); - } - // "+reg64c": the 64-register window with the full-chain address mix (outermost, a research class) - if let Some(base) = s.strip_suffix("+reg64c") { - return Some(LoadClass::parse(base)?.with_reg64().with_reg64_chain()); - } - // "+reg64": the 64-register window over any class (the suffix is outermost, a research class) - if let Some(base) = s.strip_suffix("+reg64") { - return Some(LoadClass::parse(base)?.with_reg64()); - } // "+state": the state leaves of class v5 over any class (the suffix is outermost) if let Some(base) = s.strip_suffix("+state") { return Some(LoadClass::parse(base)?.with_state()); @@ -793,7 +691,6 @@ impl LoadClass { "w4" => [100, 0, 0], "w16" => [0, 100, 0], "w64" => [0, 0, 100], - "w32" => return Some(LoadClass::fixed8(slots)), m => { let v: Vec = m.split(',').map(|x| x.trim().parse::().ok()).collect::>>()?; if v.len() != 3 || v.iter().map(|&x| x as u32).sum::() != 100 { @@ -822,18 +719,6 @@ impl LoadClass { /// An era class is the base name with "-era" appended ("w4-era401998a5", "mx4-era..."). /// A hot class appends "hotk[a]" ("hot64k4", "scr4k32+hot64k4a"; measured and not adopted). pub fn name(&self) -> String { - // class v6 lane 1: "+fold" then "+rw" / "+rw2" are the outermost suffixes ("mx8+sh256x27+state+fold+rw") - if self.rw != 0 { - return format!("{}+rw{}", LoadClass { rw: 0, ..*self }.name(), if self.rw == 1 { "" } else { "2" }); - } - if self.fold { - return format!("{}+fold", LoadClass { fold: false, ..*self }.name()); - } - if self.reg64 { - // "+reg64" / "+reg64c": the 64-register window is a suffix on any class, outermost - let base = LoadClass { reg64: false, reg64_chain: false, ..*self }.name(); - return if self.reg64_chain { format!("{base}+reg64c") } else { format!("{base}+reg64") }; - } if self.state { // "+state": class v5's leaves are a suffix on any class, outermost return format!("{}+state", LoadClass { state: false, ..*self }.name()); @@ -878,7 +763,6 @@ impl LoadClass { format!("scr{k}k{}", self.scratch_kb) } else { let base = match self.mix { - [0, 100, 0] if self.wide8 => "w32".to_string(), [100, 0, 0] => "w4".to_string(), [0, 100, 0] => "w16".to_string(), [0, 0, 100] => "w64".to_string(), @@ -907,15 +791,15 @@ impl LoadClass { for (i, &m) in self.mix.iter().enumerate() { acc += m as u64; if roll < acc { - return self.widths()[i]; + return WIDTH_WORDS[i]; } } - self.widths()[2] + WIDTH_WORDS[2] } /// Expected dataset bytes read per hash: dataset loads per hash times the mean width (scratch traffic apart). pub fn expected_bytes_per_hash(&self) -> f64 { - let mean = self.mix.iter().zip(self.widths().iter()).map(|(&m, &w)| m as f64 / 100.0 * w as f64 * 4.0).sum::(); + let mean = self.mix.iter().zip(WIDTH_WORDS.iter()).map(|(&m, &w)| m as f64 / 100.0 * w as f64 * 4.0).sum::(); (self.load_slots as usize - self.scratch_slots()) as f64 * ITERATIONS as f64 * mean } } @@ -1124,7 +1008,7 @@ impl ProgramClass { /// The program class whose load class `class` is, the era draw set aside: [`LoadClass::V2`] is v2, [`V3_CLASS`] /// is v3, [`V4_CLASS`] is v4; a measurement class (a width, a derivation length, another shadow size) is none. pub fn of_load_class(class: &LoadClass) -> Option { - let base = LoadClass { era: None, reg64: false, reg64_chain: false, ..*class }; + let base = LoadClass { era: None, ..*class }; if base == LoadClass::V2 { Some(ProgramClass::V2) } else if base == V3_CLASS { @@ -1148,52 +1032,7 @@ impl ProgramClass { } } -/// Registers per lane of a program: 64 under the reg64 flag, 8 for every other class. -pub const REG64_REGISTERS: usize = 64; -/// The reg64 full-chain fold in the program id: 1 = the load's own source out of the rotate-xor chain. -pub const REG64_CHAIN_FOLD: u8 = 1; - impl Program { - /// Registers per lane (8, or 64 under `class.reg64`). - pub fn registers(&self) -> usize { - if self.class.reg64 { REG64_REGISTERS } else { 8 } - } - /// Whether every load's address consumes all 64 registers (the reg64 full-chain variant). - pub fn address_mix(&self) -> bool { - self.class.reg64 && self.class.reg64_chain - } - /// The pack's liveness statement (the reg64 variants): which reads keep every register necessary. - pub fn reg64_liveness(&self) -> String { - if !self.class.reg64 { - return String::new(); - } - let loads = self.scheduled().iter().filter(|i| i.op == Op::Load).count(); - if self.address_mix() { - format!("every one of the 64 registers is read by the address of each of the {loads} loads per iteration (the load's source directly, the 63 others through the rotate-xor chain) and by the end fold; 64 independently necessary values for the length of the dependent chain; xoring any register with either of two seed-derived probe words at the start of an iteration moves the first load address and the final hash") - } else { - "every one of the 64 registers is read by the end fold (r[k] ^= r[k + 8] ^ ... ^ r[k + 56]); a load's address reads its own window's register only, so the chain needs 8 live values at a time and the other 56 are live across it as state (window, arithmetic-only)".to_string() - } - } - /// The instructions one iteration executes, in order, with their register fields as the kernels name them. - /// Without the reg64 flag this is the drawn program as it stands. Under the flag, instruction i of the drawn - /// program runs on window 0 with every register field widened by `8 * (i % 4)` (so the 64 instructions touch all - /// 32 registers of the window), then the same instruction runs on window 1 (the widened field + 32): 128 - /// statements per iteration over registers 0..63. The CPU verifier and every emitter read this list, so the - /// vectors and the kernel text agree by construction. - pub fn scheduled(&self) -> Vec { - if !self.class.reg64 { - return self.instrs.clone(); - } - let mut out = Vec::with_capacity(self.instrs.len() * 2); - for (i, ins) in self.instrs.iter().enumerate() { - let off = (8 * (i % 4)) as u8; - let a = Instr { dst: ins.dst + off, src: ins.src + off, src2: ins.src2 + off, ..*ins }; - let b = Instr { dst: a.dst + 32, src: a.src + 32, src2: a.src2 + 32, ..a }; - out.push(a); - out.push(b); - } - out - } pub fn loads_per_hash(&self) -> usize { self.instrs.iter().filter(|i| i.op.is_load()).count() * ITERATIONS } @@ -1254,7 +1093,7 @@ impl Program { pub fn width_counts(&self) -> [usize; 3] { let mut c = [0usize; 3]; for i in self.instrs.iter().filter(|i| i.op == Op::Load) { - if let Some(k) = self.class.widths().iter().position(|&w| w == i.width) { + if let Some(k) = WIDTH_WORDS.iter().position(|&w| w == i.width) { c[k] += 1; } } @@ -1298,7 +1137,7 @@ impl Program { let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS; // class v5 at rung 0 is `program_id(5, seed, attempt)`; above rung 0 the class-bearing id with "state/" let v5_rung_0 = self.generator == GENERATOR_VERSION_V5 && LoadClass { era: None, ..self.class } == V5_CLASS; - if !self.class.reg64 && (self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0) { + if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0 { // Spec 01 section 1.4.6: a class v3 program's id is `program_id(3, seed, attempt)`, a class v4 program's // `program_id(4, seed, attempt)` (Counter ASIC 3.0); the generator version in the preimage separates // them from every version 2 program of the same seed @@ -1308,20 +1147,6 @@ impl Program { } } - /// The derivation of [`Program::program_id`] as text, from the same byte recipe (program.json's - /// "program_id_derivation"; spec 01 section 1.4.6). - pub fn program_id_derivation(&self) -> String { - let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS; - // class v5 at rung 0 takes the plain form as `program_id` does (the text follows the id; the first form of this - // function named the class recipe for the v5 packs whose id was the plain one) - let v5_rung_0 = self.generator == GENERATOR_VERSION_V5 && LoadClass { era: None, ..self.class } == V5_CLASS; - if !self.class.reg64 && (self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0) { - program_id_recipe(self.generator, &self.seed, self.attempt).text() - } else { - program_id_class_recipe(self.generator, &self.seed, self.attempt, &self.class).text() - } - } - /// The program class of this program, from its generator version (3 = v3, 4 = v4, everything else v2). pub fn program_class(&self) -> ProgramClass { match self.generator { @@ -1333,61 +1158,24 @@ impl Program { } } -/// The byte recipe of a program id and the text that states it: the bytes FNV-1a 64 hashes and, part by part, the label -/// of each part (`'literal'` or `field_le32`), so the derivation printed in program.json is built from the same list the id -/// is hashed from and the two cannot drift (the documentation finding of 7 October 2026: program.json and spec 1.4.6 said -/// the plain form while generator 4 appended `'sub/' || sub_version_le16`, so a client written from the text derived another -/// id). `tests/derivation.rs` re-derives every pinned pack's id from its own derivation string. -#[derive(Clone, Debug, Default)] -pub struct IdRecipe { - pub bytes: Vec, - parts: Vec, -} - -impl IdRecipe { - fn lit(&mut self, s: &[u8]) { - self.bytes.extend_from_slice(s); - self.parts.push(format!("'{}'", String::from_utf8_lossy(s))); - } - fn field(&mut self, bytes: &[u8], label: &str) { - self.bytes.extend_from_slice(bytes); - self.parts.push(label.to_string()); - } - /// The derivation as program.json prints it: `FNV-1a 64 over || || ...`. - pub fn text(&self) -> String { - format!("FNV-1a 64 over {}", self.parts.join(" || ")) - } - /// The id: FNV-1a 64 over the bytes. - pub fn id(&self) -> u64 { - fnv1a64(&self.bytes) - } -} - -/// The recipe of [`program_id`]. -pub fn program_id_recipe(generator: u32, seed: &[u32; 8], attempt: u32) -> IdRecipe { - let mut r = IdRecipe::default(); - r.lit(PROGRAM_ID_TAG); - r.field(&generator.to_le_bytes(), "generator_le32"); - let mut sw = Vec::with_capacity(32); +pub fn program_id(generator: u32, seed: &[u32; 8], attempt: u32) -> u64 { + let mut b = Vec::with_capacity(PROGRAM_ID_TAG.len() + 4 + 32 + 4 + 6); + b.extend_from_slice(PROGRAM_ID_TAG); + b.extend_from_slice(&generator.to_le_bytes()); for w in seed { - sw.extend_from_slice(&w.to_le_bytes()); + b.extend_from_slice(&w.to_le_bytes()); } - r.field(&sw, "seed_words as little-endian bytes"); - r.field(&attempt.to_le_bytes(), "attempt_le32"); + b.extend_from_slice(&attempt.to_le_bytes()); if generator == GENERATOR_VERSION_V4 { // The class v4 sub-version (AP-F8-1 amendment, 7 October 2026): `"sub/" || sub_version as little-endian u16` // appended for generator 4 only, so a binary from before the load-source rule (sub-version 0, no suffix) and // one after it never share a program id for one seed; the node's id check then catches a split. v2 and v3 // ids are byte-identical. The node reads the sub-version from [`PROGRAM_SUBVERSION_V4`]; packs carry it as // IGNEUM_PROGRAM_SUBVERSION and program.json "sub_version". - r.lit(b"sub/"); - r.field(&PROGRAM_SUBVERSION_V4.to_le_bytes(), "sub_version_le16"); + b.extend_from_slice(b"sub/"); + b.extend_from_slice(&PROGRAM_SUBVERSION_V4.to_le_bytes()); } - r -} - -pub fn program_id(generator: u32, seed: &[u32; 8], attempt: u32) -> u64 { - program_id_recipe(generator, seed, attempt).id() + fnv1a64(&b) } /// The sub-version of class v4's program stream, in every generator-4 program id and pack (AP-F8-3: 3 = the @@ -1401,84 +1189,57 @@ pub const PROGRAM_ID_TAG_RW: &[u8] = b"igneum-program-rw/"; /// The program id of a non-default class: the tag, then the same fields as [`program_id`], then the three mix /// percentages and the slot count as bytes. -/// The recipe of [`program_id_class`]. -pub fn program_id_class_recipe(generator: u32, seed: &[u32; 8], attempt: u32, class: &LoadClass) -> IdRecipe { - let mut r = IdRecipe::default(); - r.lit(PROGRAM_ID_TAG_RW); - r.field(&generator.to_le_bytes(), "generator_le32"); - let mut sw = Vec::with_capacity(32); +pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &LoadClass) -> u64 { + let mut b = Vec::with_capacity(PROGRAM_ID_TAG_RW.len() + 4 + 32 + 4 + 4); + b.extend_from_slice(PROGRAM_ID_TAG_RW); + b.extend_from_slice(&generator.to_le_bytes()); for w in seed { - sw.extend_from_slice(&w.to_le_bytes()); + b.extend_from_slice(&w.to_le_bytes()); } - r.field(&sw, "seed_words as little-endian bytes"); - r.field(&attempt.to_le_bytes(), "attempt_le32"); - r.field(&class.mix, "mix[3]"); - r.field(&[class.load_slots], "load_slots_u8"); + b.extend_from_slice(&attempt.to_le_bytes()); + b.extend_from_slice(&class.mix); + b.push(class.load_slots); if let Some(k) = class.scratch { - r.lit(b"scratch/"); - r.field(&[k], "scratch_k_u8"); - r.field(&[class.scratch_kb], "scratch_kb_u8"); + b.extend_from_slice(b"scratch/"); + b.push(k); + b.push(class.scratch_kb); } if class.mixer_mult != 1 || class.growth { // Counter ASIC 2.0: the mixer multiplier and the growth rule are part of the construction, so a program of // the same seed under a different mixer carries a different id (under the v3 seam the id is // program_id(3, seed, attempt) and this branch is not taken) - r.lit(b"mixer/"); - r.field(&[class.mixer_mult], "mixer_mult_u8"); - r.field(&[class.growth as u8], "growth_u8"); + b.extend_from_slice(b"mixer/"); + b.push(class.mixer_mult); + b.push(class.growth as u8); } if class.derive_len != 0 { // Counter ASIC 3.0 item 2: the derivation program's length is part of the construction - r.lit(b"derive/"); - r.field(&class.derive_len.to_le_bytes(), "derive_len_le16"); + b.extend_from_slice(b"derive/"); + b.extend_from_slice(&class.derive_len.to_le_bytes()); } if let Some(e) = class.era { - r.lit(b"era/"); - r.field(&e.id_bytes(), "era_id_bytes (allowed[3] || width_words_u8 || stride_mul_le32 || stride_rot_le32 || interleave[4])"); + b.extend_from_slice(b"era/"); + b.extend_from_slice(&e.id_bytes()); } if let Some(sh) = class.shadow { // Counter ASIC 3.0 item 8: the shadow block's size and repeat count are part of the construction - r.lit(b"shadow/"); - r.field(&sh.instrs.to_le_bytes(), "shadow_instrs_le16"); - r.field(&sh.reps.to_le_bytes(), "shadow_reps_le16"); + b.extend_from_slice(b"shadow/"); + b.extend_from_slice(&sh.instrs.to_le_bytes()); + b.extend_from_slice(&sh.reps.to_le_bytes()); } if let Some(h) = class.hot { - r.lit(b"hot/"); - r.field(&[h.mb], "hot_mb_u8"); - r.field(&[h.k], "hot_k_u8"); + b.extend_from_slice(b"hot/"); + b.push(h.mb); + b.push(h.k); if h.added { - r.lit(b"added"); + b.extend_from_slice(b"added"); } } if class.state { // class v5: the state leaves are part of the construction - r.lit(b"state/"); + b.extend_from_slice(b"state/"); } - if class.fold { - // class v6 lane 1: the index fold moves every era load address - r.lit(b"fold/"); - } - if class.rw != 0 { - // class v6 lane 1: the re-weight table moves the op draw - r.lit(b"rw/"); - r.field(&[class.rw], "rw_table_u8"); - } - if class.reg64 { - // the 64-register window is part of the construction - r.lit(b"reg64/"); - if class.reg64_chain { - // the fold text is part of the construction: 1 = the load's own source out of the chain (the sound - // fold); the benched text of 8 October 2026 (the source inside the chain, id 0x3deee2320e70e1bf for the - // pinned devnet seeds) carried no byte here, so the two texts never share an id - r.lit(b"chain/"); - r.field(&[REG64_CHAIN_FOLD], "reg64_chain_fold_u8"); - } - } - r -} - -pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &LoadClass) -> u64 { - program_id_class_recipe(generator, seed, attempt, class).id() + fnv1a64(&b) } /// Weights of the ten non-load families under version 2, in draw order. Sum 75. The load family has no @@ -1496,38 +1257,6 @@ pub const NONLOAD_WEIGHTS: [(Op, u64); 10] = [ (Op::Or, 4), ]; -/// Class v6 lane 1, the re-weight table behind the index fold (`+rw`): the k lane's optimiser split in draw order, -/// sum [`NONLOAD_WEIGHTS_RW_SUM`] (83). Shuffle stays at 4 (a shuffle-heavy draw is the worst thing the class can do -/// on the chip side, k 0.021 routed) and `or` is never drawn. The roll ranges over the table's own sum. -pub const NONLOAD_WEIGHTS_RW: [(Op, u64); 10] = [ - (Op::Add, 16), - (Op::Xor, 14), - (Op::Mul, 4), - (Op::Mad, 12), - (Op::Shfl, 4), - (Op::Rotl, 11), - (Op::Sub, 10), - (Op::MulHi, 2), - (Op::Rotr, 10), - (Op::Or, 0), -]; -pub const NONLOAD_WEIGHTS_RW_SUM: u64 = 83; - -/// Class v6 lane 1, the census lane's neighbouring table (`+rw2`), sum [`NONLOAD_WEIGHTS_RW2_SUM`] (75). -pub const NONLOAD_WEIGHTS_RW2: [(Op, u64); 10] = [ - (Op::Add, 13), - (Op::Xor, 11), - (Op::Mul, 6), - (Op::Mad, 10), - (Op::Shfl, 8), - (Op::Rotl, 8), - (Op::Sub, 7), - (Op::MulHi, 2), - (Op::Rotr, 6), - (Op::Or, 4), -]; -pub const NONLOAD_WEIGHTS_RW2_SUM: u64 = 75; - /// Version 1 weights (retired). Sum 100, load at 25 percent. pub const OP_WEIGHTS: [(Op, u64); 11] = [ (Op::Load, 25), @@ -1649,12 +1378,8 @@ pub fn candidate_from_words_class( // era set aside) on EVERY draw path, era or not, so a census through candidate_class reads the same stream as // the chain; v2, v3 and every other class take no part. The draw order and the stream are otherwise the same. // class v5 (docs/design/class-v5-stored-state.md) draws under the same rule: its state flag is set aside here too - // class v6 lane 1: the index fold and the re-weight table are set aside too (the address path and the table are not - // the shape; a +fold or +rw program draws its sources under the same rule) let source_rule_v4 = matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. })) - && LoadClass { era: None, shadow: None, state: false, fold: false, rw: 0, ..class } == LoadClass { shadow: None, ..V4_CLASS }; - // the op table and the roll's range: the plain table at 75 for every class without the re-weight flag - let (weights, weights_sum) = class.nonload_weights(); + && LoadClass { era: None, shadow: None, state: false, ..class } == LoadClass { shadow: None, ..V4_CLASS }; let mut fresh = [false; 8]; let mut fresh_value = [true; 8]; // the shared-operand idiom (AP-F8-1, sub-version 3): after `or d |= s`, a later `xor d ^= s` or `sub d -= s` with @@ -1664,9 +1389,9 @@ pub fn candidate_from_words_class( let mut pair_op: [Option<(Op, usize)>; 8] = [None; 8]; let mut instrs = Vec::with_capacity(INSTR_COUNT); for k in 0..INSTR_COUNT { - let mut roll = rng.below(weights_sum); + let mut roll = rng.below(75); let mut op = Op::Add; - for &(o, w) in weights { + for &(o, w) in &NONLOAD_WEIGHTS { if roll < w { op = o; break; @@ -1771,9 +1496,9 @@ pub fn candidate_from_words_class( loop { shadow.clear(); for _ in 0..sh.instrs { - let mut roll = rng.below(weights_sum); + let mut roll = rng.below(75); let mut op = Op::Add; - for &(o, w) in weights { + for &(o, w) in &NONLOAD_WEIGHTS { if roll < w { op = o; break; @@ -1864,12 +1589,7 @@ pub fn try_generate_class(seed_string: &str, seed_bytes: &[u8], class: LoadClass } if crate::accept::is_class_v4_shape(&class) { // AP-F8-2 (7 October 2026, main's ruling: the draw is total and no consensus path panics): a class v4 seed that - // exhausts its attempts takes the last-resort program, deterministic. Sub-version 3's is accepted as drawn - // (unreachable at 4.6e-44 per epoch and unverified against the rule: adv-accept-3's finding, 223 of 2,500 - // rewritten candidates fail it, 209 by part (a)); class v5's is the verified one. - if class.state { - return Ok(last_resort_v5(seed_string, seed_bytes, cap, class)); - } + // exhausts its attempts takes the last-resort program, deterministic and accepted as drawn return Ok(last_resort_v4(candidate_class(seed_string, seed_bytes, cap, class))); } Err(Exhausted { seed_string: seed_string.to_string(), attempts: cap, last: last.unwrap() }) @@ -1893,55 +1613,6 @@ pub fn last_resort_v4(mut p: Program) -> Program { p } -/// Class v5's last resort, verified (adv-accept-3's finding of 7 October 2026: the sub-version 3 rewrite alone fails the -/// rule on 223 of 2,500 seeds, 209 by part (a), a load reading a register no instruction wrote since the previous load -/// from it). From attempt `cap` on, each candidate is rewritten as [`last_resort_v4`] does, its stale loads re-sourced -/// by [`repair_stale_loads`], and the result checked against the whole rule; the first that passes is the program. The -/// scan runs [`LAST_RESORT_SCAN`] candidates; past it the first repaired candidate stands as drawn, so the draw is total, -/// and that fallback sits behind the 4.6e-44 of reaching the last resort at all times the rejection rate of a repaired -/// candidate (about 0.09 per adv-accept-3) to the power of the scan: under 1e-300. -pub fn last_resort_v5(seed_string: &str, seed_bytes: &[u8], cap: u32, class: LoadClass) -> Program { - for k in cap..cap + LAST_RESORT_SCAN { - let p = repair_stale_loads(last_resort_v4(candidate_class(seed_string, seed_bytes, k, class))); - if check(&p).is_ok() { - return p; - } - } - repair_stale_loads(last_resort_v4(candidate_class(seed_string, seed_bytes, cap, class))) -} - -/// The candidates class v5's last resort scans past the attempt cap before the unchecked fallback. -pub const LAST_RESORT_SCAN: u32 = 256; - -/// Part (a) by construction: a load whose source register no instruction wrote since the previous load from it (in -/// cyclic order over the 64 instructions, the two-pass walk of the acceptance's `check_stale_loads`) is re-sourced to -/// the lowest register that was written since its last load. The walk repeats until a full two-pass walk changes -/// nothing (a re-sourced load can make a later load from the new register stale, which the next walk re-sources); -/// with 16 loads among 64 instructions a non-pending register always exists. Only the `src` field moves. -pub fn repair_stale_loads(mut p: Program) -> Program { - for _round in 0..16 { - let mut changed = false; - let mut pending = [false; 8]; - for _pass in 0..2 { - for ins in p.instrs.iter_mut() { - if ins.op.is_load() && pending[ins.src as usize] { - let q = (0..8usize).find(|&q| !pending[q]).expect("a register written since its last load") as u8; - ins.src = q; - changed = true; - } - pending[ins.dst as usize] = false; - if ins.op.is_load() { - pending[ins.src as usize] = true; - } - } - } - if !changed { - break; - } - } - p -} - /// [`try_generate_from_seed_bytes`], treating exhaustion as the consensus fault it is. pub fn generate_from_seed_bytes(seed_string: &str, seed_bytes: &[u8]) -> Program { try_generate_from_seed_bytes(seed_string, seed_bytes).unwrap_or_else(|e| panic!("{e}")) @@ -2227,14 +1898,6 @@ mod tests { assert!(v2.instrs.iter().all(|i| i.width == 1)); assert_eq!(v2.bytes_per_hash(), 512); assert_eq!(LoadClass::parse("w16"), Some(LoadClass::fixed(4, 16))); - // W = 8 (8 October 2026): the w32 class reads 8 words a load; w16 reads 4 (known-failed first: the plain set has no 8) - assert!(!WIDTH_WORDS.contains(&8)); - let w32 = LoadClass::parse("w32").unwrap(); - assert!(w32.wide8 && w32.mix == [0, 100, 0] && w32.widths() == [1, 8, 16]); - assert_eq!(w32.name(), "w32"); - assert_eq!(LoadClass::parse("w32x8").unwrap().load_slots, 8); - assert_eq!(w32.expected_bytes_per_hash(), 2.0 * LoadClass::fixed(4, 16).expected_bytes_per_hash()); - assert!(!LoadClass::fixed(4, 16).wide8); assert_eq!(LoadClass::parse("w64x4"), Some(LoadClass::fixed(16, 4))); assert_eq!(LoadClass::parse("50,35,15"), Some(LoadClass::mixed([50, 35, 15]))); assert_eq!(LoadClass::parse("v2"), Some(LoadClass::V2)); @@ -2610,44 +2273,6 @@ mod tests { } } - /// Class v5's last resort is verified (adv-accept-3's finding, 7 October 2026): on seed adv3/steer/2 of that lane's - /// label space (`seed_words_from_bytes("igneum-adv-accept-3/steer/2")` as the epoch bytes, Devnet 3's genesis as the - /// era) the sub-version 3 rewrite of the candidate at the cap fails the rule by part (a), a cyclic stale load, and - /// would be handed to the chain; class v5's repair re-sources the load and the result passes the whole rule, as - /// does the program `last_resort_v5` returns. The class v4 path is unchanged (frozen, unreachable, recorded). - #[test] - fn class_v5_last_resort_is_verified_known_failed_adv3_steer_2() { - use crate::accept::{check, check_static, Reject}; - use crate::seed::seed_words_from_bytes; - let epoch: Vec = seed_words_from_bytes(b"igneum-adv-accept-3/steer/2").iter().flat_map(|x| x.to_le_bytes()).collect(); - let era = crate::bind::unhex("4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925").unwrap(); - let v4 = LoadClass::era(V4_CLASS, &era, &V3_ALLOWED); - let v5 = LoadClass::era(V5_CLASS, &era, &V3_ALLOWED); - let cap = max_attempts_for(&v4); - let old = last_resort_v4(candidate_class("adv3/steer/2", &epoch, cap, v4)); - match check_static(&old) { - Err(Reject::StaleLoadSource { instr, reg }) => println!("adv3/steer/2: the sub-version 3 last resort fails part (a) at instruction {instr} reading r{reg}"), - other => panic!("the known-failed case must fail part (a): {other:?}"), - } - let repaired = repair_stale_loads(old.clone()); - assert!(check_static(&repaired).is_ok(), "the repair restores part (a): {:?}", check_static(&repaired).err()); - assert_eq!(repaired.instrs.len(), old.instrs.len()); - assert!(repaired.instrs.iter().zip(old.instrs.iter()).all(|(a, b)| a.op == b.op && a.dst == b.dst), "only load sources move"); - let lr = last_resort_v5("adv3/steer/2", &epoch, cap, v5); - assert!(check(&lr).is_ok(), "class v5's last resort passes the whole rule: {:?}", check(&lr).err()); - assert!(lr.attempt >= cap && lr.attempt < cap + LAST_RESORT_SCAN); - assert!(lr.class.state); - println!("adv3/steer/2: class v5 last resort at attempt {} id {:016x}", lr.attempt, lr.program_id()); - // the sub-version 3 path is byte for byte what it was - assert_eq!(try_generate_class("adv3/steer/2", &epoch, v4).map(|p| p.attempt).ok(), Some(try_generate_class("adv3/steer/2", &epoch, v4).unwrap().attempt)); - // a few more of the label space: every class v5 last resort passes - for i in [11u32, 33, 56, 58, 77] { - let e: Vec = seed_words_from_bytes(format!("igneum-adv-accept-3/steer/{i}").as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect(); - let lr = last_resort_v5(&format!("adv3/steer/{i}"), &e, cap, v5); - assert!(check(&lr).is_ok(), "steer/{i}: {:?}", check(&lr).err()); - } - } - #[test] fn class_v4_draw_is_total_with_the_last_resort() { assert_eq!(max_attempts_for(&V4_CLASS), MAX_ATTEMPTS_V4); diff --git a/igneum-pow/src/main.rs b/igneum-pow/src/main.rs index 393b1eb81..95795cd5d 100644 --- a/igneum-pow/src/main.rs +++ b/igneum-pow/src/main.rs @@ -1,7 +1,7 @@ //! igneum-pow CLI. //! //! igneum-pow bench --seed [--day ] [--closed-form] [--dataset-log2 28] [--warps 20] -//! igneum-pow export --seed --out [--day ] [--closed-form] [--dataset-log2 28 | --dataset-words N] [--epoch-hex <64 hex> --day-hex ] +//! igneum-pow export --seed --out [--day ] [--closed-form] [--dataset-log2 28] [--epoch-hex <64 hex> --day-hex ] //! igneum-pow hash --seed --nonce [--day ] [--closed-form] [--dataset-log2 28] //! igneum-pow hash-bound --seed --prehash <64 hex> --nonce [--day ] [--closed-form] [--dataset-log2 28] //! [--epoch-hex <64 hex> --day-hex ] byte seeds instead of strings (Epoch::from_seed_bytes) @@ -22,7 +22,7 @@ use igneum_pow::emit::export_pack; use igneum_pow::generator::{LoadClass, ProgramClass}; use igneum_pow::memhard::{Cache, Shape}; use igneum_pow::seed::day_key; -use igneum_pow::verify::{DatasetGeom, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2}; +use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2}; use std::time::Instant; struct Args { @@ -32,10 +32,6 @@ struct Args { out: Option, closed_form: bool, dataset_log2: u32, - /// `--dataset-words N`: the dataset at N words (research class ds55, 8 October 2026): any multiple of 2^16 in - /// 2^28 ..= 2^31; a non-power-of-two takes the multiply-shift range reduction of spec 01 section 1.13.3 in every - /// load, a power of two is the `--dataset-log2` path. Applied after the class's own sizing. - dataset_words: Option, warps: usize, nonce: u64, /// `hash-bound --count N`: N consecutive nonces from --nonce, one epoch build (gate G2, 5 October 2026). @@ -60,12 +56,6 @@ struct Args { /// generator 3 and the era bytes recorded (the measurement packs: v2's mixer under the era layout). era: Option<(u64, Vec, String)>, era_widths: Vec, - /// `--reg64`: the 64-register window per lane over the chosen class (the hash lane's measurement, 8 October - /// 2026, a research class; also `--class +reg64`). The draw is the class's; the flag is stamped on the - /// program after it, as the era is. - reg64: bool, - /// `--reg64-chain`: the full-chain address mix over the reg64 window (`+reg64c`). - reg64_chain: bool, } /// `igneum-era-test/` or `:<64 hex>` -> (index, 32 era bytes, label). @@ -90,7 +80,6 @@ fn parse_widths(s: &str) -> Option> { .map(|x| match x.trim() { "4" => Some(1u8), "16" => Some(4), - "32" => Some(8), // W = 8, the 32-byte sector (the w32 class, 8 October 2026) "64" => Some(16), _ => None, }) @@ -119,10 +108,7 @@ fn usage() -> ! { \x20 --state class v5 (or any --class ...+state): the window's state stream (IGSD1 file, igneum-day-stream --out), whose leaves key every item\n\ \x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\ \x20 --era E era layout over --class: igneum-era-test/ or :<64 hex> (the 32-byte era seed E_n)\n\ - \x20 --era-widths 4[,16,32,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it; 32 only with the w32 class)\n\ - \x20 --dataset-words N research class ds55: the dataset at N words (a multiple of 65,536 in 2^28 ..= 2^31; 1476395008 = 5.5 GiB); a non-power-of-two uses idx = (src * N) >> 32 in every load (spec 01 section 1.13.3), a power of two is --dataset-log2\n\\ - \x20 --reg64 the 64-register window per lane over the class (research; the same as --class +reg64; CUDA, OpenCL and Metal texts)\n\ - \x20 --reg64-chain reg64 with the full-chain address mix: every load's address consumes all 64 registers (the same as --class +reg64c)" + \x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)" ); std::process::exit(2) } @@ -136,7 +122,6 @@ fn parse() -> Args { closed_form: false, state: None, dataset_log2: DEFAULT_DATASET_LOG2, - dataset_words: None, warps: 20, nonce: 0, count: 1, @@ -150,8 +135,6 @@ fn parse() -> Args { shadow_reps: 0, era: None, era_widths: vec![1], - reg64: false, - reg64_chain: false, }; let mut it = std::env::args().skip(1); a.cmd = it.next().unwrap_or_else(|| usage()); @@ -163,13 +146,6 @@ fn parse() -> Args { "--out" => a.out = Some(val()), "--closed-form" => a.closed_form = true, "--dataset-log2" => a.dataset_log2 = val().parse().unwrap_or_else(|_| usage()), - "--dataset-words" => { - let n: u64 = val().parse().unwrap_or_else(|_| usage()); - a.dataset_words = Some(DatasetGeom::words(n).unwrap_or_else(|err| { - eprintln!("{err}"); - std::process::exit(2) - })); - } "--warps" => a.warps = val().parse().unwrap_or_else(|_| usage()), "--nonce" => a.nonce = val().parse().unwrap_or_else(|_| usage()), "--count" => a.count = val().parse().unwrap_or_else(|_| usage()), @@ -184,11 +160,6 @@ fn parse() -> Args { "--shadow-reps" => a.shadow_reps = val().parse().unwrap_or_else(|_| usage()), "--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())), "--era-widths" => a.era_widths = parse_widths(&val()).unwrap_or_else(|| usage()), - "--reg64" => a.reg64 = true, - "--reg64-chain" => { - a.reg64 = true; - a.reg64_chain = true; - } _ => usage(), } } @@ -251,22 +222,6 @@ fn main() { fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) { let (mut e, label) = epoch_of_class(a, mode); stamp_era(&mut e, a); - // research class ds55: the dataset at the word count of --dataset-words, the cache and the items unchanged; a - // state class sizes its leaves by the power-of-two count and is refused here - if let Some(geom) = a.dataset_words { - if e.program.class.state { - eprintln!("--dataset-words is not supported with a state class ({}): the leaves are sized by --dataset-log2", e.program.class.name()); - std::process::exit(2); - } - e.dataset = e.dataset.with_geom(geom); - } - if a.reg64 { - // the 64-register window over the drawn program (the same text in CUDA, OpenCL and Metal since class-v6) - e.program.class = e.program.class.with_reg64(); - if a.reg64_chain { - e.program.class = e.program.class.with_reg64_chain(); - } - } // class v5: the leaves of --state, built for the dataset's size; a state class without --state is refused here // rather than at the first derivation if e.program.class.state { @@ -426,10 +381,10 @@ fn export(a: &Args, mode: DatasetMode) { let build_ms = t0.elapsed().as_secs_f64() * 1e3; println!("igneum-pow export {out}"); println!( - "seed \"{}\", day \"{}\", dataset {} ({}), generator v{} attempt {} program id {:016x}, loads/hash {}; epoch built in {build_ms:.1} ms", + "seed \"{}\", day \"{}\", dataset 2^{} words ({}), generator v{} attempt {} program id {:016x}, loads/hash {}; epoch built in {build_ms:.1} ms", e.program.seed_string, day_label, - e.dataset.geom.describe(), + e.dataset.log2_words, e.dataset.mode().name(), e.program.generator, e.program.attempt, @@ -437,10 +392,6 @@ fn export(a: &Args, mode: DatasetMode) { e.program.loads_per_hash() ); println!("op mix: {}; class {}, {} bytes/hash, widths (1,4,16 words) {:?}", e.program.op_mix(), e.program.class.name(), e.program.bytes_per_hash(), e.program.width_counts()); - println!("seed words {}", e.program.seed.iter().map(|w| format!("0x{w:08x}")).collect::>().join(" ")); - if e.dataset.geom.mulshift { - println!("dataset mapping: multiply-shift, idx = (src * {}) >> 32; {} items, {} bytes (research class ds55)", e.dataset.geom.words, e.dataset.geom.items(), e.dataset.geom.bytes()); - } let source = format!("igneum-pow (Rust) CPU interpreter, generator v{}, {} dataset", e.program.generator, e.dataset.mode().name()); let pack = export_pack(&e, &day_label, &source); let dir = std::path::Path::new(&out); diff --git a/igneum-pow/src/memhard.rs b/igneum-pow/src/memhard.rs index 644896ffb..63a35ee50 100644 --- a/igneum-pow/src/memhard.rs +++ b/igneum-pow/src/memhard.rs @@ -210,15 +210,12 @@ pub struct MixParams { pub redraws: u32, } -/// Class v5's mixer-draw rule (AP-F4-1, the form the attack-pass lane and adv-mixer-2 agreed on 7 October 2026, 22:0x -/// UK; it replaces the first form's "NAF sum under 163"): the adder-datapath cost of a mixer block is -/// `A = 64 + sum over the 16 multipliers of (w32(m) - 1)`, `w32` the NAF weight over bit positions 0 to 31 only (the -/// carry digit at position 32 a 32-bit multiplier never pays; the first census counted it, so its median read 231 -/// where the agreed median is 226). A block whose cost is at most this is rejected (a 1.1x gain against the median). -pub const MIXER_COST_REJECT_MAX: u32 = 205; -/// Class v5's mixer-draw rule: every multiplier's `w32` at least this (a multiplier with a two-adder chain, `w32 <= 3`, -/// is rejected on its own: adv-mixer-2's `k >= 1`). +/// Class v5's mixer-draw rule (AP-F4-1): the NAF sum of the 16 multipliers at least this. +pub const MIXER_NAF_SUM_MIN: u32 = 163; +/// Class v5's mixer-draw rule: every multiplier's NAF weight at least this. pub const MIXER_NAF_WORD_MIN: u32 = 4; +/// Class v5's mixer-draw rule: at least this many distinct rotation amounts among the eight. +pub const MIXER_DISTINCT_ROT_MIN: usize = 4; /// Class v5's mixer-draw rule: redraws before the last block stands as drawn (never reached at 6.1e-4 per try). pub const MIXER_REDRAW_CAP: u32 = 64; @@ -241,40 +238,14 @@ pub fn naf_weight(mut x: u64) -> u32 { w } -/// The NAF weight of a 32-bit multiplier over bit positions 0 to 31: [`naf_weight`] without the digit at position 32. -pub fn naf32_weight(m: u32) -> u32 { - let mut x = m as u64; - let mut w = 0; - let mut pos = 0; - while x != 0 { - if x & 1 == 1 { - if pos < 32 { - w += 1; - } - if x & 3 == 3 { - x += 1; - } else { - x -= 1; - } - } - x >>= 1; - pos += 1; - } - w -} - -/// The adder-datapath cost of a mixer block: `64 + sum(w32(m) - 1)` over the 16 multipliers. -pub fn mixer_cost(mul: &[u32; 16]) -> u32 { - 64 + mul.iter().map(|&m| naf32_weight(m).saturating_sub(1)).sum::() -} - -/// Whether a mixer block passes class v5's draw rule (AP-F4-1): cost over [`MIXER_COST_REJECT_MAX`], every -/// multiplier's `w32` at least [`MIXER_NAF_WORD_MIN`], and the eight rotation amounts not all equal. A rejected block -/// is redrawn whole (all forty draws) from the continuing stream. +/// Whether a mixer block passes class v5's draw rule (AP-F4-1). pub fn mixer_block_admissible(rot: &[u32; 8], mul: &[u32; 16]) -> bool { - let words = mul.iter().all(|&m| naf32_weight(m) >= MIXER_NAF_WORD_MIN); - let rot_equal = rot.iter().all(|&r| r == rot[0]); - mixer_cost(mul) > MIXER_COST_REJECT_MAX && words && !rot_equal + let sum: u32 = mul.iter().map(|&m| naf_weight(m as u64)).sum(); + let words = mul.iter().all(|&m| naf_weight(m as u64) >= MIXER_NAF_WORD_MIN); + let mut distinct = rot.to_vec(); + distinct.sort_unstable(); + distinct.dedup(); + sum >= MIXER_NAF_SUM_MIN && words && distinct.len() >= MIXER_DISTINCT_ROT_MIN } impl MixParams { @@ -298,11 +269,11 @@ impl MixParams { } let mut redraws = 0u32; if shape.state { - // Class v5 (docs/design/class-v5-stored-state.md section 11, AP-F4-1, the attack-pass lane's and adv-mixer-2's - // reconciled weak-day census): a mixer block whose adder cost is at most 205 against the median 226, or with a - // two-adder multiplier, or with one rotation amount, is redrawn whole from the next forty stream values, so no - // day is a weak day for a per-day LUT-recompute FPGA (the worst calendar day, chain day 29,337 = 2050-04-28, - // read 1.113x). About 5.69e-4 of days redraw, 15 days a century. The derive program's draws come after. + // Class v5 (docs/design/class-v5-stored-state.md section 11, AP-F4-1, the attack-pass lane's weak-day census): + // a mixer block whose multipliers are cheap on an adder datapath (NAF sum under 163, a word under NAF weight 4) + // or whose rotations repeat (under 4 distinct amounts) is redrawn from the next stream values, so no day is a + // weak day for a per-day LUT-recompute FPGA (the worst calendar day of the census, chain day 29,337, was 1.121x). + // About 6.1e-4 of days redraw. The derive program's draws (none under v5) come after, as before. while !mixer_block_admissible(&rot, &mul) && redraws < MIXER_REDRAW_CAP { for r in rot.iter_mut() { *r = 1 + rng.below(31) as u32; @@ -847,11 +818,9 @@ impl MemhardCpu { mod tests { use super::*; - /// Class v5's mixer-draw rule (AP-F4-1 in the agreed form), the known-failed case first: the day the two censuses - /// name as the worst of the century, chain day 29,337 (2050-04-28, 1.113x), draws a block the rule rejects and class - /// v5 redraws it; a block of two-adder multipliers or one rotation amount is inadmissible; a scan of day keys finds - /// the redrawn days (about 5.69e-4), every v5 block passes after the draw, and the v4 constants of the same keys - /// never move. + /// Class v5's mixer-draw rule (AP-F4-1), the known-failed case first: a block of cheap multipliers (NAF sum under + /// 163) or repeated rotations is inadmissible; a scan of day keys finds days the rule redraws (the census's 6.1e-4), + /// every v5 block passes after the draw, and the v4 constants of the same keys never move. #[test] fn class_v5_mixer_draw_rule() { assert_eq!(naf_weight(0), 0); @@ -859,34 +828,16 @@ mod tests { assert_eq!(naf_weight(3), 2, "11 = 100 - 1"); assert_eq!(naf_weight(7), 2, "111 = 1000 - 1"); assert_eq!(naf_weight(0xffff_ffff), 2); - assert_eq!(naf32_weight(0xffff_ffff), 1, "the carry digit at position 32 is not paid"); - assert_eq!(naf32_weight(0xc000_0001), 2, "2^32 - 2^30 + 1: the digit at 32 dropped, -2^30 and +1 kept"); assert_eq!(naf_weight(0b1010_1010), 4); - assert_eq!(naf32_weight(0b1010_1010), 4); let good_rot = [1u32, 5, 9, 13, 17, 21, 25, 29]; let cheap = [0x8000_0001u32; 16]; - assert_eq!(mixer_cost(&cheap), 64 + 16, "16 two-adder multipliers"); assert!(!mixer_block_admissible(&good_rot, &cheap), "the known-failed case: 16 two-adder multipliers"); let dense = [0xaaaa_aaabu32; 16]; - assert!(mixer_cost(&dense) > MIXER_COST_REJECT_MAX); assert!(mixer_block_admissible(&good_rot, &dense)); assert!(!mixer_block_admissible(&[7u32; 8], &dense), "one rotation amount"); - assert!(mixer_block_admissible(&[1u32, 2, 3, 3, 3, 3, 3, 3], &dense), "three distinct amounts pass the agreed form"); - // one two-adder multiplier among dense ones is rejected on its own (k >= 1) - let mut one_cheap = dense; - one_cheap[5] = 0x0000_0401; - assert!(mixer_cost(&one_cheap) > MIXER_COST_REJECT_MAX); - assert!(!mixer_block_admissible(&good_rot, &one_cheap), "a two-adder multiplier"); - // the known-failed day + assert!(!mixer_block_admissible(&[1u32, 2, 3, 3, 3, 3, 3, 3], &dense), "three distinct amounts"); let v5 = Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: true }; let v4 = Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false }; - let worst = crate::seed::seed_words_from_bytes(&crate::bind::day_bytes(29_337)); - let b = MixParams::with_shape(worst, v4); - println!("day 29,337 (2050-04-28) under class v4: cost {}, w32 min {}, distinct rot {}", mixer_cost(&b.mul), b.mul.iter().map(|&m| naf32_weight(m)).min().unwrap(), { let mut r = b.rot.to_vec(); r.sort_unstable(); r.dedup(); r.len() }); - assert!(!mixer_block_admissible(&b.rot, &b.mul), "the known-failed day: the sub-version 3 block of day 29,337 is one the rule rejects"); - let a = MixParams::with_shape(worst, v5); - assert!(a.redraws >= 1 && mixer_block_admissible(&a.rot, &a.mul), "class v5 redraws day 29,337"); - assert_ne!((a.rot, a.mul), (b.rot, b.mul)); let mut redrawn = 0; let mut scanned = 0; for d in 0..60_000u64 { @@ -908,7 +859,7 @@ mod tests { break; } } - assert!(redrawn >= 1, "no redraw in {scanned} days (the census says about 5.69e-4 per day)"); + assert!(redrawn >= 1, "no redraw in {scanned} days (the census says about 6.1e-4 per day)"); } #[test] diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index 75ea927b1..9d1281aeb 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -9,38 +9,18 @@ use crate::seed::day_key; /// window of the load site, `k = min(win, D - 26)` (0 when `D <= 26`), `idx = ((y & (MASK >> k)) | ((off & /// (2^k - 1)) << (D - k))) & MASK`. For every other class `idx = x & MASK`, the lottery hash's address. `mask` is /// `2^D - 1`. The acceptance mirror calls this at the rule's constant `D = 28`. -/// -/// Class v6 lane 1, the index fold (`docs/design/class-v6-rotating-family.md` section 2; `EraParams::fold`): the -/// product's low bits are folded before the rotation, `y = x * M; y ^= y >> 16; y = rotl(y, R)` ([`stride`]), so no -/// era's R lands a biased product bit (bit 0 of `x * M` for odd M is bit 0 of x; bit 1 is set at 3/8) on an address -/// bit. Every era of every other class keeps the plain product. #[inline(always)] pub fn load_index(era: Option<&EraParams>, ins: &Instr, x: u32, mask: u32, log2: u32) -> u32 { match era { None => x & mask, Some(e) => { let (wm, off) = window(ins, mask, log2); - let y = stride(e, x); + let y = x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot); ((y & wm) | off) & mask } } } -/// The era's stride of a source value: `rotl(x * M, R)`, with the index fold of class v6 lane 1 between the product -/// and the rotation when the era carries it (`y ^= y >> 16`). The one place the form lives on the CPU side; the three -/// emitters write the same text ([`crate::emit`]). -#[inline(always)] -pub fn stride(e: &EraParams, x: u32) -> u32 { - let mut y = x.wrapping_mul(e.stride_mul); - if e.fold { - y ^= y >> 16; - } - y.rotate_left(e.stride_rot) -} - -/// The fold's shift (`y ^= y >> INDEX_FOLD_SHIFT`), named for the pack texts and the tests. -pub const INDEX_FOLD_SHIFT: u32 = 16; - /// The window of a load site at a dataset of `2^log2` words: `(window mask, offset)` such that /// `idx = (y & window mask) | offset` lies in the site's aligned window of `2^(log2 - k)` words. #[inline(always)] @@ -51,122 +31,6 @@ pub fn window(ins: &Instr, mask: u32, log2: u32) -> (u32, u32) { (wm, off) } -/// The window of a load site in the 32-bit SOURCE space (the multiply-shift mapping, research class ds55, -/// 8 October 2026): `k = min(win, log2 - 26)` as [`window`] with `log2 = floor(log2(N))`, and `(window mask, -/// offset)` such that `v = (y & window mask) | offset` lies in the site's aligned window of `2^(32 - k)` source -/// values; `idx = (v * N) >> 32` then lands in a contiguous run of about `N / 2^k` words, the site's window of the -/// dataset. -#[inline(always)] -pub fn window32(ins: &Instr, log2: u32) -> (u32, u32) { - let k = (ins.win as u32).min(log2.saturating_sub(26)); - let wm = u32::MAX >> k; - let off = ((((ins.off as u32) & ((1u32 << k) - 1)) as u64) << (32 - k)) as u32; - (wm, off) -} - -/// The dataset's geometry: `2^log2` words under the lottery hash's `src AND MASK` (`mulshift` false), or `words` -/// words, any multiple of 2^16 in `2^28 ..= 2^31`, under the multiply-shift range reduction of spec 01 section -/// 1.13.3 (`mulshift` true: `idx = (src * words) >> 32` in 64 bits; research class ds55, 8 October 2026, no -/// consensus object moves). A power of two given as a word count takes the mask path, so `--dataset-words 2^28` -/// is `--dataset-log2 28` byte for byte. `log2` is `floor(log2(words))` under the multiply-shift (the era -/// window's floor rule reads it); the item index `words / 16 - 1` stays 32-bit. -#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] -pub struct DatasetGeom { - pub log2: u32, - pub words: u64, - pub mulshift: bool, -} - -impl DatasetGeom { - /// The smallest word count `--dataset-words` takes: 2^28 (1 GiB). - pub const MIN_WORDS: u64 = 1 << 28; - /// The largest: 2^31 (8 GiB; the item index stays 32-bit far beyond, the kernel's word index is 32-bit). - pub const MAX_WORDS: u64 = 1 << 31; - /// The step: 2^16 words, so the era layout's interleave (positions below 16) stays a bijection of the dataset - /// and every wide or warp-coalesced load stays inside it. - pub const WORDS_STEP: u64 = 1 << 16; - - /// A dataset of `2^log2` words (4 to 32), the lottery hash's mask path. - pub fn pow2(log2: u32) -> Self { - assert!((4..=32).contains(&log2), "dataset log2 must be in 4..=32"); - Self { log2, words: 1u64 << log2, mulshift: false } - } - - /// A dataset of `words` words: the mask path for a power of two, the multiply-shift otherwise. Refused - /// outside `MIN_WORDS ..= MAX_WORDS` or off the `WORDS_STEP` grid, with the reason. - pub fn words(words: u64) -> Result { - if !(Self::MIN_WORDS..=Self::MAX_WORDS).contains(&words) { - return Err(format!( - "--dataset-words {words}: the word count must be in 2^28 ..= 2^31 ({} ..= {})", - Self::MIN_WORDS, - Self::MAX_WORDS - )); - } - if words % Self::WORDS_STEP != 0 { - return Err(format!("--dataset-words {words}: the word count must be a multiple of 2^16 words (65,536; the era interleave and the wide loads)")); - } - if words.is_power_of_two() { - return Ok(Self::pow2(words.trailing_zeros())); - } - Ok(Self { log2: words.ilog2(), words, mulshift: true }) - } - - /// The last word index (`words - 1`). Under the mask path it is the AND mask. - pub fn last_index(&self) -> u32 { - (self.words - 1) as u32 - } - - /// The AND mask of the mask path; under the multiply-shift the last index (recorded in packs as `IGNEUM_MASK` - /// for the host's self-test, never ANDed). - pub fn mask(&self) -> u32 { - self.last_index() - } - - pub fn items(&self) -> u64 { - self.words >> 4 - } - - pub fn bytes(&self) -> u64 { - self.words << 2 - } - - /// The range reduction of a 32-bit source value to a word index. - #[inline(always)] - pub fn reduce(&self, x: u32) -> u32 { - if self.mulshift { - ((x as u64 * self.words) >> 32) as u32 - } else { - x & self.last_index() - } - } - - /// One line for logs: `2^28 words` or `1476395008 words (5.50 GiB, not a power of two, multiply-shift)`. - pub fn describe(&self) -> String { - if self.mulshift { - format!("{} words ({:.2} GiB, not a power of two, multiply-shift)", self.words, self.bytes() as f64 / (1u64 << 30) as f64) - } else { - format!("2^{} words", self.log2) - } - } -} - -/// [`load_index`] at a dataset geometry: the mask path unchanged; under the multiply-shift the plain load is -/// `(x * N) >> 32` and the era form windows the source first ([`window32`]) then reduces. -#[inline(always)] -pub fn load_index_geom(era: Option<&EraParams>, ins: &Instr, x: u32, geom: DatasetGeom) -> u32 { - if !geom.mulshift { - return load_index(era, ins, x, geom.mask(), geom.log2); - } - match era { - None => geom.reduce(x), - Some(e) => { - let (wm, off) = window32(ins, geom.log2); - let y = stride(e, x); - geom.reduce((y & wm) | off) - } - } -} - /// Read-width experiment (5 October 2026): a `load` of `W` words folds every word into `dst`: /// `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, FOLD_ROT) * FOLD_MUL) XOR w[j]; dst = x`. For `W = 1` this is the /// lottery hash's `dst XOR dataset[...]`. The fold is state-dependent (the rotate-multiply sits between the words), @@ -320,12 +184,8 @@ pub enum Dataset { /// A dataset of `2^log2` words plus the construction that fills it. pub struct DatasetSource { - /// `floor(log2(words))`: the size under the mask path, the era window's floor under the multiply-shift. pub log2_words: u32, - /// The last word index: the AND mask under the mask path (see [`DatasetGeom::mask`]). pub mask: u32, - /// The geometry (size and range reduction); `log2_words` and `mask` are its `log2` and `mask()`. - pub geom: DatasetGeom, /// The day key `K`; `d0, d1 = K[0], K[1]`. pub key: [u32; 8], /// The bytes `K` was derived from (`"day/"` for a string day, `bind::day_bytes` on the chain), recorded @@ -359,25 +219,12 @@ impl DatasetSource { /// `2^shape.cache_log2_words` words on the calling thread. pub fn from_key_shape(key: [u32; 8], mode: DatasetMode, log2_words: u32, shape: Shape) -> Self { assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32"); - Self::from_key_geom(key, mode, DatasetGeom::pow2(log2_words), shape) - } - - /// [`DatasetSource::from_key_shape`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`). - pub fn from_key_geom(key: [u32; 8], mode: DatasetMode, geom: DatasetGeom, shape: Shape) -> Self { + let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 }; let dataset = match mode { DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] }, DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)), }; - Self { log2_words: geom.log2, mask: geom.mask(), geom, key, key_bytes: Vec::new(), dataset, hot: None } - } - - /// This source at another geometry: the cache, key and construction unchanged (an item has the same value at - /// every size), only the size and the range reduction move. - pub fn with_geom(mut self, geom: DatasetGeom) -> Self { - self.log2_words = geom.log2; - self.mask = geom.mask(); - self.geom = geom; - self + Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None } } /// This source with the window's state leaves (class v5, `docs/design/class-v5-stored-state.md`): memory-hard mode @@ -399,7 +246,7 @@ impl DatasetSource { Dataset::MemoryHard(m) => Dataset::MemoryHard(m.refreshed(leaves)), Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"), }; - Self { log2_words: self.log2_words, mask: self.mask, geom: self.geom, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } } /// A copy of this source sharing its cache (and leaves), for a caller that needs an owned source from a shared one. @@ -408,7 +255,7 @@ impl DatasetSource { Dataset::MemoryHard(m) => Dataset::MemoryHard(crate::memhard::MemhardCpu { params: m.params.clone(), cache: m.cache.clone(), leaves: m.leaves.clone() }), Dataset::ClosedForm { d0, d1 } => Dataset::ClosedForm { d0: *d0, d1: *d1 }, }; - Self { log2_words: self.log2_words, mask: self.mask, geom: self.geom, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } } /// The window's state leaves, when the source carries them. @@ -452,14 +299,8 @@ impl DatasetSource { } /// `dataset[w & mask]` under a program's layout (era layout). The closed form has no items and ignores it. - /// Under the multiply-shift geometry `w` must already be a word index (below `words`). pub fn word_at(&self, layout: Layout, w: u32) -> u32 { - let w = if self.geom.mulshift { - assert!((w as u64) < self.geom.words, "word {w} outside a dataset of {} words", self.geom.words); - w - } else { - w & self.mask - }; + let w = w & self.mask; match &self.dataset { Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1), Dataset::MemoryHard(m) => m.word_at(layout, w), @@ -534,38 +375,11 @@ pub fn interpret_warp_scratch( ds: &DatasetSource, trace: bool, ) -> (WarpResult, Vec) { - interpret_warp_core(program, seed, base_nonce, ds, trace, None) -} - -/// The liveness probe of the reg64 window (`accept::check_window_liveness`): `flip` xors register `k` of every -/// lane with `x` at the start of iteration 0, after the init; `first_load_idx` receives the word index each -/// lane read at iteration 0's first load. -#[derive(Clone, Debug, Default)] -pub struct Probe { - pub flip: Option<(usize, u32)>, - pub first_load_idx: [u32; LANES], - pub seen_first_load: bool, -} - -/// [`interpret_warp_init`] under a [`Probe`] (the reg64 liveness rule): the same interpreter, one flip, one read. -pub fn interpret_warp_probe(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource, probe: &mut Probe) -> WarpResult { - interpret_warp_core(program, seed, base_nonce, ds, false, Some(probe)).0 -} - -fn interpret_warp_core( - program: &Program, - seed: &[u32; 8], - base_nonce: u32, - ds: &DatasetSource, - trace: bool, - mut probe: Option<&mut Probe>, -) -> (WarpResult, Vec) { - let geom = ds.geom; + let mask = ds.mask; + let log2 = ds.log2_words; let era = program.class.era; let layout = program.class.layout(); - // 8 registers per lane, or 64 under the reg64 flag (r8..r63 derived from r0..r7 as the kernels do it) - let nregs = program.registers(); - let mut r = vec![[0u32; LANES]; nregs]; + let mut r = [[0u32; LANES]; 8]; for lane in 0..LANES { let nonce = base_nonce.wrapping_add(lane as u32); for i in 0..8 { @@ -574,13 +388,7 @@ fn interpret_warp_core( x = splitmix32(x); r[i][lane] = x ^ seed[(i + 1) & 7]; } - for k in 8..nregs { - r[k][lane] = r[k & 7][lane].wrapping_mul(0x9e3779b9u32).wrapping_add(k as u32); - } } - // the iteration's statements: the drawn program, or its interleaved two-window form under reg64 - let scheduled = program.scheduled(); - let address_mix = program.address_mix(); let mut items_derived = 0usize; let mut idx = [0u32; LANES]; let mut val = [0u32; LANES]; @@ -595,28 +403,10 @@ fn interpret_warp_core( let h = ds.hot.as_ref().expect("a hot-table program needs the epoch's hot table on the dataset source"); assert_eq!(h.n_words(), program.hot_words(), "the hot table's size is the class's"); } - for it in 0..ITERATIONS { - if it == 0 { - // the liveness probe's flip: one register of every lane, complemented at the start of iteration 0 - if let Some(pr) = probe.as_deref_mut() { - if let Some((k, x)) = pr.flip { - for lane in 0..LANES { - r[k][lane] ^= x; - } - } - } - } + for _ in 0..ITERATIONS { let sel = r[0]; - for ins in &scheduled { - step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix); - if it == 0 && ins.op == Op::Load { - if let Some(pr) = probe.as_deref_mut() { - if !pr.seen_first_load { - pr.first_load_idx = idx; - pr.seen_first_load = true; - } - } - } + for ins in &program.instrs { + step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); if ins.op == Op::Scratch { let m = scratch.as_mut().expect("a scratch op needs a scratch class"); let (d, a) = (ins.dst as usize, ins.src as usize); @@ -630,17 +420,7 @@ fn interpret_warp_core( // iteration's `sel`; it is empty on every class without a shadow, so version 2 and class v3 run nothing here. for _ in 0..program.shadow_reps() { for ins in &program.shadow { - step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix); - } - } - } - if nregs > 8 { - // reg64: both windows fold into the eight output registers by xor before the hash fold - for k in 0..8 { - for j in (k + 8..nregs).step_by(8) { - for lane in 0..LANES { - r[k][lane] ^= r[j][lane]; - } + step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); } } } @@ -658,38 +438,19 @@ fn interpret_warp_core( #[allow(clippy::too_many_arguments)] fn step( ins: &Instr, - r: &mut [[u32; LANES]], + r: &mut [[u32; LANES]; 8], sel: &[u32; LANES], - geom: DatasetGeom, + mask: u32, + log2: u32, era: Option<&EraParams>, layout: Layout, ds: &DatasetSource, idx: &mut [u32; LANES], val: &mut [u32; LANES], items_derived: &mut usize, - address_mix: bool, ) { let d = ins.dst as usize; let a = ins.src as usize; - // reg64 full chain: a load's address source is src ^ m, m the rotate-xor chain over the 63 other registers in - // index order (m = first; m = rotl(m, 1) ^ next). The source itself stays out of the chain: with it inside, a - // register whose term lands at rotation 0 mod 32 (r31, r63) cancels its own direct term, a dead register the - // liveness rule found on the pinned draw (first load src r31, 8 October 2026, 15:5x UK). - let addr_src = |r: &[[u32; LANES]], lane: usize| -> u32 { - if !address_mix { - return r[a][lane]; - } - let mut m = 0u32; - let mut started = false; - for k in 0..r.len() { - if k == a { - continue; - } - m = if started { m.rotate_left(1) ^ r[k][lane] } else { r[k][lane] }; - started = true; - } - r[a][lane] ^ m - }; match ins.op { Op::Add => { let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32); @@ -758,7 +519,7 @@ fn step( } Op::Load if ins.width == 1 => { for lane in 0..LANES { - idx[lane] = load_index_geom(era, ins, addr_src(r, lane), geom); + idx[lane] = load_index(era, ins, r[a][lane], mask, log2); } *items_derived += ds.fetch(idx, val, layout); for lane in 0..LANES { @@ -770,7 +531,7 @@ fn step( let width = ins.width as usize; let align = !(ins.width as u32 - 1); for lane in 0..LANES { - idx[lane] = load_index_geom(era, ins, addr_src(r, lane), geom) & align; + idx[lane] = load_index(era, ins, r[a][lane], mask, log2) & align; } let mut vals = [[0u32; 16]; LANES]; *items_derived += ds.fetch_wide(idx, width, &mut vals, layout); @@ -790,8 +551,8 @@ fn step( } } Op::WLoad => { - // Lane 0's register, range-reduced, aligned down to 32 words; lane l reads word base + l. - let base = geom.reduce(r[a][0]) & !31; + // Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l. + let base = (r[a][0] & mask) & !31; for lane in 0..LANES { idx[lane] = base + lane as u32; } diff --git a/igneum-pow/tests/packs.rs b/igneum-pow/tests/packs.rs index 9b3aecf48..f66d91536 100644 --- a/igneum-pow/tests/packs.rs +++ b/igneum-pow/tests/packs.rs @@ -997,229 +997,3 @@ fn v5_pack_is_the_v4_program_over_the_state_leaves() { let mh5 = v5_read("v5-genesis", "memhard.h"); assert!(mh5.contains("s[i] ^= leaf[i]"), "the leaf XOR before the first mixer"); } - -// --------------------------------------------------------------------------------------------------------------- -// The 64-register window (the hash lane's measurement, 8 October 2026, a research class behind `+reg64`). - -/// The `source` line the pinned devnet pack was exported with (vectors.json and vectors.h carry it). -const PINNED_SOURCE: &str = "igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset"; - -/// The pinned devnet pack's epoch again, with the reg64 flag stamped on the program after the draw (as `--reg64` -/// does): the same instructions, the same dataset, 64 registers per lane. -fn epoch_reg64_of_devnet() -> Epoch { - let pack = "mx8-devnet-epoch0"; - let j = json(pack, "program.json"); - let seed = j["seed"].as_str().unwrap(); - let seed_bytes = unhex(&j["seed_bytes"]); - let day_bytes = unhex(&j["dataset"]["day_bytes"]); - let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; - let era = j.get("era_seed_bytes").map(unhex); - let mut program = generate_from_seed_bytes_program_class(seed, &seed_bytes, ProgramClass::V3, era.as_deref()); - program.class = program.class.with_reg64(); - let shape = Shape::for_class(&program.class); - let mut dataset = DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2, shape); - dataset.key_bytes = day_bytes; - Epoch { program, dataset } -} - -/// The plain path re-exports the pinned devnet pack unchanged: every file `export` writes, byte for byte (the -/// reg64 code sits behind the flag; a pack without it does not move). -#[test] -fn reg64_plain_path_reexports_the_pinned_devnet_pack_unchanged() { - let pack = "mx8-devnet-epoch0"; - let e = epoch(pack); - assert!(!e.program.class.reg64); - assert_eq!(e.program.registers(), 8); - assert_eq!(e.program.scheduled(), e.program.instrs, "without the flag the schedule is the drawn program"); - let out = export_pack(e, &day_label(pack), PINNED_SOURCE); - assert_eq!(out.files.len(), 12); - for (name, text) in &out.files { - assert_same_text(pack, name, text); - } -} - -/// The reg64 variant: the CPU verifier and the emitted CUDA text read one schedule, so the kernel's statements name -/// the registers the verifier wrote, in the verifier's order; the vectors are the verifier's; the plain pack's -/// instructions are untouched. -#[test] -fn reg64_cpu_verifier_and_cuda_text_agree() { - let plain = epoch("mx8-devnet-epoch0"); - let e = epoch_reg64_of_devnet(); - let p = &e.program; - assert!(p.class.reg64); - assert_eq!(p.registers(), 64); - assert_eq!(p.class.name(), "mx8-erad810f22d+reg64", "the pinned pack's era class, the window outermost"); - assert_eq!(LoadClass::parse("mx8+reg64"), Some(V3_CLASS.with_reg64()), "the class name parses"); - assert_eq!(LoadClass::MX8.with_reg64().name(), "mx8+reg64", "and round-trips"); - assert_eq!(p.instrs, plain.program.instrs, "the draw is the class's without the flag"); - assert_ne!(p.program_id(), plain.program.program_id(), "the id carries the flag"); - // the schedule: 128 statements, instruction i of the draw on window 0 (+8 * (i % 4)) then on window 1 (+32) - let sched = p.scheduled(); - assert_eq!(sched.len(), 128); - let mut touched = [false; 64]; - for (i, ins) in p.instrs.iter().enumerate() { - let off = (8 * (i % 4)) as u8; - let a = sched[2 * i]; - let b = sched[2 * i + 1]; - assert_eq!((a.op, a.dst, a.src, a.src2), (ins.op, ins.dst + off, ins.src + off, ins.src2 + off), "instruction {i} on window 0"); - assert_eq!((b.op, b.dst, b.src, b.src2), (ins.op, a.dst + 32, a.src + 32, a.src2 + 32), "instruction {i} on window 1"); - touched[a.dst as usize] = true; - touched[b.dst as usize] = true; - } - // the fixed extension 8 * (i % 4) gives each octet 16 of the 64 instructions, so a draw need not write every - // register: the pinned draw writes 28 of 32 per window (r11, r17, r24, r29 unwritten, read or folded only) - let written = touched.iter().filter(|&&t| t).count(); - assert!(written >= 48 && written % 2 == 0, "the schedule writes {written} of 64 registers (both windows alike)"); - // the CUDA texts: 64 registers declared, 56 derived, every statement of the schedule in order, the fold, 32 masked loads - let mp = e.dataset.memhard().map(|m| &m.params); - for (name, text) in [("kernel.cu", cuda_kernel(p, mp)), ("kernel_bound.cu", cuda_kernel_bound(p, mp))] { - assert!(text.contains("uint32_t r0, r1, r2, r3, r4, r5, r6, r7, r8, r9,") && text.contains(", r62, r63;"), "{name}: 64 registers"); - for k in 8..64 { - assert!(text.contains(&format!(" r{k} = r{} * 0x9e3779b9u + {k}u;\n", k & 7)), "{name}: r{k} derived"); - } - for k in 0..8 { - let terms: Vec = (k + 8..64).step_by(8).map(|j| format!("r{j}")).collect(); - assert!(text.contains(&format!(" r{k} = r{k} ^ {};\n", terms.join(" ^ "))), "{name}: the fold of r{k}"); - } - assert_eq!(text.matches("ds[((rotl_imm(").count(), 2 * LOAD_SLOTS, "{name}: 32 loads (the era form)"); - assert_eq!(text.matches(" & mask]").count(), 2 * LOAD_SLOTS, "{name}: 32 masked loads"); - // every statement line names the schedule's registers: dst first, then src on every op that reads one - let lines: Vec<&str> = text.lines().filter(|l| l.starts_with(" r") && l.contains(" // ")).collect(); - assert_eq!(lines.len(), 128, "{name}: 128 statements per iteration"); - for (k, (line, ins)) in lines.iter().zip(sched.iter()).enumerate() { - let stmt = line.trim_start(); - assert!(stmt.starts_with(&format!("r{} = r{} ", ins.dst, ins.dst)) || stmt.starts_with(&format!("r{} = ", ins.dst)), "{name} statement {k}: dst r{}: {stmt}", ins.dst); - assert!(line.ends_with(&format!("// {k} {}", ins.op.name())), "{name} statement {k}: numbered: {line}"); - if ins.op != Op::Rotl && ins.op != Op::Load { - assert!(stmt.contains(&format!("r{}", ins.src)) || stmt.contains(&format!("r{},", ins.src)), "{name} statement {k}: src r{}: {stmt}", ins.src); - } - } - } - // Metal and OpenCL carry the same window (class-v6, 8 October 2026): 64 registers declared, 56 derived, the fold, - // 128 statements per iteration in the schedule's order, 32 loads of the era form - for (name, text) in [ - ("program.metal", metal_program(p, e.dataset.log2_words, LoadSource::Stored)), - ("program_bound.metal", metal_program_bound(p, e.dataset.log2_words)), - ("kernel.cl", opencl_kernel(p, mp)), - ("kernel_bound.cl", opencl_kernel_bound(p, mp)), - ] { - assert!(!text.starts_with("// reg64: no "), "{name}: the window text, not a stub"); - assert!(text.contains("uint r0, r1, r2, r3, r4, r5, r6, r7, r8, r9,") && text.contains(", r62, r63;"), "{name}: 64 registers"); - for k in 8..64 { - assert!(text.contains(&format!(" r{k} = r{} * 0x9e3779b9u + {k}u;\n", k & 7)), "{name}: r{k} derived"); - } - for k in 0..8 { - let terms: Vec = (k + 8..64).step_by(8).map(|j| format!("r{j}")).collect(); - assert!(text.contains(&format!(" r{k} = r{k} ^ {};\n", terms.join(" ^ "))), "{name}: the fold of r{k}"); - } - // a statement line opens with its register, or with the shuffle's block in OpenCL ("{ uint t_; ...") - let lines: Vec<&str> = text.lines().filter(|l| (l.starts_with(" r") || l.starts_with(" { uint t_;")) && l.contains(" // ")).collect(); - assert!(lines.len() % 128 == 0 && !lines.is_empty(), "{name}: {} statement lines, a multiple of 128", lines.len()); - for (k, (line, ins)) in lines.iter().take(128).zip(sched.iter()).enumerate() { - let stmt = line.trim_start(); - assert!(stmt.starts_with(&format!("r{} = ", ins.dst)) || (stmt.starts_with("{ uint t_;") && stmt.contains(&format!("r{} = r{} ^ t_", ins.dst, ins.dst))), "{name} statement {k}: dst r{}: {stmt}", ins.dst); - } - assert_eq!(text.matches("rotl_imm(").count() >= 2 * LOAD_SLOTS, true, "{name}: the era loads"); - } - // the pack: the vectors are the reg64 verifier's and differ from the plain pack's; program.h and program.json carry the flag - let out = export_pack(&e, &day_label("mx8-devnet-epoch0"), "test"); - for (i, &base) in out.bases.iter().enumerate() { - assert_eq!(out.outs[i], e.hash_warp(base)); - assert_ne!(out.outs[i], plain.hash_warp(base), "base {base}: the reg64 hashes differ from the plain pack's"); - } - let file = |n: &str| out.files.iter().find(|(f, _)| f == n).map(|(_, t)| t.clone()).unwrap(); - assert!(file("program.h").contains("#define IGNEUM_REG64 1\n#define IGNEUM_REGISTERS 64\n")); - let j: Value = serde_json::from_str(&file("program.json")).unwrap(); - assert_eq!(j["load_class"].as_str().unwrap(), "mx8-erad810f22d+reg64"); - assert_eq!(j["reg64"]["registers"].as_u64().unwrap(), 64); - assert_eq!(j["reg64"]["statements_per_iteration"].as_u64().unwrap(), 128); - assert_eq!(j["program_class"].as_str().unwrap(), "v3"); - let v: Value = serde_json::from_str(&file("vectors.json")).unwrap(); - assert_eq!(v["warps"].as_array().unwrap().len(), 3); -} - -/// The full-chain variant (`+reg64c`): every load's address source is `src ^ m` with `m` the rotate-xor mix of all -/// 64 registers; the CUDA text carries the mix statement before each of the 32 loads, the verifier the same chain, -/// and the hashes differ from the arithmetic-only window's. -#[test] -fn reg64_full_chain_cpu_verifier_and_cuda_text_agree() { - let mut e = epoch_reg64_of_devnet(); - let arithmetic_only: Vec<[u64; 32]> = [0u32, 4096].iter().map(|&b| e.hash_warp(b)).collect(); - e.program.class = e.program.class.with_reg64_chain(); - let p = &e.program; - assert!(p.address_mix()); - assert_eq!(p.class.name(), "mx8-erad810f22d+reg64c"); - assert_eq!(LoadClass::parse("mx8+reg64c"), Some(V3_CLASS.with_reg64().with_reg64_chain())); - assert_ne!(LoadClass::parse("mx8+reg64c"), LoadClass::parse("mx8+reg64")); - assert_ne!(p.program_id(), epoch_reg64_of_devnet().program.program_id(), "the id carries the chain"); - assert_ne!(p.program_id(), 0x3deee2320e70e1bf, "the sound fold's id differs from the benched text's (the source inside the chain)"); - assert_eq!(p.scheduled().len(), 128, "the schedule is the window's"); - let mp = e.dataset.memhard().map(|m| &m.params); - let sched = p.scheduled(); - let load_srcs: Vec = sched.iter().filter(|i| i.op == Op::Load).map(|i| i.src).collect(); - for (name, text) in [("kernel.cu", cuda_kernel(p, mp)), ("kernel_bound.cu", cuda_kernel_bound(p, mp))] { - assert!(text.contains(" uint32_t m; // reg64 full chain"), "{name}: m declared"); - assert_eq!(text.matches(" ^ m)").count(), 2 * LOAD_SLOTS, "{name}: 32 mixed address sources"); - assert_eq!(text.matches(" & mask]").count(), 2 * LOAD_SLOTS, "{name}: 32 masked loads"); - // the mix line sits right before its load: 63 terms in index order, the load's own source left out, and the - // load reads (rS ^ m) with the schedule's src - let lines: Vec<&str> = text.lines().collect(); - let mut loads = 0; - for (i, l) in lines.iter().enumerate() { - let t = l.trim_start(); - if t.starts_with("m = r") && t.contains("rotl_imm(m, 1u)") { - let src = load_srcs[loads]; - let next = lines[i + 1].trim_start(); - assert!(next.contains("ds[") && next.contains(&format!("(r{src} ^ m)")), "{name}: load {loads} follows its mix with src r{src}: {next}"); - let mut want = String::new(); - for k in 0..64u8 { - if k == src { - continue; - } - if want.is_empty() { - want.push_str(&format!("m = r{k};")); - } else { - want.push_str(&format!(" m = rotl_imm(m, 1u) ^ r{k};")); - } - } - assert_eq!(t, want, "{name}: the mix of load {loads} (src r{src})"); - loads += 1; - } - } - assert_eq!(loads, 2 * LOAD_SLOTS, "{name}: the mix before each of the 32 loads"); - } - for (i, &b) in [0u32, 4096].iter().enumerate() { - assert_ne!(e.hash_warp(b), arithmetic_only[i], "base {b}: the chain's hashes differ from the window's"); - } - let out = export_pack(&e, &day_label("mx8-devnet-epoch0"), "test"); - let file = |n: &str| out.files.iter().find(|(f, _)| f == n).map(|(_, t)| t.clone()).unwrap(); - assert!(file("program.h").contains("#define IGNEUM_REG64_ADDRESS_MIX 1")); - let j: Value = serde_json::from_str(&file("program.json")).unwrap(); - assert_eq!(j["load_class"].as_str().unwrap(), "mx8-erad810f22d+reg64c"); - assert_eq!(j["reg64"]["variant"].as_str().unwrap(), "window, full chain"); - assert_eq!(j["reg64"]["address_mix"].as_u64().unwrap(), 1); - assert!(j["reg64"]["liveness"].as_str().unwrap().contains("64 independently necessary values")); - assert_eq!(j["program_class"].as_str().unwrap(), "v3"); -} - -/// The window's liveness rule (`accept::check_window_liveness`): the arithmetic-only window is the known-failed -/// fixture (a load's address reads its own window's register, so complementing a register of another octet leaves -/// iteration 0's first load address where it was); the full-chain form passes; a program without the window is -/// `NotAWindow`. -#[test] -fn reg64_liveness_rule_refuses_the_subset_fold_and_passes_the_full_chain() { - use igneum_pow::accept::{check_window_liveness, Reject}; - let plain = epoch("mx8-devnet-epoch0"); - assert_eq!(check_window_liveness(&plain.program), Err(Reject::NotAWindow)); - let window = epoch_reg64_of_devnet(); - match check_window_liveness(&window.program) { - Err(Reject::DeadWindowRegister { address_changed, result_changed, .. }) => { - assert!(!address_changed, "the subset fold leaves the first load address where it was"); - assert!(result_changed, "the end fold still reads every register"); - } - other => panic!("the arithmetic-only window must be refused as a dead register: {other:?}"), - } - let mut chain = epoch_reg64_of_devnet(); - chain.program.class = chain.program.class.with_reg64_chain(); - assert_eq!(check_window_liveness(&chain.program), Ok(()), "the full chain keeps all 64 registers live across the chain"); -} diff --git a/infra/fast-time/proving-cache-history.mjs b/infra/fast-time/proving-cache-history.mjs new file mode 100644 index 000000000..2934bd5d6 --- /dev/null +++ b/infra/fast-time/proving-cache-history.mjs @@ -0,0 +1,253 @@ +#!/usr/bin/env node +// The two-node cache-history case (review B F01, Phase 1 item (c), 8 October 2026): on a node with the verdict cache +// of verdict-cache-fix-node 3f672661, a proof's bytes refused once for CONTEXT (another statement) must still pay +// when the same bytes arrive later as their own honest record, answered from the cached facts with no second SP1 +// verify; bytes refused once for being INVALID must be refused again at a later carrier, from the cache, with no +// second verify. Three local nodes at fast time as in proving-enforcement.mjs: H1 and H2 honest (the verifier in +// consensus), A the carrier (skip rule, trust verify) whose templates carry what its pool holds. +// +// The real proof is one of this chain's own blocks: after the chain has run, block n is exported from H1 +// (igneum_exportSegments), cut by igneum-prove-export and proven compressed by igneum-prove-host on the box's CPU +// (about 70 to 110 s), so its statement is H1's native statement for (n, shard 0, PAYOUT). The node binaries and the +// proving binaries must carry the SAME pair (the node's elf/ overlay and the host's embedded pin: pin C tonight), and +// the node's object names that pair (proving_shard_program_id / proving_aggregator_id from the manifest). +// +// node infra/fast-time/proving-cache-history.mjs --prove-bin +// [--before 90] [--after 120] [--slot 0] [--out ] +// +// Verdict: (1) block n shard 0 paid exactly once, to PAYOUT; (2) H1's verifies.run grew by exactly 2 over the run (the +// real proof once, the invalid bytes once) and verifies.cacheAnswers by at least 2; (3) the context refusal of phase 1 +// and the invalid refusal of phase 3 both read in H1's log; (4) the second use of the invalid bytes produced a refusal +// with no new verify. +import { spawn, spawnSync } from 'node:child_process'; +import { mkdirSync, rmSync, writeFileSync, readFileSync, openSync, existsSync, appendFileSync } from 'node:fs'; +import { createHash, randomBytes } from 'node:crypto'; + +const ROOT = new URL('../../', import.meta.url).pathname; +const FILE = `${ROOT}infra/fast-time/override-60x.json`; +const MANIFEST = `${ROOT}proving/igneum-prove/elf/manifest.json`; +const IGNEUMD = process.env.IGNEUMD || `${ROOT}vendor/igneum-node/target/release/igneumd`; +const CPU_MINER = process.env.IGNEUM_MINER || `${ROOT}vendor/igneum-node/target/release/igneum-miner`; +const args = process.argv.slice(2); +const flag = (name, dflt) => { const i = args.indexOf(`--${name}`); return i >= 0 ? Number(args[i + 1]) : dflt; }; +const sflag = (name, dflt) => { const i = args.indexOf(`--${name}`); return i >= 0 ? args[i + 1] : dflt; }; +const BEFORE = flag('before', 90), AFTER = flag('after', 120), SLOT = flag('slot', 0); +const PROVE_BIN = sflag('prove-bin', `${ROOT}proving/igneum-prove/target/release`); +const GENESIS_BITS = flag('genesis-bits', 0x1f010000); +const OUT = sflag('out') || `${ROOT}docs/plans/proving-enforcement/cache-history.json`; +const BASE = 30890 + SLOT * 40, SUFFIX = 985 + SLOT; +const CHAIN = `igneum-devnet-${SUFFIX}`; +const TMP = `/tmp/igneum-fast-time-ch${SLOT}`; +const PIDS = `${TMP}/pids`; +const NEVER = '18446744073709551615'; +for (const b of [IGNEUMD, CPU_MINER, `${PROVE_BIN}/igneum-prove-host`, `${PROVE_BIN}/igneum-prove-export`]) if (!existsSync(b)) { console.error(`missing ${b}`); process.exit(2); } + +const t0 = Date.now(); +const log = (s) => { const t = new Date().toISOString().slice(11, 23); console.log(`${t} ch${SLOT} ${s}`); }; +const sleep = (ms) => new Promise(r => setTimeout(r, ms)); +const hex = (b) => '0x' + Buffer.from(b).toString('hex'); +const sha256 = (b) => createHash('sha256').update(b).digest('hex'); + +// leftovers of an earlier run on this slot: by the pid file only +rmSync(TMP, { recursive: true, force: true }); mkdirSync(TMP, { recursive: true }); +const started = []; +const track = (proc) => { started.push(proc); try { appendFileSync(PIDS, `${proc.pid}\n`); } catch { } }; + +// the override: the 60x file with the pair named from the manifest, the verifier in consensus from genesis, no succession +let baseText = readFileSync(FILE, 'utf8'); +for (const k of ['proving_payment_activation_daa']) baseText = baseText.replace(new RegExp(`\\s*"${k}":\\s*[^,}\\n]+,?`), ''); +const manifest = JSON.parse(readFileSync(MANIFEST, 'utf8')); +function mergeOverrideText(text, fields) { + let out = text; + for (const k of Object.keys(fields)) out = out.replace(new RegExp(`\\s*"${k}":\\s*[^,}\\n]+,?`), ''); + const extra = Object.entries(fields).map(([k, v]) => `"${k}": ${typeof v === 'string' && !/^\d+$/.test(v) && v !== 'true' && v !== 'false' ? JSON.stringify(v) : v}`).join(', '); + return out.replace(/,?\s*}\s*$/, `,\n ${extra}\n}\n`); +} +const override = `${TMP}/override.json`; +writeFileSync(override, mergeOverrideText(baseText, { + genesis_bits: GENESIS_BITS, skip_proof_of_work: false, proving_v0_activation_daa: '0', proving_consensus_verify_daa: '0', + proving_shard_program_id: manifest.shard.program_id, proving_aggregator_id: manifest.aggregator.program_id, + verifier_in_consensus: 'true', proving_key_succession_daa: NEVER, proving_key_succession_window_daa: '120', + proving_next_shard_program_id: '', proving_next_aggregator_id: '', + program_class_v3_activation_daa: NEVER, program_class_v4_activation_daa: NEVER, +})); +const DAY_MS = (() => { const m = /"pow_day_ms":\s*([0-9]+)/.exec(baseText); return m ? +m[1] : 1440000; })(); + +class Node { + constructor(name, i, peers = [], env = {}) { + this.name = name; this.i = i; this.grpcPort = BASE + i * 10; this.p2pPort = BASE + i * 10 + 1; this.jsonPort = BASE + i * 10 + 2; this.evmPort = BASE + i * 10 + 3; + this.peers = peers; this.env = env; this.dir = `${TMP}/${name}`; this.logFile = `${this.dir}/node.log`; + } + get grpc() { return `grpc://127.0.0.1:${this.grpcPort}`; } + async start() { + mkdirSync(this.dir, { recursive: true }); + const out = openSync(this.logFile, 'a'); + const a = ['--devnet', `--devnet-suffix=${SUFFIX}`, '--nodnsseed', '--disable-upnp', '--nologfiles', '--enable-unsynced-mining', '--utxoindex', + `--appdir=${this.dir}`, `--rpclisten=127.0.0.1:${this.grpcPort}`, `--rpclisten-json=127.0.0.1:${this.jsonPort}`, `--evm-rpclisten=127.0.0.1:${this.evmPort}`, + `--listen=127.0.0.1:${this.p2pPort}`, `--override-params-file=${override}`, '--loglevel=info', '--yes']; + if (this.peers.length) for (const p of this.peers) a.push(`--addpeer=127.0.0.1:${p}`); else a.push('--outpeers=0'); + this.proc = spawn(IGNEUMD, a, { stdio: ['ignore', out, out], env: { ...process.env, ...this.env } }); + track(this.proc); + for (let k = 0; k < 60; k++) { + await sleep(1000); + try { await this.exec('igneum_getExecStatus', []); log(`${this.name} up (pid ${this.proc.pid}) after ${k + 1} s`); return; } catch { } + if (this.proc.exitCode !== null) throw new Error(`${this.name} exited ${this.proc.exitCode}: ${readFileSync(this.logFile, 'utf8').split('\n').slice(-5).join(' | ')}`); + } + throw new Error(`${this.name} did not answer in 60 s`); + } + async exec(method, params) { + const r = await fetch(`http://127.0.0.1:${this.evmPort}`, { method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify({ jsonrpc: '2.0', id: 1, method, params }) }); + const j = await r.json(); + if (j.error) throw new Error(`${method}: ${j.error.message || JSON.stringify(j.error)}`); + return j.result; + } + logText() { try { return readFileSync(this.logFile, 'utf8'); } catch { return ''; } } + stop() { try { this.proc.kill('SIGINT'); } catch { } } +} +function startMiner(node, label, secs) { + const out = openSync(`${TMP}/${label}.log`, 'a'); + const p = spawn(CPU_MINER, ['mine', node.grpc, '1', String(secs), label, '--engine', 'igneum-pow', '--payout-label', label, '--status-secs', '30', '--no-vote'], { stdio: ['ignore', out, out], env: { ...process.env, IGNEUM_POW_DAY_MS: String(DAY_MS) } }); + track(p); + return p; +} +function signRecord(label, chain, block, number, shard, payout, statement, proofHash) { + const r = spawnSync(CPU_MINER, ['sign-record', label, chain, block, String(number), String(shard), payout, statement, proofHash], { encoding: 'utf8' }); + if (r.status !== 0) throw new Error(`sign-record failed: ${r.stderr || r.stdout}`); + return JSON.parse(r.stdout.trim().split('\n').pop()); +} +const H1 = new Node('H1', 0); +const H2 = new Node('H2', 1, [H1.p2pPort]); +const A = new Node('A', 2, [H1.p2pPort, H2.p2pPort], { IGNEUM_TEST_SKIP_PROOF_RULE: '1', IGNEUM_PROOF_VERIFY: 'trust' }); +const nodes = [H1, H2, A]; +let miners = []; +async function stopAll() { + for (const m of miners) { try { m.kill('SIGTERM'); } catch { } } + for (const n of nodes) n.stop(); + await sleep(2000); + for (const p of started) { try { p.kill('SIGKILL'); } catch { } } +} +process.on('SIGINT', async () => { await stopAll(); process.exit(130); }); +const PAYOUT = '0x' + 'c3'.repeat(20); +async function tipNumber(node) { const s = await node.exec('igneum_getExecStatus', []); return Number(s.executedTip ?? 0); } +async function sameTip(a, h) { + try { const x = await a.exec('igneum_getExecStatus', []); const y = await h.exec('igneum_getExecStatus', []); return x.executedTipHash && x.executedTipHash === y.executedTipHash; } catch { return false; } +} +async function rejoin(label) { + for (let k = 0; k < 120; k++) { if (await sameTip(A, H1)) { log(`${label}: A is on H1's tip (${k * 5} s)`); return true; } await sleep(5000); } + log(`${label}: A did not rejoin H1's tip in 600 s`); return false; +} +async function verifies(node) { const s = await node.exec('igneum_getProvingStatus', []); return s.verifies || { run: 0, cacheAnswers: 0 }; } +const refusals = (node) => (node.logText().match(/a carried proof record's proof does not verify: [^\n]*/g) || []); +async function submitVia(node, name, record, proof) { + let out; + try { out = await node.exec('igneum_submitProofRecord', [{ record, proof }]); } catch (e) { out = { accepted: false, reason: String(e.message) }; } + log(`${name}: A's pool ${out.accepted ? 'accepted' : 'refused'} (${out.reason || ''})`); + return out; +} +async function observe(n) { + try { + const r = await H1.exec('igneum_getProofRecords', ['0x' + n.toString(16)]); + return { paid: (r.paid || []).filter(Boolean), carried: (r.carried || []).map(c => ({ key: c.keyHash, carrier: c.carrierNumber, rejected: c.rejected, paidWei: c.paidWei })) }; + } catch (e) { return { error: e.message }; } +} + +/// A real proof of block n of this chain: exported from H1, cut and proven by the host (CPU), returning the proof +/// bytes and the statement the host printed (which must equal H1's native statement for (n, 0, PAYOUT)). +function proveBlock(n) { + const dir = `${TMP}/prove-${n}`; mkdirSync(dir, { recursive: true }); + const body = JSON.stringify({ jsonrpc: '2.0', id: 1, method: 'igneum_exportSegments', params: ['0x0', '0x' + n.toString(16)] }); + const r = spawnSync('curl', ['-s', '-m', '600', '-X', 'POST', `http://127.0.0.1:${H1.evmPort}`, '-H', 'Content-Type: application/json', '--data-binary', body], { encoding: 'utf8', maxBuffer: 1 << 30 }); + if (r.status !== 0) throw new Error(`export failed: ${r.stderr}`); + writeFileSync(`${dir}/export.json`, JSON.stringify(JSON.parse(r.stdout).result)); + const e = spawnSync(`${PROVE_BIN}/igneum-prove-export`, [`${dir}/export.json`, String(n), `${dir}/block-${n}.json`, '--source', `fast-time cache-history ${CHAIN}`], { encoding: 'utf8' }); + if (e.status !== 0) throw new Error(`igneum-prove-export failed: ${(e.stdout + e.stderr).slice(-400)}`); + log(`block ${n} cut: ${(e.stdout.trim().split('\n').pop() || '').slice(0, 160)}`); + const h = spawnSync(`${PROVE_BIN}/igneum-prove-host`, [`${dir}/block-${n}.json`, '--mode', 'compressed', '--shard', '0', '--prover', PAYOUT, '--out', `${dir}/results.json`], { encoding: 'utf8', env: { ...process.env, SP1_PROVER: 'cpu' } }); + const line = (h.stdout.match(/RESULT compressed shard 0: [^\n]*/) || [''])[0]; + log(`host: ${line.slice(0, 200)}${h.status !== 0 ? ` (exit ${h.status}: ${(h.stderr || '').slice(-300)})` : ''}`); + if (h.status !== 0 || !/VERIFIED/.test(line)) throw new Error(`the host did not prove block ${n}`); + const file = `${dir}/block-${n}-shard-0-compressed.bin`; + if (!existsSync(file)) throw new Error(`no proof file ${file}`); + const statement = (line.match(/statement (0x[0-9a-f]{64})/) || [])[1]; + return { bytes: readFileSync(file), statement, line }; +} + +const result = { case: 'cache-history', node: IGNEUMD, proveBin: PROVE_BIN, pair: { shard: manifest.shard.program_id, aggregator: manifest.aggregator.program_id }, startedAt: new Date().toISOString(), phases: {} }; +try { + await H1.start(); await H2.start(); await A.start(); + miners = [startMiner(H1, 'h1', BEFORE + 3 * AFTER + 900), startMiner(H2, 'h2', BEFORE + 3 * AFTER + 900), startMiner(A, 'attacker', BEFORE + 3 * AFTER + 900)]; + log(`phase 0: mining ${BEFORE} s`); + await sleep(BEFORE * 1000); + const v0 = await verifies(H1); + // the block to prove: outside the exclusive window, executed on every node + let tip = await tipNumber(H1); + const n = Math.max(1, tip - 15); + const plan = await H1.exec('igneum_getShardPlan', ['0x' + n.toString(16), PAYOUT]); + const native = plan.shards[0].statement; + log(`block ${n} ${plan.hash}: H1's native statement for shard 0 and ${PAYOUT}: ${native}`); + const proof = proveBlock(n); + result.proof = { block: n, statement: proof.statement, native, bytes: proof.bytes.length, sha256: '0x' + sha256(proof.bytes), hostLine: proof.line }; + if (proof.statement !== native) throw new Error(`the host's statement ${proof.statement} is not H1's ${native} (host and node layouts or pairs differ)`); + // phase 1: the real bytes under a record for another block (the statement of block m): refused for context + await rejoin('before phase 1'); + tip = await tipNumber(A); + const m = Math.max(1, tip - 15); + const planM = await A.exec('igneum_getShardPlan', ['0x' + m.toString(16), PAYOUT]); + const recWrong = signRecord('ch-wrong', CHAIN, planM.hash, m, 0, PAYOUT, planM.shards[0].statement, '0x' + sha256(proof.bytes)).record; + const r1 = await submitVia(A, 'phase 1 (real bytes, another block\'s statement)', recWrong, hex(proof.bytes)); + await sleep(AFTER * 1000); + const v1 = await verifies(H1); + const o1 = await observe(m); + result.phases.context = { block: m, submit: r1, observed: o1, verifies: v1, refusals: refusals(H1) }; + log(`phase 1: H1 verifies ${JSON.stringify(v1)}; block ${m} paid ${JSON.stringify(o1.paid)}; refusals ${refusals(H1).length}`); + // phase 2: the same bytes as their own honest record: answered from the cached facts, paid once + await rejoin('before phase 2'); + const recHonest = signRecord('ch-honest', CHAIN, plan.hash, n, 0, PAYOUT, native, '0x' + sha256(proof.bytes)).record; + const r2 = await submitVia(A, 'phase 2 (the same bytes, the honest record)', recHonest, hex(proof.bytes)); + await sleep(AFTER * 1000); + const v2 = await verifies(H1); + const o2 = await observe(n); + result.phases.honest = { block: n, submit: r2, observed: o2, verifies: v2 }; + log(`phase 2: H1 verifies ${JSON.stringify(v2)}; block ${n} paid ${JSON.stringify(o2.paid)}; carried ${JSON.stringify(o2.carried)}`); + // phase 3: invalid bytes twice, under two carriers: verified once, refused from the cache the second time + await rejoin('before phase 3'); + const bad = randomBytes(2048); + tip = await tipNumber(A); + const p = Math.max(1, tip - 15); + const planP = await A.exec('igneum_getShardPlan', ['0x' + p.toString(16), PAYOUT]); + const recBad1 = signRecord('ch-bad1', CHAIN, planP.hash, p, 0, PAYOUT, planP.shards[0].statement, '0x' + sha256(bad)).record; + const r3 = await submitVia(A, 'phase 3a (invalid bytes)', recBad1, hex(bad)); + await sleep(AFTER * 1000); + const v3a = await verifies(H1); + const ref3a = refusals(H1).length; + await rejoin('before phase 3b'); + tip = await tipNumber(A); + const q = Math.max(1, tip - 15); + const planQ = await A.exec('igneum_getShardPlan', ['0x' + q.toString(16), PAYOUT]); + const recBad2 = signRecord('ch-bad2', CHAIN, planQ.hash, q, 0, PAYOUT, planQ.shards[0].statement, '0x' + sha256(bad)).record; + const r4 = await submitVia(A, 'phase 3b (the same invalid bytes, a later carrier)', recBad2, hex(bad)); + await sleep(AFTER * 1000); + const v3b = await verifies(H1); + const ref3b = refusals(H1).length; + result.phases.invalid = { blocks: [p, q], submits: [r3, r4], verifies: { after3a: v3a, after3b: v3b }, refusals: { after3a: ref3a, after3b: ref3b }, reasons: refusals(H1) }; + log(`phase 3: H1 verifies after 3a ${JSON.stringify(v3a)}, after 3b ${JSON.stringify(v3b)}; refusals ${ref3a} then ${ref3b}`); + // the verdict + const paidOnce = (o2.paid || []).length === 1 && (o1.paid || []).length === 0; + const runs = Number(v3b.run) - Number(v0.run); + const cache = Number(v3b.cacheAnswers) - Number(v0.cacheAnswers); + const contextRefused = result.phases.context.refusals.length >= 1; + const invalidRefusedTwiceNoVerify = ref3b > ref3a && Number(v3b.run) === Number(v3a.run); + const pass = paidOnce && runs === 2 && cache >= 2 && contextRefused && invalidRefusedTwiceNoVerify; + result.verdict = { paidOnce, verifiesRun: runs, cacheAnswers: cache, contextRefused, invalidRefusedTwiceNoVerify, pass }; + result.endedAt = new Date().toISOString(); + writeFileSync(OUT, JSON.stringify(result, null, 2)); + log(`RESULT ${pass ? 'PASS' : 'FAIL'}: paid once=${paidOnce}; SP1 verifies run=${runs} (expect 2); cache answers=${cache} (expect >= 2); context refusal read=${contextRefused}; invalid refused again without a verify=${invalidRefusedTwiceNoVerify}; ${OUT}`); + await stopAll(); + process.exit(pass ? 0 : 1); +} catch (e) { + log(`ERROR: ${e.message}`); + result.error = e.message; result.endedAt = new Date().toISOString(); + try { writeFileSync(OUT, JSON.stringify(result, null, 2)); } catch { } + await stopAll(); + process.exit(1); +} From 6ea01336a36f68512a188972a225f38b0e77cd1d Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 20:49:20 +0000 Subject: [PATCH 04/13] V6-07: one floor patch; the canonical copy carried no grow_for_main and built a server that overflows its key buffer at prove proving/prover-floor/sp1-gpu-6.8.1-floor.patch and tools/fleet/floor.patch are now byte-equal to tools/fleet/floor-v5.patch (the complete patch: the key buffer grown for the main traces, the pool release threshold, the stage hook on a panic). Measured 8 October 2026 20:5x UK: the server built from the old canonical copy (floor-build.py's input; sha 75d0b4be) panicked at sp1-gpu/crates/jagged_tracegen/src/lib.rs:240 "range end index 37428736 out of range for slice of length 36700160" on the first recursion prove (the RTX 4060, chain mode); the server built from floor-v5 (db37c38b) proves. box-setup.sh re-pinned to the one sha. Co-Authored-By: Claude Fable 5.1 --- .../prover-floor/sp1-gpu-6.8.1-floor.patch | 181 +++++++++++++++++- tools/fleet/box-setup.sh | 2 +- tools/fleet/floor.patch | 80 ++++++++ 3 files changed, 259 insertions(+), 4 deletions(-) diff --git a/proving/prover-floor/sp1-gpu-6.8.1-floor.patch b/proving/prover-floor/sp1-gpu-6.8.1-floor.patch index dcc6fbad4..c3a8ec532 100644 --- a/proving/prover-floor/sp1-gpu-6.8.1-floor.patch +++ b/proving/prover-floor/sp1-gpu-6.8.1-floor.patch @@ -1,5 +1,26 @@ +diff --git a/sp1-gpu/crates/cuda/src/task.rs b/sp1-gpu/crates/cuda/src/task.rs +index a503a86..813016b 100644 +--- a/sp1-gpu/crates/cuda/src/task.rs ++++ b/sp1-gpu/crates/cuda/src/task.rs +@@ -149,7 +149,15 @@ pub enum GlobalTaskPoolBuildError { + + impl TaskPoolBuilder { + pub fn new() -> Self { +- Self { capacity: None, device: CudaDevice(0), mem_release_threshold: u64::MAX } ++ // Igneum prover-floor patch: upstream keeps every freed device allocation in the pool for the process's ++ // life (threshold u64::MAX), so the prover holds its high-water mark between shards on a card it shares ++ // with a miner. `SP1_GPU_MEM_RELEASE_THRESHOLD=` sets the pool's release threshold (0 returns ++ // freed memory to the driver at once); unset, upstream's behaviour. ++ let mem_release_threshold = std::env::var("SP1_GPU_MEM_RELEASE_THRESHOLD") ++ .ok() ++ .and_then(|s| s.parse::().ok()) ++ .unwrap_or(u64::MAX); ++ Self { capacity: None, device: CudaDevice(0), mem_release_threshold } + } + + pub fn num_tasks(mut self, num_tasks: usize) -> Self { diff --git a/sp1-gpu/crates/jagged_tracegen/src/lib.rs b/sp1-gpu/crates/jagged_tracegen/src/lib.rs -index 579f70a..09e73e8 100644 +index 579f70a..2264044 100644 --- a/sp1-gpu/crates/jagged_tracegen/src/lib.rs +++ b/sp1-gpu/crates/jagged_tracegen/src/lib.rs @@ -481,6 +481,33 @@ async fn device_preprocessed_tracegen>( @@ -64,7 +85,81 @@ index 579f70a..09e73e8 100644 log_stacking_height, max_log_row_count, backend, -@@ -984,9 +1021,15 @@ pub async fn full_tracegen>( +@@ -906,6 +943,11 @@ pub async fn main_tracegen, A: CudaTracegenAir>( + + log_chip_stats(machine, &chip_set, &traces); + ++ // Igneum prover-floor patch: the key's buffer is sized to its preprocessed traces at setup (upstream sized it ++ // for a whole shard), so grow it here to what this shard needs before the main traces are appended: a bigger ++ // dense buffer and column index, the preprocessed region copied device to device, swapped into the key. ++ grow_for_main(&mut jagged_traces.preprocessed_traces, &traces, log_stacking_height, backend); ++ + copy_main_jagged_traces( + traces, + &mut jagged_traces.preprocessed_traces, +@@ -918,6 +960,61 @@ pub async fn main_tracegen, A: CudaTracegenAir>( + (public_values, chip_set, permit) + } + ++/// Igneum prover-floor patch: see `main_tracegen`. The need is the preprocessed phase as laid out (its padded ++/// end, `preprocessed_offset`) plus the main traces padded to the stacking height plus one stacking height of ++/// slack; a buffer at least that big is left alone. The process aborts, loudly, if the copy cannot be made, ++/// because a panic inside a prover task is what left sweep 2 hanging on the client's socket. ++fn grow_for_main( ++ jagged: &mut JaggedTraceMle, ++ main_traces: &BTreeMap>, ++ log_stacking_height: u32, ++ backend: &TaskScope, ++) { ++ let pre_end = jagged.dense().preprocessed_offset; ++ let needed = pre_end ++ + padded_trace_elements(main_traces, log_stacking_height) ++ + (1 << log_stacking_height); ++ let have = jagged.dense().dense.capacity(); ++ if have >= needed { ++ return; ++ } ++ let mut new_dense: Buffer = Buffer::with_capacity_in(needed, backend.clone()); ++ let mut new_col_index: Buffer = ++ Buffer::with_capacity_in(needed >> 1, backend.clone()); ++ unsafe { ++ new_dense.assume_init(); ++ new_col_index.assume_init(); ++ } ++ { ++ let JaggedMle { dense_data, col_index, .. } = &mut **jagged; ++ let src_dense: &Slice<_, _> = &dense_data.dense[..pre_end]; ++ let dst_dense: &mut Slice<_, _> = &mut new_dense[..pre_end]; ++ let src_col: &Slice<_, _> = &col_index[..pre_end >> 1]; ++ let dst_col: &mut Slice<_, _> = &mut new_col_index[..pre_end >> 1]; ++ unsafe { ++ if dst_dense.copy_from_slice(src_dense, backend).is_err() ++ || dst_col.copy_from_slice(src_col, backend).is_err() ++ { ++ eprintln!("FLOOR grow FAILED: could not copy the preprocessed region ({pre_end} elements) into the grown buffer ({needed} elements); aborting instead of hanging"); ++ std::process::abort(); ++ } ++ } ++ } ++ if std::env::var("SP1_GPU_FLOOR_LOG").is_ok() { ++ eprintln!( ++ "FLOOR grow key buffer {have} -> {needed} elements (preprocessed {pre_end}, {} bytes)", ++ needed * 6 ++ ); ++ } ++ let JaggedMle { dense_data, col_index, .. } = &mut **jagged; ++ dense_data.dense = new_dense; ++ *col_index = new_col_index; ++ unsafe { ++ dense_data.dense.set_len(pre_end); ++ col_index.set_len(pre_end >> 1); ++ } ++} ++ + #[allow(clippy::too_many_arguments)] + pub async fn main_tracegen_permit, A: CudaTracegenAir>( + machine: &Machine, +@@ -984,9 +1081,15 @@ pub async fn full_tracegen>( log_chip_stats(machine, &chip_set, &main_traces); @@ -81,7 +176,7 @@ index 579f70a..09e73e8 100644 log_stacking_height, max_log_row_count, backend, -@@ -1002,6 +1045,18 @@ pub async fn full_tracegen>( +@@ -1002,6 +1105,18 @@ pub async fn full_tracegen>( ) .await; @@ -251,6 +346,86 @@ index 5dccd9d..574d4fa 100644 .await, recursion_verifier, ) +diff --git a/sp1-gpu/crates/server/src/main.rs b/sp1-gpu/crates/server/src/main.rs +index 65e94f5..498357c 100644 +--- a/sp1-gpu/crates/server/src/main.rs ++++ b/sp1-gpu/crates/server/src/main.rs +@@ -17,9 +17,55 @@ struct Args { + version: bool, + } + ++/// Igneum prover-floor patch (6 October 2026, the GPU fleet's finding): a panic inside a prover task (an allocation ++/// the card cannot meet, `cudaMallocAsync` failing and `Buffer::with_capacity_in` panicking in a tokio worker) left ++/// the request's future waiting for ever, the client on its socket, the card at 0% for the 15 minutes until someone ++/// killed it (the 8 GB and 10 GB cards at threshold 2^27). The server must fail the shard instead: this hook names ++/// the stage from the panic's location and exits, so the client's proof fails at once with the server's last line. ++fn stage_of(location: &str) -> &'static str { ++ let l = location.to_ascii_lowercase(); ++ if l.contains("jagged_tracegen") || l.contains("/tracegen") { ++ "trace generation" ++ } else if l.contains("commit") || l.contains("basefold") || l.contains("merkle") { ++ "the commit (codewords and Merkle trees)" ++ } else if l.contains("logup_gkr") { ++ "LogUp GKR" ++ } else if l.contains("zerocheck") { ++ "the zerocheck" ++ } else if l.contains("jagged") { ++ "the jagged sumcheck" ++ } else if l.contains("prover_components") || l.contains("recursion") || l.contains("sp1-prover") || l.contains("sp1_prover") { ++ "the recursion (compression)" ++ } else if l.contains("cuda") || l.contains("slop") { ++ "a device allocation" ++ } else { ++ "the prover" ++ } ++} ++ ++fn install_abort_on_panic() { ++ std::panic::set_hook(Box::new(|info| { ++ let location = info.location().map(|l| format!("{}:{}", l.file(), l.line())).unwrap_or_else(|| "unknown".into()); ++ let message = info ++ .payload() ++ .downcast_ref::<&str>() ++ .map(|s| s.to_string()) ++ .or_else(|| info.payload().downcast_ref::().cloned()) ++ .unwrap_or_default(); ++ let oom = message.to_ascii_lowercase().contains("alloc") || message.contains("MemoryAllocation") || message.contains("OUT_OF_MEMORY"); ++ eprintln!( ++ "FLOOR abort: {} failed at {location}: {message}{}; the server exits so the client's proof fails instead of waiting", ++ stage_of(&location), ++ if oom { " (the card's memory could not meet an allocation: lower SP1_GPU_ELEMENT_THRESHOLD one notch)" } else { "" } ++ ); ++ std::process::exit(70); ++ })); ++} ++ + #[tokio::main] + #[allow(clippy::print_stdout)] + async fn main() { ++ install_abort_on_panic(); + tracing_subscriber::fmt::init(); + + let args = Args::parse(); +@@ -40,3 +86,19 @@ async fn main() { + eprintln!("Error running server: {e}"); + } + } ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::stage_of; ++ ++ #[test] ++ fn the_stage_is_named_from_the_panic_location() { ++ assert_eq!(stage_of("sp1-gpu/crates/jagged_tracegen/src/lib.rs:240"), "trace generation"); ++ assert_eq!(stage_of("sp1-gpu/crates/basefold/src/fri.rs:97"), "the commit (codewords and Merkle trees)"); ++ assert_eq!(stage_of("sp1-gpu/crates/logup_gkr/src/tracegen.rs:72"), "LogUp GKR"); ++ assert_eq!(stage_of("sp1-gpu/crates/zerocheck/src/prover.rs:1163"), "the zerocheck"); ++ assert_eq!(stage_of("sp1-gpu/crates/prover_components/src/builder.rs:70"), "the recursion (compression)"); ++ assert_eq!(stage_of("sp1-gpu/crates/cuda/src/stream.rs:330"), "a device allocation"); ++ assert_eq!(stage_of("somewhere/else.rs:1"), "the prover"); ++ } ++} diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs index 4035f1f..0d0d907 100644 --- a/sp1-gpu/crates/server/src/server.rs diff --git a/tools/fleet/box-setup.sh b/tools/fleet/box-setup.sh index 30e8e308e..bec5c23b1 100755 --- a/tools/fleet/box-setup.sh +++ b/tools/fleet/box-setup.sh @@ -22,7 +22,7 @@ fail() { echo "RESULT setup_failed $1 $(stamp)"; exit 2; } ARCHS="${ARCHS:-86,89,120}"; LABEL="${LABEL:-box}"; WALLET="${WALLET:-}" PKG_URL=https://dl.igneum.network/dl/public/igneum-hive-0.3.12.tar.gz PKG_SHA=7972af92e7cd9a032303eca4d95b533f53e0e68d1b9cae5bfe406a5b7c30a454 -PATCH_SHA=f53871d92e27b743b82545c03bfa6ef21cf799f43fae6138ad4136637f3ce2c2 +PATCH_SHA=9098c3e5979d057031f43588668201d1f7a53ab17e980c1eaf896cd5e6f38575 SEED=188.245.5.161:26611 echo "RESULT start $(stamp) label=$LABEL archs=$ARCHS host=$(hostname) nproc=$(nproc) ram_gb=$(( $(awk '/MemTotal/{print $2}' /proc/meminfo) / 1048576 )) disk_avail=$(df -BG /root | awk 'NR==2{print $4}')" echo "RESULT gpu $(nvidia-smi --query-gpu=name,memory.total,driver_version,pci.bus_id,power.limit,power.min_limit,power.max_limit,clocks.max.sm --format=csv,noheader 2>&1 | tr '\n' ';')" diff --git a/tools/fleet/floor.patch b/tools/fleet/floor.patch index 3d01b9ee5..c3a8ec532 100644 --- a/tools/fleet/floor.patch +++ b/tools/fleet/floor.patch @@ -346,6 +346,86 @@ index 5dccd9d..574d4fa 100644 .await, recursion_verifier, ) +diff --git a/sp1-gpu/crates/server/src/main.rs b/sp1-gpu/crates/server/src/main.rs +index 65e94f5..498357c 100644 +--- a/sp1-gpu/crates/server/src/main.rs ++++ b/sp1-gpu/crates/server/src/main.rs +@@ -17,9 +17,55 @@ struct Args { + version: bool, + } + ++/// Igneum prover-floor patch (6 October 2026, the GPU fleet's finding): a panic inside a prover task (an allocation ++/// the card cannot meet, `cudaMallocAsync` failing and `Buffer::with_capacity_in` panicking in a tokio worker) left ++/// the request's future waiting for ever, the client on its socket, the card at 0% for the 15 minutes until someone ++/// killed it (the 8 GB and 10 GB cards at threshold 2^27). The server must fail the shard instead: this hook names ++/// the stage from the panic's location and exits, so the client's proof fails at once with the server's last line. ++fn stage_of(location: &str) -> &'static str { ++ let l = location.to_ascii_lowercase(); ++ if l.contains("jagged_tracegen") || l.contains("/tracegen") { ++ "trace generation" ++ } else if l.contains("commit") || l.contains("basefold") || l.contains("merkle") { ++ "the commit (codewords and Merkle trees)" ++ } else if l.contains("logup_gkr") { ++ "LogUp GKR" ++ } else if l.contains("zerocheck") { ++ "the zerocheck" ++ } else if l.contains("jagged") { ++ "the jagged sumcheck" ++ } else if l.contains("prover_components") || l.contains("recursion") || l.contains("sp1-prover") || l.contains("sp1_prover") { ++ "the recursion (compression)" ++ } else if l.contains("cuda") || l.contains("slop") { ++ "a device allocation" ++ } else { ++ "the prover" ++ } ++} ++ ++fn install_abort_on_panic() { ++ std::panic::set_hook(Box::new(|info| { ++ let location = info.location().map(|l| format!("{}:{}", l.file(), l.line())).unwrap_or_else(|| "unknown".into()); ++ let message = info ++ .payload() ++ .downcast_ref::<&str>() ++ .map(|s| s.to_string()) ++ .or_else(|| info.payload().downcast_ref::().cloned()) ++ .unwrap_or_default(); ++ let oom = message.to_ascii_lowercase().contains("alloc") || message.contains("MemoryAllocation") || message.contains("OUT_OF_MEMORY"); ++ eprintln!( ++ "FLOOR abort: {} failed at {location}: {message}{}; the server exits so the client's proof fails instead of waiting", ++ stage_of(&location), ++ if oom { " (the card's memory could not meet an allocation: lower SP1_GPU_ELEMENT_THRESHOLD one notch)" } else { "" } ++ ); ++ std::process::exit(70); ++ })); ++} ++ + #[tokio::main] + #[allow(clippy::print_stdout)] + async fn main() { ++ install_abort_on_panic(); + tracing_subscriber::fmt::init(); + + let args = Args::parse(); +@@ -40,3 +86,19 @@ async fn main() { + eprintln!("Error running server: {e}"); + } + } ++ ++#[cfg(test)] ++mod floor_tests { ++ use super::stage_of; ++ ++ #[test] ++ fn the_stage_is_named_from_the_panic_location() { ++ assert_eq!(stage_of("sp1-gpu/crates/jagged_tracegen/src/lib.rs:240"), "trace generation"); ++ assert_eq!(stage_of("sp1-gpu/crates/basefold/src/fri.rs:97"), "the commit (codewords and Merkle trees)"); ++ assert_eq!(stage_of("sp1-gpu/crates/logup_gkr/src/tracegen.rs:72"), "LogUp GKR"); ++ assert_eq!(stage_of("sp1-gpu/crates/zerocheck/src/prover.rs:1163"), "the zerocheck"); ++ assert_eq!(stage_of("sp1-gpu/crates/prover_components/src/builder.rs:70"), "the recursion (compression)"); ++ assert_eq!(stage_of("sp1-gpu/crates/cuda/src/stream.rs:330"), "a device allocation"); ++ assert_eq!(stage_of("somewhere/else.rs:1"), "the prover"); ++ } ++} diff --git a/sp1-gpu/crates/server/src/server.rs b/sp1-gpu/crates/server/src/server.rs index 4035f1f..0d0d907 100644 --- a/sp1-gpu/crates/server/src/server.rs From a556c3ad53d0eaf2e70ed769884776e4c8988c19 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 21:20:21 +0000 Subject: [PATCH 05/13] V6-07: the aggregate and chain floors at 8,500 MiB from the measured aggregation peak; the rows and their consequences docs/analysis/floor-memory-profile-2026-10-08.md: the 3060 and 4060 rows on the default, unoverridden job path (the served 0317 host on the V6-07 server: 3060 shard 11.4 s at 7,601 MiB, the whole chain 27.4 s at an 8,306 MiB aggregation peak, both VERIFIED; 4060 shard 16.3 s at 7,504 MiB VERIFIED, the chain aborts at the aggregation's 486 MiB allocation on 176 MiB free), the host's profile and refusal rows (exit 78 under the floor and beside the miner on both cards, the miner unharmed), the recursion constant isolated (no change of the device peak), the known-failed first server, the master-host 0-bytes fault with its NOT RUN rows, and the consequences per card tier. memory_profile.rs: aggregate and chain floors 8,500 MiB (an 8 GB card proves shards and is refused the chain before setup), tests re-pinned; suite 7 passed on box 3. Co-Authored-By: Claude Fable 5.1 --- .../floor-memory-profile-2026-10-08.md | 121 ++++++++++++++++++ .../igneum-prove/host/src/memory_profile.rs | 26 ++-- 2 files changed, 138 insertions(+), 9 deletions(-) create mode 100644 docs/analysis/floor-memory-profile-2026-10-08.md diff --git a/docs/analysis/floor-memory-profile-2026-10-08.md b/docs/analysis/floor-memory-profile-2026-10-08.md new file mode 100644 index 000000000..baab87dd9 --- /dev/null +++ b/docs/analysis/floor-memory-profile-2026-10-08.md @@ -0,0 +1,121 @@ +# The floor's memory profile (V6-07), 8 October 2026 + +Master review R1, residual V6-07 (`docs/plans/igneum-2.0-master/evidence/04_full_system/IGNEUM_V6_Full_System_Review.md`, pages 210 to 212): the floor patch picked its limits from the card's total VRAM, not from what was free; its small-card element threshold defaulted to 2^27 where every passing row had set 2^26 by hand; its recursion-allocation budget returned one constant in both branches. The order: a pinned memory profile per workload read from free memory, with the app lane's device coordinator, and the 3060 and 4060 rows rerun on the default, unoverridden job path. This document is the record: the code (section 1), the table (section 2), the rows as the pods wrote them (section 3), the consequences per card tier beside each number (section 4), the coordinator hook and what is next (section 5). + +Branch `v607-floor-memory` off the box mirror master cef5234b5. Code commits e6abe8c2c (the patch and the host profile), 573dad0ad (the lesser of grant and free), 8c1fb754c (one floor patch), then the floors re-pinned from the rows (the commit carrying this document). Clocks below are UTC as the pods wrote them; UK time is one hour later. + +## 1. What changed + +| Item | Where | Before | After | +|---|---|---|---| +| The limits' source | `proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, `builder.rs` `gpu_memory_gb()` (the fleet's copies `tools/fleet/floor.patch`, `floor-v5.patch` carry the same hunk; `box-setup.sh` re-pinned) | `cuda_memory_info().1` (the total), +4 as upstream | `cuda_memory_info().0` (the FREE memory as the driver reports it), +4, read ONCE per process in a `OnceLock` so the core opts and the recursion prover, built at different moments, sit on one tier; `SP1_GPU_MEMORY_BUDGET_GB` (the host's lease) still overrides | +| The small-card element threshold | same, `element_threshold_for_budget` | `1 << 27` under the 18 tier | `1 << 26`, the value the passing rows used (section 2 cites them) | +| The recursion-allocation budget | same, `recursion_trace_allocation_for_budget` | `RECURSION_TRACE_ALLOCATION` in both branches | upstream's 2^27 on the 24 GB tier and above; `RECURSION_TRACE_ALLOCATION_SMALL` = 2^26 + 2^25 = 100,663,296 under it (a recursion key or shard uses 90,177,536 elements, `docs/analysis/prover-floor.md` sweep 1; the patch's `floor_capacity` sizes the device buffer to the need, so the constant caps the buffer and shrinks the four pinned host copies per prover); `floor_tests::recursion_branches_differ` pins that the two branches differ and that the small one clears the measured use with a stacking height of slack; `floor_tests::small_tier_is_two_to_the_26` pins the tiers | +| The FLOOR opts line | same, `local_gpu_opts` | `gpu_memory_gb`, thresholds | adds `free_mib` and `total_mib` so a log names what the server read | +| The pinned profile per workload | `proving/igneum-prove/host/src/memory_profile.rs` (new), wired in `main.rs` before the SP1 client is built | nothing: the server guessed from the total; the fleet set `SP1_GPU_ELEMENT_THRESHOLD` by hand | one table (section 2) with tests pinning every value; the row is chosen from the engine's lease or, with no engine, from `nvidia-smi memory.free` on the device, and handed to the server by environment (`SP1_GPU_ELEMENT_THRESHOLD`, `SP1_GPU_RECURSION_TRACE_ALLOCATION`, `SP1_GPU_MEMORY_BUDGET_GB`); a hand override already in the environment is kept and named; under the workload's floor the host prints one line and exits 78 before any setup | +| The fleet's default path | `tools/fleet/box-prover.py` | `SP1_GPU_ELEMENT_THRESHOLD` from `THRESHOLD`, else the server's total-VRAM guess | `IGNEUM_PROVE_WORKLOAD=chain` and `IGNEUM_PROVE_DEVICE` on the host run, no threshold unless `THRESHOLD` is set by hand; a refusal (exit 78) closes the segment as cancelled/memory | + +Tests: `cargo test -p igneum-prove-host memory_profile` (box 3, section 6) and the patch's `floor_tests` (run on the build pod after the server build, section 6). + +## 2. The table + +The host's `TIERS` (largest first; the free-memory lines are the card tiers as the floor patch reads them, free GiB rounded up plus 4 as upstream computed its tiers from the total): + +| Tier | Free memory at start | Element threshold | Recursion trace allocation | Who lands here | +|---|---|---|---|---| +| full | 26,624 MiB and up | 2^28 + 2^27 = 402,653,184 | 2^27 = 134,217,728 | a 32 GB card alone | +| 24gb | 20,480 MiB and up | 285,212,672 (upstream's 24 GB figure) | 2^27 | a 24 GB card alone | +| 16gb | 14,336 MiB and up | 2^27 + 2^26 = 201,326,592 (the patch's own figure, unmeasured on this fixture) | 2^26 + 2^25 = 100,663,296 | a 16 GB card alone | +| small | under 14,336 MiB | 2^26 = 67,108,864 | 100,663,296 | a 12 GB or 8 GB card alone; any card beside a miner's resident set | + +Floors per workload (the least free memory the host runs in; under it the refusal, exit 78): shard 7,700 MiB, aggregate 8,500 MiB, chain 8,500 MiB. The shard floor is the small tier's measured peak (7,525 MiB on the 3060, 7,532 on the 4060, section 3) plus headroom for the driver's own context. The aggregate and chain floors are 8,500 MiB: the aggregation is the peak of a chain run, 8,306 MiB on the RTX 3060 (section 3, D2), and the RTX 4060 (7,807 MiB free) aborted at it, so an 8 GB card is refused the chain and the aggregation before any setup and proves shards only. + +The rows cited for 2^26 on small cards: `docs/analysis/prover-tiers-real-cards.md` (6 October 2026, the matrix's `alone-comp-26-v1` point, compressed, verified: RTX 3060 7.4 GB 14.4 s; RTX 3080 8.0 GB 7.1 s; RTX 4060 7.4 GB 18.4 s; RTX 4060 Ti 8 GB 7.6 GB 9.6 s; RTX 4060 Ti 16 GB 7.8 GB 11.6 s; RTX 4070 7.6 GB 12.1 s; RTX 5070 7.6 GB 4.8 s) and `docs/analysis/class-v6/coexist-rows.md` (8 October 2026, `proof_alone` at `SP1_GPU_ELEMENT_THRESHOLD=67108864`: RTX 3060 peak 7,525 MiB 13.2 s verified; RTX 4060 peak 7,532 MiB 8.2 s verified). No small-card row ever passed at 2^27 on this host path; the 10 GB tier's 2^27 reading of 7 October (8,642 MiB alone on a 3080) ran on the segment host and is not this path. + +Every value above is pinned by `memory_profile::tests::table_is_pinned`, `tiers_from_free_memory`, `refuses_under_the_floor`, `recursion_branches_differ`, `workload_from_mode` and `env_for_the_server`; a change of the table is a change of the profile and needs its rows. + +## 3. The rows, rerun on the default, unoverridden job path + +Every row is a RESULT line as the pod wrote it (the raw run logs, host logs and 1 Hz `nvidia-smi` samples are kept under the lane's scratch `v607/rows/