From 4f545000247af7763e6f3d64b01fe2fdf9c2e059 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:53:00 +0000 Subject: [PATCH] igneum-pow at the class-v5 lane's 764a41d9 (StateStream, StateLeaves: the stored-state side the 0.3.24 node line's kaspa-pow calls); the testnet re-cut on that line builds against it Co-Authored-By: Claude Fable 5.1 --- igneum-pow/src/accept.rs | 301 +++++++++++++++++++++++++++++++++--- igneum-pow/src/blake2b.rs | 152 ++++++++++++++++++ igneum-pow/src/emit.rs | 178 +++++++++++++++++---- igneum-pow/src/generator.rs | 272 ++++++++++++++++++++++++++++++-- igneum-pow/src/lib.rs | 5 +- igneum-pow/src/main.rs | 35 ++++- igneum-pow/src/memhard.rs | 200 +++++++++++++++++++++--- igneum-pow/src/packcheck.rs | 14 +- igneum-pow/src/state.rs | 287 ++++++++++++++++++++++++++++++++++ igneum-pow/src/verify.rs | 36 +++++ igneum-pow/tests/derive.rs | 20 +-- igneum-pow/tests/mixer.rs | 2 +- igneum-pow/tests/packs.rs | 102 +++++++++++- 13 files changed, 1499 insertions(+), 105 deletions(-) create mode 100644 igneum-pow/src/blake2b.rs create mode 100644 igneum-pow/src/state.rs diff --git a/igneum-pow/src/accept.rs b/igneum-pow/src/accept.rs index 5777be4b6..9937dba06 100644 --- a/igneum-pow/src/accept.rs +++ b/igneum-pow/src/accept.rs @@ -9,6 +9,7 @@ //! | (a) static | for every `load`, some instruction between the previous `load` from the same source register and this one, in cyclic order over the 64 instructions, writes that register | //! | (b) static | every register `r0..r7` is the destination of at least one `add`, `sub`, `xor`, `mad`, `shfl` or `load` | //! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) | +//! | (c''') class v5 | the per-site distinct-index floor of (c'') raised to [`MIN_DISTINCT_RATIO_V5`] (0.995) on the same 2^20-evaluation run, keyed on the state flag; the ratio at the 0.98 floor is still named (c'') first (`docs/design/class-v5-stored-state.md` section 14) | //! //! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and //! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by @@ -40,6 +41,21 @@ pub const MIN_DISTINCT_RATIO_V4: f64 = 0.98; /// Kept for the record and the driver, not wired: the most repeated source value per site over the (c) units' /// 16,384 evaluations (a uniform site repeats a value 2 or 3 times; the finding's bands sit under the ratio instead). pub const MAX_SOURCE_REPEAT_V4: u32 = 8; +/// (c'''), class v5 (`docs/design/class-v5-stored-state.md` section 14): the per-site distinct-index floor of the +/// ratio pass raised from sub-version 3's 0.98 to 0.995. The residual class the in-house pass attributed (adv-accept, +/// 7 October 2026: seed 100767 of the f8 label space, program 9d68e6286fc817d4; its site 6 reads the multiples of +/// 2^19 through `rotl(x * stride, rot)` of a `mad` value that is near zero in a value class, 3.35 percent of its live +/// reads on the top 0.1 percent of items, 1,677 reads of index 0 in 10^6 nonces) reads 0.9919 on the closed form and +/// 0.9920 on the live day at the rule's own 2^20 sample: 0.012 above the 0.98 floor, 0.003 under this one, where the +/// window model's spread is 0.0001 (the collision count is near Poisson with mean n^2 / 2W = 8,192 on the quarter +/// window). The census behind the number (`v5_hot_census`, 4,600 seeds of the f8 label space, 7 October 2026, 21:03 +/// UK): the class v4 sub-version 3 accepted programs' minimum site ratio reads min 0.9807, p1 0.9906, p5 0.9962, +/// median 0.9999; 112 of 4,600 (2.435 percent) sit under 0.995 and take one or more further attempts under class v5 +/// (the attempts mean 2.174 to 2.248, the tail unchanged at 24); 0 class v5 accepted programs sit under the floor. +/// 0.99 would pass the exemplar (43 under it); 0.995 is the lowest round floor that refuses it with the model's +/// spread under it. The cheapest fix that reaches the class (main's order of 7 October 2026): no new sample, the +/// verdict at the attempt itself, no 2^24 run, and a per-site hot-item test at that sample is the same statistic. +pub const MIN_DISTINCT_RATIO_V5: f64 = 0.995; /// Hashes the dynamic test evaluates: 2,048. pub const ACCEPT_HASHES: usize = ACCEPT_UNITS * LANES; /// Domain tag of the base-nonce stream. @@ -88,6 +104,11 @@ pub enum Reject { /// `evaluations`, `ratio_milli` / 1000 of a uniform source on its window, under the floor: a low-entropy index band /// (F8's p23, p18, p19, p15, p56). LowEntropySite { site: u8, distinct: u32, evaluations: u32, ratio_milli: u32 }, + /// (c'''), class v5: the load at `site` read `distinct` distinct dataset word indices over `evaluations`, + /// `ratio_milli` / 1000 of the window model, under [`MIN_DISTINCT_RATIO_V5`] (and at or above sub-version 3's + /// floor, else (c'') names it); `top_index` was its most read index, `top_count` times: a hot set from a + /// value-level constant upstream of the site that neither the lineage rule nor the 0.98 floor reaches (seed 100767). + HotItemSite { site: u8, distinct: u32, evaluations: u32, ratio_milli: u32, top_index: u32, top_count: u32 }, /// (c): output bit `bit` was set in `ones` of 2,048 hashes. OutputBias { bit: u8, ones: u32 }, /// (c): the distinct-address sum was `sum`. @@ -108,6 +129,7 @@ impl std::fmt::Display for Reject { Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"), Reject::UnfreshLoadSource { instr, reg } => write!(f, "(a') load at {instr} reads r{reg}, not fresh by dataflow in the loop's steady state (class v4 sub-version 2)"), Reject::RepeatedSource { site, value, count } => write!(f, "(c'') load site {site} read the value {value:#010x} in {count} of 16384 evaluations (limit {})", MAX_SOURCE_REPEAT_V4 - 1), + Reject::HotItemSite { site, distinct, evaluations, ratio_milli, top_index, top_count } => write!(f, "(c''') load site {site} read {distinct} distinct word indices over {evaluations} evaluations, {ratio_milli}/1000 of the window model, under the class v5 floor (most read: index {top_index:#x}, {top_count} times): a hot set from a value-level constant upstream"), Reject::LowEntropySite { site, distinct, evaluations, ratio_milli } => write!(f, "(c'') load site {site} read {distinct} distinct word indices over {evaluations} evaluations, {}.{:03} of a uniform source on its window (floor {MIN_DISTINCT_RATIO_V4} at 2^20)", ratio_milli / 1000, ratio_milli % 1000), Reject::SaturatedSource { site, count } => write!(f, "(c') load site {site} read a saturated source value in {count} of 16384 evaluations (limit 163)"), Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"), @@ -195,31 +217,37 @@ pub fn check_distinct_indices_v4(p: &Program) -> Result<(), Reject> { /// window (`N - N^2 / 2W`, the window `2^28 >> min(win, 2)` words of the closed-form dataset), `Err` at the first /// site under `floor`, else the minimum ratio and its site. pub fn distinct_ratio_pass(p: &Program, units: usize, floor: f64) -> Result<(f64, usize), Reject> { - let n = (units * LANES * ITERATIONS) as f64; - let d = distinct_indices_v4(p, units)?; - let mut min = (f64::MAX, 0usize); - let mut site = 0usize; - for i in &p.instrs { - if !i.op.is_load() { - continue; - } - let wsize = ((1u64 << ACCEPT_DATASET_LOG2) >> (i.win as u64).min(2)) as f64; - let ratio = d[site] as f64 / (n - n * n / (2.0 * wsize)); - if ratio < floor { - return Err(Reject::LowEntropySite { site: site as u8, distinct: d[site], evaluations: n as u32, ratio_milli: (ratio * 1000.0) as u32 }); - } - if ratio < min.0 { - min = (ratio, site); - } - site += 1; - } - Ok(min) + let stats = site_index_stats(p, units)?; + distinct_ratio_on(p, &stats, units, floor) } /// The distinct dataset word indices every load site reads over `units` units of the seed's acceptance stream on /// the closed-form words (the sample caps the count near `units x 32 x 8`, so a site's index entropy is read only /// below about log2 of that). pub fn distinct_indices_v4(p: &Program, units: usize) -> Result, Reject> { + Ok(site_index_stats(p, units)?.into_iter().map(|s| s.distinct).collect()) +} + +/// What one load site's word indices over the ratio pass look like: the distinct count (c'' and c'''), the sum of +/// `C(run, 2)` over the indices (the colliding pairs), and the most read index with its count (the diagnostic the +/// reject names). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct SiteIndexStats { + pub distinct: u32, + pub pairs: u64, + pub top_index: u32, + pub top_count: u32, +} + +/// The window of a load site in words at the rule's dataset: `2^28 >> min(win, 2)`. +pub fn site_window_words(ins: &Instr) -> u64 { + (1u64 << ACCEPT_DATASET_LOG2) >> (ins.win as u64).min(2) +} + +/// One interpreter run over `units` units with every load site's word indices kept, then per site the sorted +/// run lengths: [`SiteIndexStats`] per site in load order. The ratio pass (c'') and the hot-item rule (c''') read +/// the same run, so class v5 pays the sort once. +pub fn site_index_stats(p: &Program, units: usize) -> Result, Reject> { let loads = p.loads_per_hash(); let sites = loads / ITERATIONS; let mut acc = Acc { @@ -239,17 +267,95 @@ pub fn distinct_indices_v4(p: &Program, units: usize) -> Result, Reject let mut out = Vec::with_capacity(sites); for ix in acc.indices.take().unwrap().iter_mut() { ix.sort_unstable(); - ix.dedup(); - out.push(ix.len() as u32); + let mut st = SiteIndexStats { distinct: 0, pairs: 0, top_index: 0, top_count: 0 }; + let mut i = 0; + while i < ix.len() { + let mut j = i + 1; + while j < ix.len() && ix[j] == ix[i] { + j += 1; + } + let run = (j - i) as u32; + st.distinct += 1; + st.pairs += (run as u64) * (run as u64 - 1) / 2; + if run > st.top_count { + st.top_count = run; + st.top_index = ix[i]; + } + i = j; + } + out.push(st); } Ok(out) } +/// (c'''), class v5: the raised floor on the ratio pass's sample, `Err` at the first load site under `floor`, else +/// the minimum ratio and its site. Pure in the program as (c'') is (the closed form, the seed's acceptance stream); +/// the state leaves move the dataset's words, not the indices a site reads (adv-accept read the exemplar's site 6 +/// at 0.9919 closed and 0.9920 on the live day). +pub fn hot_item_pass(p: &Program, stats: &[SiteIndexStats], units: usize, floor: f64) -> Result<(f64, usize), Reject> { + let n = (units * LANES * ITERATIONS) as f64; + let mut min = (f64::MAX, 0usize); + let mut site = 0usize; + for i in &p.instrs { + if !i.op.is_load() { + continue; + } + let st = &stats[site]; + let ratio = site_ratio(st.distinct, n, site_window_words(i)); + if ratio < floor { + return Err(Reject::HotItemSite { site: site as u8, distinct: st.distinct, evaluations: n as u32, ratio_milli: (ratio * 1000.0) as u32, top_index: st.top_index, top_count: st.top_count }); + } + if ratio < min.0 { + min = (ratio, site); + } + site += 1; + } + Ok(min) +} + +/// A site's distinct-index ratio against the window model `N - N^2 / 2W` (the (c'') form, kept so the census +/// numbers of sub-version 3 read on the same scale). +pub fn site_ratio(distinct: u32, n: f64, window: u64) -> f64 { + distinct as f64 / (n - n * n / (2.0 * window as f64)) +} + +/// The ratio floor (c'') on precomputed site stats: `Err` at the first site under `floor`, else the minimum ratio +/// and its site (the pass [`distinct_ratio_pass`] runs when it has no stats yet). +pub fn distinct_ratio_on(p: &Program, stats: &[SiteIndexStats], units: usize, floor: f64) -> Result<(f64, usize), Reject> { + let n = (units * LANES * ITERATIONS) as f64; + let mut min = (f64::MAX, 0usize); + let mut site = 0usize; + for i in &p.instrs { + if !i.op.is_load() { + continue; + } + let d = stats[site].distinct; + let ratio = site_ratio(d, n, site_window_words(i)); + if ratio < floor { + return Err(Reject::LowEntropySite { site: site as u8, distinct: d, evaluations: n as u32, ratio_milli: (ratio * 1000.0) as u32 }); + } + if ratio < min.0 { + min = (ratio, site); + } + site += 1; + } + Ok(min) +} + +/// (c'') then (c'''), class v5: one 2^20-evaluation run, the ratio floor of sub-version 3 and the hot-item rule on +/// it. The order is the module table's: a low-entropy band is named before a hot item on the same site. +pub fn check_indices_v5(p: &Program) -> Result<(), Reject> { + let stats = site_index_stats(p, ACCEPT_UNITS_DISTINCT_V4)?; + distinct_ratio_on(p, &stats, ACCEPT_UNITS_DISTINCT_V4, MIN_DISTINCT_RATIO_V4)?; + hot_item_pass(p, &stats, ACCEPT_UNITS_DISTINCT_V4, MIN_DISTINCT_RATIO_V5).map(|_| ()) +} + /// Whether `class` is the class v4 shape (the 256-instruction shadow block over the class v3 base, the pass count and /// the era set aside): the shape the sub-version 2 rules (a') and (c') apply to, on every draw path. pub fn is_class_v4_shape(class: &LoadClass) -> bool { matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. })) - && LoadClass { era: None, shadow: None, ..*class } == LoadClass { shadow: None, ..V4_CLASS } + // class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside + && LoadClass { era: None, shadow: None, state: false, ..*class } == LoadClass { shadow: None, ..V4_CLASS } } /// One pass of the dataflow freshness over the base program then the shadow block (the order of one iteration), @@ -600,8 +706,14 @@ pub fn check_dynamic(p: &Program) -> Result { if let Some((site, &count)) = acc.sat_source.iter().enumerate().find(|(_, &c)| c >= MAX_SATURATED) { return Err(Reject::SaturatedSource { site: site as u8, count }); } - // (c''), the ratio on the candidate that passed everything else (the draw's last and dearest test) - check_distinct_indices_v4(p)?; + // (c''), the ratio on the candidate that passed everything else (the draw's last and dearest test); class v5 + // reads (c''') the hot-item rule on the same run (`docs/design/class-v5-stored-state.md` section 14), keyed + // on the state flag so no class v4 verdict moves + if p.class.state { + check_indices_v5(p)?; + } else { + check_distinct_indices_v4(p)?; + } } let half = (ACCEPT_HASHES / 2) as u32; let mut bias_max = 0u32; @@ -630,6 +742,147 @@ mod tests { use crate::generator::{candidate, candidate_class, generate, generate_class, GeneratorConfig, generate_v1, LoadClass}; use crate::verify::{DatasetMode, DatasetSource}; + /// The f8 label space's chain draw (adv-accept's census space): epoch and era seed bytes for seed `k`, and the + /// candidate at `attempt` under `base` (the chain's class v4 sub-version 3 draw for `V4_CLASS`, the class v5 draw + /// for `V5_CLASS`), stamped as `generate_era` stamps it. + fn f8_seed(k: u32) -> (Vec, Vec) { + use crate::seed::seed_words_from_bytes; + let w = |s: String| -> Vec { seed_words_from_bytes(s.as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect() }; + (w(format!("igneum-attack-f8/program/{k}")), w(format!("igneum-attack-f8/era/{k}"))) + } + fn f8_label(epoch: &[u8]) -> String { + format!("igneum-epoch/{}", epoch.iter().map(|b| format!("{b:02x}")).collect::()) + } + fn f8_candidate(epoch: &[u8], era: &[u8], attempt: u32, base: LoadClass) -> Program { + use crate::generator::{era_generator_of, V3_ALLOWED}; + let mut p = candidate_class(&f8_label(epoch), epoch, attempt, LoadClass::era(base, era, &V3_ALLOWED)); + p.generator = era_generator_of(&base); + p.era_bytes = Some(era.to_vec()); + p + } + fn f8_draw(epoch: &[u8], era: &[u8], base: LoadClass) -> Program { + use crate::generator::{generate_era, V3_ALLOWED}; + generate_era(&f8_label(epoch), epoch, base, era, &V3_ALLOWED) + } + + /// (c''') known-failed first (`docs/design/class-v5-stored-state.md` section 14; main's order of 7 October + /// 2026): adv-accept's seed 100767 is the chain's class v4 sub-version 3 draw at attempt 2 (program + /// 9d68e6286fc817d4), passes every part of the sub-version 3 rule with its site 6 at 0.9919, and reads a hot set + /// live (3.35 percent of site 6's reads on the top 0.1 percent of items, the multiples of 2^19). The same shape + /// under class v5 is refused at attempt 2 by the raised floor, naming site 6 under 0.995, and the class v5 draw + /// moves past it; the class v4 verdict does not move. A clean seed (the genesis draw of both classes) clears + /// the floor with room. + #[test] + fn class_v5_hot_set_rule_known_failed_seed_100767() { + use crate::generator::V5_CLASS; + let (epoch, era) = f8_seed(100767); + let v4 = f8_candidate(&epoch, &era, 2, V4_CLASS); + assert_eq!(format!("{:016x}", v4.program_id()), "9d68e6286fc817d4", "the exemplar is the chain's sub-version 3 program"); + assert_eq!(v4.seed, [0xcc5466bf, 0x375211d6, 0xf22b6e63, 0x638d9f8b, 0x85903c90, 0x03605228, 0x5f97734b, 0x910fb493]); + assert!(check(&v4).is_ok(), "class v4 sub-version 3 accepts it: {:?}", check(&v4).err()); + let v4_stats = site_index_stats(&v4, ACCEPT_UNITS_DISTINCT_V4).unwrap(); + let (min, site) = distinct_ratio_on(&v4, &v4_stats, ACCEPT_UNITS_DISTINCT_V4, 0.0).unwrap(); + println!("seed 100767 class v4: min site {site} ratio {min:.4} (distinct {}, most read index {:#x} {} times)", v4_stats[site].distinct, v4_stats[site].top_index, v4_stats[site].top_count); + assert_eq!(site, 6); + assert!(min > 0.98 && min < 0.995, "the exemplar sits between the floors: {min:.4}"); + + let v5 = f8_candidate(&epoch, &era, 2, V5_CLASS); + assert_eq!(v5.instrs, v4.instrs, "the same base program under class v5"); + match check(&v5) { + Err(Reject::HotItemSite { site, ratio_milli, distinct, evaluations, top_index, top_count }) => { + println!("seed 100767 class v5 attempt 2: (c''') site {site} {distinct} of {evaluations}, ratio {}.{:03}, most read {top_index:#x} x{top_count}", ratio_milli / 1000, ratio_milli % 1000); + assert_eq!(site, 6); + assert!(ratio_milli >= 980 && ratio_milli < 995); + } + other => panic!("class v5 must refuse the exemplar at attempt 2 by (c'''), got {other:?}"), + } + let drawn = f8_draw(&epoch, &era, V5_CLASS); + println!("seed 100767 class v5 draw: attempt {} id {:016x}", drawn.attempt, drawn.program_id()); + assert!(drawn.attempt > 2, "the class v5 draw moves past the exemplar"); + assert!(check(&drawn).is_ok()); + let drawn_v4 = f8_draw(&epoch, &era, V4_CLASS); + assert_eq!(drawn_v4.attempt, 2, "the class v4 draw still lands on it"); + + for (name, p) in [("genesis v4", generate_class("igneum-genesis", V4_CLASS)), ("genesis v5", generate_class("igneum-genesis", V5_CLASS))] { + let st = site_index_stats(&p, ACCEPT_UNITS_DISTINCT_V4).unwrap(); + let (min, site) = distinct_ratio_on(&p, &st, ACCEPT_UNITS_DISTINCT_V4, 0.0).unwrap(); + println!("{name}: min site {site} ratio {min:.4}"); + assert!(min >= MIN_DISTINCT_RATIO_V5, "{name}: a clean seed clears the class v5 floor: {min:.4}"); + } + } + + /// The census behind the floor (run by hand on a box: `cargo test --release --lib -- --ignored v5_hot_census + /// --nocapture`, `IGNEUM_V5_CENSUS_SEEDS` seeds of the f8 label space from `IGNEUM_V5_CENSUS_FROM`, default + /// 4,600 from 0, `IGNEUM_V5_CENSUS_THREADS` threads, default 88): per seed the class v4 sub-version 3 draw's + /// attempt and its accepted program's minimum site ratio, the class v5 draw's attempt, and the counts under + /// candidate floors, so the clean rejection rate and the attempts histogram of the chosen floor are on record. + #[test] + #[ignore] + fn v5_hot_census() { + use crate::generator::V5_CLASS; + use std::sync::atomic::{AtomicU32, Ordering}; + let seeds: u32 = std::env::var("IGNEUM_V5_CENSUS_SEEDS").ok().and_then(|v| v.parse().ok()).unwrap_or(4600); + let from: u32 = std::env::var("IGNEUM_V5_CENSUS_FROM").ok().and_then(|v| v.parse().ok()).unwrap_or(0); + let threads: usize = std::env::var("IGNEUM_V5_CENSUS_THREADS").ok().and_then(|v| v.parse().ok()).unwrap_or(88); + let next = AtomicU32::new(from); + let t0 = std::time::Instant::now(); + // per seed: (k, v4 attempt, v4 min ratio, v4 min site, v5 attempt, v5 min ratio) + let rows: Vec<(u32, u32, f64, usize, u32, f64)> = std::thread::scope(|sc| { + let hs: Vec<_> = (0..threads).map(|_| sc.spawn(|| { + let mut out = Vec::new(); + loop { + let k = next.fetch_add(1, Ordering::Relaxed); + if k >= from + seeds { + break out; + } + let (epoch, era) = f8_seed(k); + let v4 = f8_draw(&epoch, &era, V4_CLASS); + let st = site_index_stats(&v4, ACCEPT_UNITS_DISTINCT_V4).unwrap(); + let (min4, site4) = distinct_ratio_on(&v4, &st, ACCEPT_UNITS_DISTINCT_V4, 0.0).unwrap(); + let v5 = f8_draw(&epoch, &era, V5_CLASS); + let st5 = site_index_stats(&v5, ACCEPT_UNITS_DISTINCT_V4).unwrap(); + let (min5, _) = distinct_ratio_on(&v5, &st5, ACCEPT_UNITS_DISTINCT_V4, 0.0).unwrap(); + out.push((k, v4.attempt, min4, site4, v5.attempt, min5)); + } + })).collect(); + let mut rows: Vec<_> = hs.into_iter().flat_map(|h| h.join().unwrap()).collect(); + rows.sort_by_key(|r| r.0); + rows + }); + let n = rows.len() as f64; + let mut sorted: Vec = rows.iter().map(|r| r.2).collect(); + sorted.sort_by(|a, b| a.partial_cmp(b).unwrap()); + let q = |p: f64| sorted[((p * (sorted.len() - 1) as f64).round() as usize).min(sorted.len() - 1)]; + println!("v5_hot_census: {} seeds {from}..{} in {:.0} s on {threads} threads", rows.len(), from + seeds, t0.elapsed().as_secs_f64()); + println!("class v4 sub-version 3 accepted programs' minimum site ratio: min {:.4} p0.1 {:.4} p1 {:.4} p5 {:.4} median {:.4} max {:.4}", sorted[0], q(0.001), q(0.01), q(0.05), q(0.5), sorted[sorted.len() - 1]); + for floor in [0.98, 0.99, 0.995, 0.998, 0.999] { + let under = rows.iter().filter(|r| r.2 < floor).count(); + println!("floor {floor:.3}: {under} of {} class v4 accepted programs under it ({:.3} percent)", rows.len(), under as f64 * 100.0 / n); + } + let mut lowest: Vec<_> = rows.iter().collect(); + lowest.sort_by(|a, b| a.2.partial_cmp(&b.2).unwrap()); + for r in lowest.iter().take(25) { + println!("low: seed {} v4 attempt {} min site {} ratio {:.4}; v5 attempt {} min ratio {:.4}", r.0, r.1, r.3, r.2, r.4, r.5); + } + let hist = |pick: fn(&(u32, u32, f64, usize, u32, f64)) -> u32| { + let mut h = std::collections::BTreeMap::new(); + for r in &rows { + *h.entry(pick(r)).or_insert(0u32) += 1; + } + let mean = rows.iter().map(|r| pick(r) as f64).sum::() / n; + (h, mean) + }; + let (h4, m4) = hist(|r| r.1); + let (h5, m5) = hist(|r| r.4); + println!("attempts histogram class v4 (mean {m4:.3}): {:?}", h4); + println!("attempts histogram class v5 (mean {m5:.3}): {:?}", h5); + let moved = rows.iter().filter(|r| r.1 != r.4).count(); + println!("seeds whose class v5 draw lands on another attempt than the class v4 draw: {moved} of {} ({:.3} percent)", rows.len(), moved as f64 * 100.0 / n); + let v5_under = rows.iter().filter(|r| r.5 < MIN_DISTINCT_RATIO_V5).count(); + println!("class v5 accepted programs under the class v5 floor {MIN_DISTINCT_RATIO_V5}: {v5_under} (must be 0)"); + assert_eq!(v5_under, 0); + } + #[test] fn distinct_bound_scales_with_the_load_count() { assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM); diff --git a/igneum-pow/src/blake2b.rs b/igneum-pow/src/blake2b.rs new file mode 100644 index 000000000..4b04cc622 --- /dev/null +++ b/igneum-pow/src/blake2b.rs @@ -0,0 +1,152 @@ +//! BLAKE2b (RFC 7693), the chain's own hash family (spec 01 section 0.6), written out here so the crate keeps its +//! rule of no dependency outside the standard library. Used by class v5's state leaves (`crate::state`): +//! `blake2b_512` for a leaf digest, `blake2b_256` for the sample order. Unkeyed, no salt, no personalisation. +//! Checked against the RFC's "abc" vector and the empty-input vector in the tests. + +const IV: [u64; 8] = [ + 0x6a09e667f3bcc908, + 0xbb67ae8584caa73b, + 0x3c6ef372fe94f82b, + 0xa54ff53a5f1d36f1, + 0x510e527fade682d1, + 0x9b05688c2b3e6c1f, + 0x1f83d9abfb41bd6b, + 0x5be0cd19137e2179, +]; + +const SIGMA: [[usize; 16]; 12] = [ + [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15], + [14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3], + [11, 8, 12, 0, 5, 2, 15, 13, 10, 14, 3, 6, 7, 1, 9, 4], + [7, 9, 3, 1, 13, 12, 11, 14, 2, 6, 5, 10, 4, 0, 15, 8], + [9, 0, 5, 7, 2, 4, 10, 15, 14, 1, 11, 12, 6, 8, 3, 13], + [2, 12, 6, 10, 0, 11, 8, 3, 4, 13, 7, 5, 15, 14, 1, 9], + [12, 5, 1, 15, 14, 13, 4, 10, 0, 7, 6, 3, 9, 2, 8, 11], + [13, 11, 7, 14, 12, 1, 3, 9, 5, 0, 15, 4, 8, 6, 2, 10], + [6, 15, 14, 9, 11, 3, 0, 8, 12, 2, 13, 7, 1, 4, 10, 5], + [10, 2, 8, 4, 7, 6, 1, 5, 15, 11, 9, 14, 3, 12, 13, 0], + [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15], + [14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3], +]; + +#[inline(always)] +fn g(v: &mut [u64; 16], a: usize, b: usize, c: usize, d: usize, x: u64, y: u64) { + v[a] = v[a].wrapping_add(v[b]).wrapping_add(x); + v[d] = (v[d] ^ v[a]).rotate_right(32); + v[c] = v[c].wrapping_add(v[d]); + v[b] = (v[b] ^ v[c]).rotate_right(24); + v[a] = v[a].wrapping_add(v[b]).wrapping_add(y); + v[d] = (v[d] ^ v[a]).rotate_right(16); + v[c] = v[c].wrapping_add(v[d]); + v[b] = (v[b] ^ v[c]).rotate_right(63); +} + +fn compress(h: &mut [u64; 8], block: &[u8; 128], t: u128, last: bool) { + let mut m = [0u64; 16]; + for (i, w) in m.iter_mut().enumerate() { + *w = u64::from_le_bytes(block[i * 8..i * 8 + 8].try_into().unwrap()); + } + let mut v = [0u64; 16]; + v[..8].copy_from_slice(h); + v[8..].copy_from_slice(&IV); + v[12] ^= t as u64; + v[13] ^= (t >> 64) as u64; + if last { + v[14] = !v[14]; + } + for s in SIGMA.iter() { + g(&mut v, 0, 4, 8, 12, m[s[0]], m[s[1]]); + g(&mut v, 1, 5, 9, 13, m[s[2]], m[s[3]]); + g(&mut v, 2, 6, 10, 14, m[s[4]], m[s[5]]); + g(&mut v, 3, 7, 11, 15, m[s[6]], m[s[7]]); + g(&mut v, 0, 5, 10, 15, m[s[8]], m[s[9]]); + g(&mut v, 1, 6, 11, 12, m[s[10]], m[s[11]]); + g(&mut v, 2, 7, 8, 13, m[s[12]], m[s[13]]); + g(&mut v, 3, 4, 9, 14, m[s[14]], m[s[15]]); + } + for i in 0..8 { + h[i] ^= v[i] ^ v[i + 8]; + } +} + +/// Unkeyed BLAKE2b of `data` with an output of `out_len` bytes (1..=64), written into `out[..out_len]`. +pub fn blake2b(out: &mut [u8], out_len: usize, data: &[u8]) { + assert!((1..=64).contains(&out_len) && out.len() >= out_len); + let mut h = IV; + h[0] ^= 0x0101_0000 ^ out_len as u64; + let mut t: u128 = 0; + let n = data.len(); + // every full block but the last; the last block (possibly empty) is compressed with the final flag + let full = if n == 0 { 0 } else { (n - 1) / 128 }; + for i in 0..full { + let block: &[u8; 128] = data[i * 128..i * 128 + 128].try_into().unwrap(); + t += 128; + compress(&mut h, block, t, false); + } + let mut last = [0u8; 128]; + let rest = &data[full * 128..]; + last[..rest.len()].copy_from_slice(rest); + t += rest.len() as u128; + compress(&mut h, &last, t, true); + let mut bytes = [0u8; 64]; + for (i, w) in h.iter().enumerate() { + bytes[i * 8..i * 8 + 8].copy_from_slice(&w.to_le_bytes()); + } + out[..out_len].copy_from_slice(&bytes[..out_len]); +} + +/// BLAKE2b-512 of the concatenation of `parts`. +pub fn blake2b_512(parts: &[&[u8]]) -> [u8; 64] { + let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum()); + for p in parts { + data.extend_from_slice(p); + } + let mut out = [0u8; 64]; + blake2b(&mut out, 64, &data); + out +} + +/// BLAKE2b-256 of the concatenation of `parts`. +pub fn blake2b_256(parts: &[&[u8]]) -> [u8; 32] { + let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum()); + for p in parts { + data.extend_from_slice(p); + } + let mut out = [0u8; 32]; + blake2b(&mut out, 32, &data); + out +} + +#[cfg(test)] +mod tests { + use super::*; + + fn hex(b: &[u8]) -> String { + b.iter().map(|x| format!("{x:02x}")).collect() + } + + /// RFC 7693 appendix A ("abc"), the empty input, and a two-block input against the reference implementation's + /// known values (the three-block "The quick brown fox" vector of the BLAKE2 test suite). + #[test] + fn rfc_7693_vectors() { + assert_eq!( + hex(&blake2b_512(&[b"abc"])), + "ba80a53f981c4d0d6a2797b69f12f6e94c212f14685ac4b74b12bb6fdbffa2d17d87c5392aab792dc252d5de4533cc9518d38aa8dbf1925ab92386edd4009923" + ); + assert_eq!( + hex(&blake2b_512(&[b""])), + "786a02f742015903c6c6fd852552d272912f4740e15847618a86e217f71f5419d25e1031afee585313896444934eb04b903a685b1448b755d56f701afe9be2ce" + ); + assert_eq!(hex(&blake2b_256(&[b"abc"])), "bddd813c634239723171ef3fee98579b94964e3bb1cb3e427262c8c068d52319"); + assert_eq!(hex(&blake2b_256(&[b""])), "0e5751c026e543b2e8ab2eb06099daa1d1e5df47778f7787faab45cdf12fe3a8"); + // a 128-byte input is exactly one full block compressed as the last; 129 bytes takes two + let one = [0x61u8; 128]; + let two = [0x61u8; 129]; + assert_ne!(blake2b_512(&[&one]), blake2b_512(&[&two])); + assert_eq!(blake2b_512(&[&one[..64], &one[64..]]), blake2b_512(&[&one]), "parts concatenate"); + assert_eq!( + hex(&blake2b_512(&[b"The quick brown fox jumps over the lazy dog"])), + "a8add4bdddfd93e4877d2746e62817b116364a1fa7bc148d95090bc7333b3673f82401cf7aa2e4cb1ecd90296e3f14cb5413f8ed77be73045b13914cdcd6a918" + ); + } +} diff --git a/igneum-pow/src/emit.rs b/igneum-pow/src/emit.rs index 8d93a59a3..90105ba01 100644 --- a/igneum-pow/src/emit.rs +++ b/igneum-pow/src/emit.rs @@ -164,7 +164,11 @@ fn program_class_header_lines(p: &Program) -> String { return String::new(); } let mut s = String::new(); - if p.program_class() == ProgramClass::V4 { + if p.program_class() == ProgramClass::V5 { + s.push_str("// Program class v5 (proof of stored state and of following, docs/design/class-v5-stored-state.md): generator version 5,\n"); + s.push_str("// class v4 over a dataset whose every item is keyed by the window's execution state (IGNEUM_STATE_* below, leaves.bin);\n"); + s.push_str("// a worker that runs another class refuses this pack, and a job line names the class it wants (class=v5 era=).\n"); + } else if p.program_class() == ProgramClass::V4 { s.push_str("// Program class v4 (Counter ASIC 3.0, docs/plans/counter-asic-3-node.md): generator version 4, class v3 plus the\n"); s.push_str("// latency-shadow block (IGNEUM_SHADOW_INSTRS x IGNEUM_SHADOW_REPS per iteration); a worker that runs another class\n"); s.push_str("// refuses this pack, and a job line names the class it wants (class=v4 era=).\n"); @@ -183,6 +187,24 @@ fn program_class_header_lines(p: &Program) -> String { s } +/// The state lines of program.h (class v5): the window's reference block and state root, the leaf count, the FNV of +/// `leaves.bin` and the file's name. Empty for every dataset without leaves, so no pinned pack changes. +fn state_header_lines(ds: &DatasetSource) -> String { + let Some(l) = ds.leaves() else { return String::new() }; + let mut s = String::new(); + s.push_str("// Class v5 state (docs/design/class-v5-stored-state.md): the window's reference chain block and the state root after it;\n"); + s.push_str("// leaves.bin holds IGNEUM_STATE_LEAVES leaves of 16 little-endian words, leaf(t) = leaves[t mod IGNEUM_STATE_LEAVES].\n"); + s.push_str(&format!("#define IGNEUM_STATE_BLOCK_HEX {}\n", jstr(&hex_bytes(&l.block)))); + s.push_str(&format!("#define IGNEUM_STATE_BLOCK_NUMBER {}\n", l.number)); + s.push_str(&format!("#define IGNEUM_STATE_ROOT_HEX {}\n", jstr(&hex_bytes(&l.root)))); + s.push_str(&format!("#define IGNEUM_STATE_LEAVES {}\n", l.n())); + s.push_str(&format!("#define IGNEUM_STATE_RECORDS {}\n", l.records_total)); + s.push_str(&format!("#define IGNEUM_STATE_SAMPLED {}\n", l.sampled as u8)); + s.push_str(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}\n", hex64(l.fnv1a64()))); + s.push_str("#define IGNEUM_STATE_LEAVES_FILE \"leaves.bin\"\n"); + s +} + /// The load class lines of program.h (empty for the lottery hash, so the pinned packs do not change). fn class_header_lines(p: &Program) -> String { if p.class.is_v2() { @@ -666,28 +688,35 @@ pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: La s.push_str("}\n"); } s.push_str(&format!("// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of (round program r, cache line s[0] & mask); round program {ITEM_ROUNDS}.\n")); - s.push_str(&format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n")); + s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr)); for i in 0..8 { s.push_str(&format!(" s[{i}] = {};\n", hex(k[i]))); } for i in 0..8 { s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i]))); } + if shape.state { + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n")); + } for r in 0..ITEM_ROUNDS { s.push_str(&format!(" mh_round_{r}(s);\n")); s.push_str(&format!(" {{ {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i]; }}\n")); } s.push_str(&format!(" mh_round_{ITEM_ROUNDS}(s);\n")); s.push_str("}\n"); - return finish_memhard_core(s, layout, u, fn_, cptr); + return finish_memhard_core(s, layout, u, fn_, cptr, shape.state); } - s.push_str(&format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n")); + s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr)); for i in 0..8 { s.push_str(&format!(" s[{i}] = {};\n", hex(k[i]))); } for i in 0..8 { s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i]))); } + if shape.state { + // class v5: the window's state leaf of item t, before the first mixer (docs/design/class-v5-stored-state.md) + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n")); + } s.push_str(&format!(" for ({u} r = 0u; r < {ITEM_ROUNDS}u; ++r) {{\n")); if m == 1 { s.push_str(" mh_mixer(s, 0x9E3779B9u * (r + 1u));\n"); @@ -706,22 +735,48 @@ pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: La )); } s.push_str("}\n"); - finish_memhard_core(s, layout, u, fn_, cptr) + finish_memhard_core(s, layout, u, fn_, cptr, shape.state) } -/// The tail of the memhard core: `mh_word` (and the era layout helpers) after `mh_item`. -fn finish_memhard_core(mut s: String, layout: Layout, u: &str, fn_: &str, cptr: &str) -> String { +/// The `mh_item` signature: under a state shape (class v5) the item takes its 16-word leaf (`leaves + 16 (t mod n)`). +fn item_signature(state: bool, fn_: &str, cptr: &str, u: &str, lptr: &str) -> String { + if state { + format!("{fn_} void mh_item({cptr} cache, {cptr} leaf, {u} t, {lptr} s) {{\n") + } else { + format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n") + } +} + +/// The tail of the memhard core: `mh_word` (and the era layout helpers) after `mh_item`. Under a state shape +/// `mh_word` takes the leaves and their count and derives item t's leaf as `leaves + 16 (t mod nLeaves)`. +fn finish_memhard_core(mut s: String, layout: Layout, u: &str, fn_: &str, cptr: &str, state: bool) -> String { + if state { + s.push_str("// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).\n"); + s.push_str(&format!("{fn_} {cptr} mh_leaf({cptr} leaves, {u} nLeaves, {u} t) {{ return leaves + ((t % nLeaves) * 16u); }}\n")); + } if layout.is_linear() { s.push_str("// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.\n"); - s.push_str(&format!( - "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n" - )); + if state { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }}\n" + )); + } else { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n" + )); + } } else { s.push_str(&layout_helpers(layout, u, fn_)); s.push_str("// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).\n"); - s.push_str(&format!( - "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n" - )); + if state { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }}\n" + )); + } else { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n" + )); + } } s } @@ -780,12 +835,23 @@ pub fn metal_memhard_layout(mp: &MixParams, layout: Layout) -> String { s.push_str(" mh_cache_segment(cache, gid);\n"); s.push_str("}\n"); s.push_str("// One thread per 64-byte item (dataset words / 16 threads).\n"); - s.push_str( - "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n", - ); - s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); - s.push_str(" uint s[16];\n"); - s.push_str(" mh_item(cache, gid, s);\n"); + if mp.shape.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) in buffer 2, their count in buffer 3.\n"); + s.push_str( + "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n", + ); + s.push_str(" device const uint* leaves [[buffer(2)]], constant uint& nLeaves [[buffer(3)]],\n"); + s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" uint s[16];\n"); + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, gid), gid, s);\n"); + } else { + s.push_str( + "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n", + ); + s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" uint s[16];\n"); + s.push_str(" mh_item(cache, gid, s);\n"); + } s.push_str(&build_store(layout, CoreDialect::Metal, "dataset", "gid")); s.push_str("}\n"); s @@ -945,7 +1011,7 @@ fn generated_by(seed: &str) -> String { format!("// Generated by igneum-pow export (generator v{GENERATOR_VERSION}) for seed \"{seed}\". Do not edit by hand.\n") } -fn hex_bytes(b: &[u8]) -> String { +pub fn hex_bytes(b: &[u8]) -> String { b.iter().map(|x| format!("{x:02x}")).collect() } @@ -1054,11 +1120,20 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n"); s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n"); s.push_str("}\n"); - s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); + if p.class.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n"); + s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n"); + } else { + s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); + } s.push_str(" uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;\n"); s.push_str(" if (t < nItems) {\n"); s.push_str(" uint32_t s[16];\n"); - s.push_str(" mh_item(cache, t, s);\n"); + if p.class.state { + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n"); + } else { + s.push_str(" mh_item(cache, t, s);\n"); + } s.push_str(&build_store(layout, CoreDialect::Cuda, "ds", "t")); s.push_str(" }\n"); s.push_str("}\n"); @@ -1125,11 +1200,20 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); s.push('\n'); - s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); - s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n"); + if p.class.state { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n"); + s.push_str(" if (nItems == 0u || nLeaves == 0u) return cudaErrorInvalidValue;\n"); + } else { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); + s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n"); + } s.push_str(" uint32_t block = 256u;\n"); s.push_str(" uint32_t grid = (nItems + block - 1u) / block;\n"); - s.push_str(" igneum_build<<>>(ds, cache, nItems);\n"); + if p.class.state { + s.push_str(" igneum_build<<>>(ds, cache, leaves, nLeaves, nItems);\n"); + } else { + s.push_str(" igneum_build<<>>(ds, cache, nItems);\n"); + } s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); s.push('\n'); @@ -1469,11 +1553,20 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str(" uint seg = (uint)get_global_id(0);\n"); s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n"); s.push_str("}\n"); - s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n"); + if p.class.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n"); + s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {\n"); + } else { + s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n"); + } s.push_str(" uint t = (uint)get_global_id(0);\n"); s.push_str(" if (t < nItems) {\n"); s.push_str(" uint s[16];\n"); - s.push_str(" mh_item(cache, t, s);\n"); + if p.class.state { + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n"); + } else { + s.push_str(" mh_item(cache, t, s);\n"); + } s.push_str(&build_store(layout, CoreDialect::OpenCl, "ds", "t")); s.push_str(" }\n"); s.push_str("}\n"); @@ -1602,6 +1695,7 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix()))); s.push_str(&program_class_header_lines(p)); s.push_str(&class_header_lines(p)); + s.push_str(&state_header_lines(ds)); s.push_str(&scratch_header_lines(p)); s.push_str(&era_header_lines(p)); s.push_str(&hot_header_lines(p)); @@ -1638,7 +1732,11 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str("#ifndef IGNEUM_NO_CUDA\n"); s.push_str("// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().\n"); s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);\n"); - s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n"); + if p.class.state { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);\n"); + } else { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n"); + } if p.has_hot() { s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments);\n"); } @@ -1836,6 +1934,19 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!(" \"era_seed_bytes\": {},\n", jstr(&hex_bytes(era)))); } } + if let Some(l) = ds.leaves() { + s.push_str(" \"state\": {\n"); + s.push_str(&format!(" \"block\": {},\n", jstr(&hex_bytes(&l.block)))); + s.push_str(&format!(" \"block_number\": {},\n", l.number)); + s.push_str(&format!(" \"root\": {},\n", jstr(&hex_bytes(&l.root)))); + s.push_str(&format!(" \"leaves\": {},\n", l.n())); + s.push_str(&format!(" \"records\": {},\n", l.records_total)); + s.push_str(&format!(" \"sampled\": {},\n", l.sampled)); + s.push_str(&format!(" \"leaves_fnv1a64\": {},\n", jhex64(l.fnv1a64()))); + s.push_str(" \"leaf_derivation\": \"leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer\",\n"); + s.push_str(" \"file\": \"leaves.bin\"\n"); + s.push_str(" },\n"); + } if !p.class.is_v2() { let c = p.width_counts(); s.push_str(&format!(" \"load_class\": {},\n", jstr(&p.class.name()))); @@ -2119,6 +2230,8 @@ pub fn vectors_json( /// A program pack: the files `--export-pack` writes, as (name, text). pub struct Pack { pub files: Vec<(String, String)>, + /// Binary files beside the texts: `leaves.bin` of a class v5 pack (empty for every other pack). + pub binaries: Vec<(String, Vec)>, pub bases: Vec, pub outs: Vec<[u64; 32]>, pub vectors: PackVectors, @@ -2130,6 +2243,9 @@ impl Pack { for (name, text) in &self.files { std::fs::write(dir.join(name), text)?; } + for (name, bytes) in &self.binaries { + std::fs::write(dir.join(name), bytes)?; + } Ok(()) } } @@ -2183,7 +2299,11 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp))); files.push(("memhard.metal".to_string(), metal_memhard_for(p, mp))); } - Pack { files, bases, outs, vectors: v } + let binaries = match ds.leaves() { + Some(l) => vec![("leaves.bin".to_string(), l.bytes())], + None => Vec::new(), + }; + Pack { files, binaries, bases, outs, vectors: v } } /// The dataset mode a pack was written in, from its program.json text (no JSON parser needed). diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index 9c2ebe05a..5deb79d69 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -238,6 +238,10 @@ pub struct LoadClass { /// Latency-shadow program work (Counter ASIC 3.0 item 8, measured 6 October 2026 and not adopted): `Some` adds a /// block of ALU instructions run `reps` times per iteration. `None` for every other class, class v3 included. pub shadow: Option, + /// Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026): + /// the item derivation XORs the window's state leaf into every item before the first mixer (`crate::state`, + /// `memhard::derive_items_leaves`). The program draw does not read it. `false` for every other class. + pub state: bool, } /// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of @@ -415,13 +419,13 @@ impl LoadClass { impl LoadClass { /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash. pub const V2: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false }; /// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): /// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the /// mixer applied 4 times per round, and the cache growth rule. Name "mx4". pub const MX4: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false }; /// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed` /// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base @@ -547,6 +551,16 @@ impl LoadClass { LoadClass { shadow: Some(ShadowClass { instrs, reps }), ..self } } + /// This class with another class's era draw (tests: a rung's class composed with the chain's era). + pub fn with_era_of(self, other: &LoadClass) -> LoadClass { + LoadClass { era: other.era, ..self } + } + + /// The class with the state leaves of class v5 folded into every item ("mx8+sh256x27+state"). + pub fn with_state(self) -> LoadClass { + LoadClass { state: true, ..self } + } + /// Shadow instructions per hash (0 without a shadow). pub fn shadow_instrs_per_hash(&self) -> usize { self.shadow.map(|s| s.instrs_per_hash()).unwrap_or(0) @@ -578,6 +592,10 @@ impl LoadClass { /// "mx4": the v3 construction; a trailing "m" and "g" set the mixer multiplier and the growth rule on any /// load class, "w16m4g" for example). pub fn parse(s: &str) -> Option { + // "+state": the state leaves of class v5 over any class (the suffix is outermost) + if let Some(base) = s.strip_suffix("+state") { + return Some(LoadClass::parse(base)?.with_state()); + } // "+shx": the latency-shadow block over any class (Counter ASIC 3.0 item 8) if let Some((base, sh)) = s.rsplit_once("+sh") { let (instrs, reps) = sh.split_once('x')?; @@ -701,6 +719,10 @@ impl LoadClass { /// An era class is the base name with "-era" appended ("w4-era401998a5", "mx4-era..."). /// A hot class appends "hotk[a]" ("hot64k4", "scr4k32+hot64k4a"; measured and not adopted). pub fn name(&self) -> String { + if self.state { + // "+state": class v5's leaves are a suffix on any class, outermost + return format!("{}+state", LoadClass { state: false, ..*self }.name()); + } if let Some(sh) = self.shadow { // "+shx": the shadow block is a suffix on any class ("mx8+sh256x13") return format!("{}+sh{}x{}", LoadClass { shadow: None, ..*self }.name(), sh.instrs, sh.reps); @@ -794,6 +816,10 @@ pub const GENERATOR_VERSION_V3: u32 = 3; /// Generator version of a class v4 program (Counter ASIC 3.0, 6 October 2026, PROPOSED: `program_id(4, seed, attempt)`). pub const GENERATOR_VERSION_V4: u32 = 4; +/// Generator version of a class v5 program (proof of stored state and of following, 7 October 2026, PROPOSED: +/// `program_id(5, seed, attempt)`; `docs/design/class-v5-stored-state.md`). +pub const GENERATOR_VERSION_V5: u32 = 5; + /// The program class of an epoch (Counter ASIC 2.0, 5 October 2026, `docs/plans/counter-asic-2-rollout.md`): one /// height switch in the node, `program_class_v3_activation_daa`, rounded up to an epoch boundary, decides which /// class an epoch's program is drawn from. V2 is the lottery hash as adopted on 4 October 2026, byte for byte. @@ -806,6 +832,9 @@ pub enum ProgramClass { V2, V3, V4, + /// Class v5 (`docs/design/class-v5-stored-state.md`, behind `program_class_v5_activation_daa`): class v4's program + /// over a dataset whose every item is keyed by the window's execution state ([`V5_CLASS`]), generator 5. + V5, } /// The load class of program class v3, decided 5 October 2026 (Counter ASIC 2.0, `docs/plans/counter-asic-2-status.md` @@ -823,6 +852,12 @@ pub const V3_CLASS: LoadClass = LoadClass { era: None, hot: None, ..LoadClass::M /// for draw, so a v4 epoch's day cache and dataset are the v3 day's. Composed with the era exactly as V3 is. pub const V4_CLASS: LoadClass = LoadClass { shadow: Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps: V4_SHADOW_REPS }), ..V3_CLASS }; +/// The load class of program class v5 (`docs/design/class-v5-stored-state.md`, 7 October 2026): class v4 with the +/// state leaves of the window's reference block folded into every item of the dataset ("mx8+sh256x27+state"). The +/// program draw, the shadow block, the era draw and the ladder rung are class v4's, draw for draw; only the item +/// derivation and the program id change. +pub const V5_CLASS: LoadClass = LoadClass { state: true, ..V4_CLASS }; + /// The shadow block size of class v4 at every rung of the latency ladder (`docs/design/latency-ladder.md`): 256 /// instructions. The ladder moves the pass count alone. pub const V4_SHADOW_INSTRS: u16 = 256; @@ -841,10 +876,18 @@ pub fn v4_class_at(reps: u16) -> LoadClass { } } +/// Class v5 at a rung of the latency ladder: [`v4_class_at`] with the state leaves (`v5_class_at(0) == V5_CLASS`). +pub fn v5_class_at(reps: u16) -> LoadClass { + v4_class_at(reps).with_state() +} + /// The shadow passes of a class v4 load class at any rung of the ladder, the era draw set aside (`Some(27)` for /// [`V4_CLASS`] itself); `None` for every other class, a measurement class with another block size included. pub fn v4_rung_reps(class: &LoadClass) -> Option { let base = LoadClass { era: None, ..*class }; + if base.state { + return None; + } match base.shadow { Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps }) if LoadClass { shadow: None, ..base } == V3_CLASS => Some(reps), _ => None, @@ -875,7 +918,20 @@ pub fn generate_era(seed_string: &str, seed_bytes: &[u8], base: LoadClass, era_b /// path on `mx8+sh256x27` were stamped generator 3 and so carried the v3 control's program id (`program_id(3, seed, /// attempt)` is class-independent inside a generator version); a class v4 program is generator 4 wherever it is made. pub fn era_generator_of(base: &LoadClass) -> u32 { - if ProgramClass::of_load_class(base) == Some(ProgramClass::V4) { GENERATOR_VERSION_V4 } else { GENERATOR_VERSION_V3 } + match ProgramClass::of_load_class(base) { + Some(ProgramClass::V4) => GENERATOR_VERSION_V4, + Some(ProgramClass::V5) => GENERATOR_VERSION_V5, + _ if base.state => GENERATOR_VERSION_V5, + _ => GENERATOR_VERSION_V3, + } +} + +/// The shadow passes of a class v5 load class at any rung (the state flag set aside); `None` for every other class. +pub fn v5_rung_reps(class: &LoadClass) -> Option { + if !class.state { + return None; + } + v4_rung_reps(&LoadClass { state: false, ..*class }) } /// [`generate_era`] with the generator version stamped by the caller: 3 for class v3 over [`V3_CLASS`], 4 for @@ -894,6 +950,7 @@ impl ProgramClass { ProgramClass::V2 => LoadClass::V2, ProgramClass::V3 => V3_CLASS, ProgramClass::V4 => V4_CLASS, + ProgramClass::V5 => V5_CLASS, } } @@ -903,15 +960,22 @@ impl ProgramClass { ProgramClass::V2 => GENERATOR_VERSION, ProgramClass::V3 => GENERATOR_VERSION_V3, ProgramClass::V4 => GENERATOR_VERSION_V4, + ProgramClass::V5 => GENERATOR_VERSION_V5, } } + /// Whether the class's dataset is keyed by the window's execution state (class v5). + pub fn has_state(&self) -> bool { + *self == ProgramClass::V5 + } + /// The class of a generator version: 2, 3 and 4 are the three classes, anything else is no class this crate runs. pub fn from_generator(generator: u32) -> Option { match generator { GENERATOR_VERSION => Some(ProgramClass::V2), GENERATOR_VERSION_V3 => Some(ProgramClass::V3), GENERATOR_VERSION_V4 => Some(ProgramClass::V4), + GENERATOR_VERSION_V5 => Some(ProgramClass::V5), _ => None, } } @@ -922,6 +986,7 @@ impl ProgramClass { ProgramClass::V2 => "v2", ProgramClass::V3 => "v3", ProgramClass::V4 => "v4", + ProgramClass::V5 => "v5", } } @@ -930,6 +995,7 @@ impl ProgramClass { "v2" => Some(ProgramClass::V2), "v3" => Some(ProgramClass::V3), "v4" => Some(ProgramClass::V4), + "v5" => Some(ProgramClass::V5), _ => None, } } @@ -949,6 +1015,8 @@ impl ProgramClass { Some(ProgramClass::V3) } else if base == V4_CLASS { Some(ProgramClass::V4) + } else if base == V5_CLASS { + Some(ProgramClass::V5) } else { None } @@ -1067,7 +1135,9 @@ impl Program { // pack of another rung is refused as a pack of another class is. Rung 0 keeps `program_id(4, seed, attempt)` // byte for byte, so every v4 id written before the ladder stands. let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS; - if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 { + // class v5 at rung 0 is `program_id(5, seed, attempt)`; above rung 0 the class-bearing id with "state/" + let v5_rung_0 = self.generator == GENERATOR_VERSION_V5 && LoadClass { era: None, ..self.class } == V5_CLASS; + if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0 { // Spec 01 section 1.4.6: a class v3 program's id is `program_id(3, seed, attempt)`, a class v4 program's // `program_id(4, seed, attempt)` (Counter ASIC 3.0); the generator version in the preimage separates // them from every version 2 program of the same seed @@ -1082,6 +1152,7 @@ impl Program { match self.generator { GENERATOR_VERSION_V3 => ProgramClass::V3, GENERATOR_VERSION_V4 => ProgramClass::V4, + GENERATOR_VERSION_V5 => ProgramClass::V5, _ => ProgramClass::V2, } } @@ -1164,6 +1235,10 @@ pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &L b.extend_from_slice(b"added"); } } + if class.state { + // class v5: the state leaves are part of the construction + b.extend_from_slice(b"state/"); + } fnv1a64(&b) } @@ -1218,6 +1293,46 @@ pub fn candidate_from_words(seed_string: &str, seed_bytes: &[u8], seed: [u32; 8] /// [`candidate_from_words`] for a load class. For [`LoadClass::V2`] this is the version 2 draw stream exactly; /// for any other class the slot count is the class's and every instruction takes a tenth draw, `below(100)`, /// the width roll (used only on a load slot, drawn on every slot so the stream stays uniform). +/// Class v5's shadow rule (AP-F1-1): the peephole-removable share a block may have, per mille of its instructions. +pub const SHADOW_REMOVABLE_MAX_PERMILLE: usize = 30; +/// Class v5's shadow rule: redraws before the last block stands as drawn (never reached at 4e-3 per try). +pub const SHADOW_REDRAW_CAP: u32 = 64; + +/// The instructions of a shadow block an honest compiler removes (the AP-F1-1 census's classes): an instruction whose +/// destination's last writer, with no write to the destination or to the source in between, is the same op on the +/// same source and immediates and the pair cancels (xor: `x ^= s` twice) or is idempotent (or: `x |= s` twice), or a +/// rotate of a register last written by a rotate (the two merge into one), or an add followed by a sub (or a sub by an +/// add) of the same source and immediate (sum-cancel). Counted per removable instruction, never across a pass. +pub fn shadow_removable_count(block: &[Instr]) -> usize { + let mut last: [Option; 8] = [None; 8]; + let mut removable = 0; + for (k, ins) in block.iter().enumerate() { + let d = ins.dst as usize; + if let Some(j) = last[d] { + let prev = &block[j]; + // the source must not have been written between j and k (its value is the same) + let src_untouched = !block[j + 1..k].iter().any(|i| i.dst == ins.src); + let same_operands = prev.src == ins.src && prev.imm == ins.imm && prev.imm2 == ins.imm2 && prev.src2 == ins.src2 && prev.rot == ins.rot && prev.bit == ins.bit && prev.mask == ins.mask; + let hit = match (prev.op, ins.op) { + (Op::Xor, Op::Xor) | (Op::Or, Op::Or) => same_operands && src_untouched, + // every op reads its own destination (`d = d op a`), so two rotates on one register always fold on + // paper; the census counts "written twice from one source": the same rotate with the same amount + // (rotl, an immediate) or the same amount register unwritten between (rotr), which keeps the + // metric at the census's 0.62 percent average instead of every rotate pair + (Op::Rotl, Op::Rotl) => prev.rot == ins.rot, + (Op::Rotr, Op::Rotr) => prev.src == ins.src && src_untouched, + (Op::Add, Op::Sub) | (Op::Sub, Op::Add) => same_operands && src_untouched, + _ => false, + }; + if hit { + removable += 1; + } + } + last[d] = Some(k); + } + removable +} + pub fn candidate_from_words_class( seed_string: &str, seed_bytes: &[u8], @@ -1262,8 +1377,9 @@ pub fn candidate_from_words_class( // Keyed on the class v4 shape (the 256-instruction shadow block over the class v3 base, the pass count and the // era set aside) on EVERY draw path, era or not, so a census through candidate_class reads the same stream as // the chain; v2, v3 and every other class take no part. The draw order and the stream are otherwise the same. + // class v5 (docs/design/class-v5-stored-state.md) draws under the same rule: its state flag is set aside here too let source_rule_v4 = matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. })) - && LoadClass { era: None, shadow: None, ..class } == LoadClass { shadow: None, ..V4_CLASS }; + && LoadClass { era: None, shadow: None, state: false, ..class } == LoadClass { shadow: None, ..V4_CLASS }; let mut fresh = [false; 8]; let mut fresh_value = [true; 8]; // the shared-operand idiom (AP-F8-1, sub-version 3): after `or d |= s`, a later `xor d ^= s` or `sub d -= s` with @@ -1371,7 +1487,14 @@ pub fn candidate_from_words_class( // op from the non-load table, the source as on an ALU slot, the same per-instruction draws (the width roll and // the era windows included when the class takes them, drawn and ignored) so the stream shape is the program's. let mut shadow = Vec::new(); + let mut shadow_redraws = 0u32; if let Some(sh) = class.shadow { + // Class v5 (docs/design/class-v5-stored-state.md section 11, AP-F1-1): a shadow block whose peephole-removable + // share exceeds SHADOW_REMOVABLE_MAX_PERMILLE (3.0 percent of the block; the census's histogram puts 384 of 100,000 + // draws there) is redrawn from the continuing stream, so an honest compiler's simplification cannot take more + // than 3 percent of the shadow's useful work. Every other class keeps its first draw. + loop { + shadow.clear(); for _ in 0..sh.instrs { let mut roll = rng.below(75); let mut op = Op::Add; @@ -1400,7 +1523,13 @@ pub fn candidate_from_words_class( } shadow.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask, width: 1, win: 0, off: 0 }); } + if !class.state || shadow_removable_count(&shadow) * 1000 <= sh.instrs as usize * SHADOW_REMOVABLE_MAX_PERMILLE || shadow_redraws >= SHADOW_REDRAW_CAP { + break; + } + shadow_redraws += 1; + } } + let _ = shadow_redraws; Program { seed_string: seed_string.to_string(), seed_bytes: seed_bytes.to_vec(), @@ -1504,6 +1633,8 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u match (class, era_bytes) { (ProgramClass::V3, Some(era)) => return generate_era(seed_string, seed_bytes, V3_CLASS, era, &V3_ALLOWED), (ProgramClass::V4, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V4_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V4), + // Class v5: the same draw inside V5_CLASS (V4_CLASS plus the state flag, which the draw does not read), generator 5. + (ProgramClass::V5, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V5_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V5), _ => {} } let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, class.load_class()); @@ -1520,15 +1651,16 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u /// and class v4 at rung 0, is [`generate_from_seed_bytes_program_class`] byte for byte. The base program, the 16 loads /// and the era draw do not move with the rung: only the pass count of the shadow block does. pub fn generate_from_seed_bytes_program_class_shadow(seed_string: &str, seed_bytes: &[u8], class: ProgramClass, era_bytes: Option<&[u8]>, shadow_reps: u16) -> Program { - if class != ProgramClass::V4 || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS { + if !matches!(class, ProgramClass::V4 | ProgramClass::V5) || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS { return generate_from_seed_bytes_program_class(seed_string, seed_bytes, class, era_bytes); } - let base = v4_class_at(shadow_reps); + // class v5 at a rung: class v4's rung with the state flag, generator 5 + let (base, generator) = if class == ProgramClass::V5 { (v5_class_at(shadow_reps), GENERATOR_VERSION_V5) } else { (v4_class_at(shadow_reps), GENERATOR_VERSION_V4) }; match era_bytes { - Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, GENERATOR_VERSION_V4), + Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, generator), None => { let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, base); - p.generator = GENERATOR_VERSION_V4; + p.generator = generator; p.era_bytes = None; p } @@ -1861,17 +1993,131 @@ mod tests { assert_eq!(v3.program_id(), program_id(GENERATOR_VERSION_V3, &v3.seed, v3.attempt)); assert_ne!(v3.program_id(), program_id(GENERATOR_VERSION, &v3.seed, v3.attempt)); assert_ne!(v3.program_id(), v2.program_id()); - for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4] { + for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4, ProgramClass::V5] { assert_eq!(ProgramClass::parse(c.name()), Some(c)); assert_eq!(ProgramClass::from_generator(c.generator_version()), Some(c)); assert_eq!(ProgramClass::from_u8(c.as_u8()), Some(c)); } assert_eq!(ProgramClass::from_generator(1), None); - assert_eq!(ProgramClass::from_generator(5), None); - assert_eq!(ProgramClass::parse("v5"), None); + assert_eq!(ProgramClass::from_generator(6), None); + assert_eq!(ProgramClass::parse("v6"), None); assert_eq!(ProgramClass::default(), ProgramClass::V2); assert_eq!(ProgramClass::V2.load_class(), LoadClass::V2); - assert!(!ProgramClass::V2.has_era() && ProgramClass::V3.has_era() && ProgramClass::V4.has_era()); + assert!(!ProgramClass::V2.has_era() && ProgramClass::V3.has_era() && ProgramClass::V4.has_era() && ProgramClass::V5.has_era()); + assert!(!ProgramClass::V4.has_state() && ProgramClass::V5.has_state()); + } + + /// Class v5's shadow rule (AP-F1-1), the known-failed case first: a synthetic block of xor-cancel pairs is counted and + /// is over the bound; a scan of seeds finds a v5 draw whose first block was redrawn (the census's 4e-3), every v5 + /// block is under 3.0 percent removable, and the same seed's class v4 block (never redrawn) is the first draw. + #[test] + fn class_v5_shadow_redundancy_rule() { + let mk = |op: Op, dst: u8, src: u8| Instr { op, dst, src, src2: 0, imm: 7, imm2: 9, rot: 3, bit: 0, mask: 1, width: 1, win: 0, off: 0 }; + let pair = vec![mk(Op::Xor, 1, 2), mk(Op::Xor, 1, 2)]; + assert_eq!(shadow_removable_count(&pair), 1, "the known-failed case: an xor-cancel pair"); + let broken = vec![mk(Op::Xor, 1, 2), mk(Op::Add, 2, 3), mk(Op::Xor, 1, 2)]; + assert_eq!(shadow_removable_count(&broken), 0, "the source moved between the two"); + let rot = vec![mk(Op::Rotl, 4, 0), mk(Op::Rotl, 4, 0)]; + assert_eq!(shadow_removable_count(&rot), 1, "two rotl by one amount merge"); + let mixed = vec![mk(Op::Rotl, 4, 0), mk(Op::Rotr, 4, 5)]; + assert_eq!(shadow_removable_count(&mixed), 0, "a rotl and a rotr are two sources: not the census's merge"); + let rotr = vec![mk(Op::Rotr, 4, 5), mk(Op::Rotr, 4, 5)]; + assert_eq!(shadow_removable_count(&rotr), 1, "two rotr by one register merge"); + let sum = vec![mk(Op::Add, 6, 7), mk(Op::Sub, 6, 7)]; + assert_eq!(shadow_removable_count(&sum), 1, "sum-cancel"); + let mut bad = Vec::new(); + for _ in 0..128 { + bad.push(mk(Op::Or, 3, 5)); + bad.push(mk(Op::Or, 3, 5)); + } + assert!(shadow_removable_count(&bad) * 1000 > bad.len() * SHADOW_REMOVABLE_MAX_PERMILLE, "a block of or-idempotent pairs is over the bound"); + // The scan is over the shadow DRAW (the candidate at attempt 0 of each class, no acceptance run): a full draw + // under class v4 or v5 costs seconds per candidate (the 2^20 ratio pass over 64 + 256 x 27 instructions per + // evaluation), and the first form of this scan, 6,000 seeds x two full draws, ran for over an hour on box 2 + // (7 October 2026, 21:4x UK). The redraw rule lives in the candidate draw, so the candidate is where it shows. + let era = [7u8; 32]; + let v5_class = LoadClass::era(V5_CLASS, &era, &V3_ALLOWED); + let v4_class = LoadClass::era(V4_CLASS, &era, &V3_ALLOWED); + let mut redrawn = 0; + let mut scanned = 0; + for i in 0..6_000u32 { + let seed = format!("igneum-shadow-scan/{i}"); + let v5 = candidate_class(&seed, seed.as_bytes(), 0, v5_class); + assert!(shadow_removable_count(&v5.shadow) * 1000 <= v5.shadow.len() * SHADOW_REMOVABLE_MAX_PERMILLE, "{seed}: a v5 block over the bound"); + let v4 = candidate_class(&seed, seed.as_bytes(), 0, v4_class); + if v5.shadow == v4.shadow { + assert_eq!(v5.instrs, v4.instrs, "{seed}: an unredrawn seed is the amended v4 draw for draw"); + } else { + // the first v4 block was over the bound and the rule redrew it from the continuing stream + redrawn += 1; + assert!(shadow_removable_count(&v4.shadow) * 1000 > v4.shadow.len() * SHADOW_REMOVABLE_MAX_PERMILLE, "{seed}: v5 redrew a block the rule admits"); + } + scanned += 1; + if redrawn >= 2 && scanned >= 1_000 { + break; + } + } + assert!(redrawn >= 1, "no redraw in {scanned} seeds (the census says about 4e-3 per draw)"); + assert!(redrawn * 50 <= scanned, "{redrawn} redraws in {scanned} seeds: the metric is far above the census's 4e-3"); + } + + /// Class v5 (docs/design/class-v5-stored-state.md) takes the amended class v4 draw (AP-F8-1) as the chain draws it: + /// with an era present every load's source was last written by an injecting op or a rotate, the base program and + /// the shadow block equal the amended v4's of the same seed, and the same holds at a ladder rung; the state flag + /// changes the id and the dataset, never the draw. The known-failed case first: a v5 draw with a lossy-sourced + /// load would fail the same scan the amended v4 passes. + #[test] + fn class_v5_chain_draw_is_the_amended_v4_draw() { + let era = [7u8; 32]; + let scan = |p: &Program| { + let mut kept = [false; 8]; + let mut lossy = 0; + for i in &p.instrs { + if i.op.is_load() && !kept[i.src as usize] { + lossy += 1; + } + kept[i.dst as usize] = i.op.injects() || matches!(i.op, Op::Rotl | Op::Rotr); + } + lossy + }; + for seed in ["igneum-genesis", "igneum-epoch-7", "igneum-epoch-99"] { + let v4 = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V4, Some(&era)); + let v5 = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V5, Some(&era)); + if v5.shadow == v4.shadow { + assert_eq!(v5.instrs, v4.instrs, "{seed}: the base program is the amended v4's"); + assert_eq!((v5.seed, v5.attempt), (v4.seed, v4.attempt)); + } else { + // the AP-F1-1 shadow rule redrew this seed's block (and the acceptance, which reads the shadow, may have + // moved the attempt): the first v4 block must have been over the bound + let first = candidate_class(seed, seed.as_bytes(), v4.attempt, v4.class); + assert!(shadow_removable_count(&first.shadow) * 1000 > first.shadow.len() * SHADOW_REMOVABLE_MAX_PERMILLE || v5.attempt != v4.attempt, "{seed}: v5 differs from v4 without a redraw"); + } + if v5.attempt != v4.attempt { + // the attempt moved: the v4 draw's program under the v5 shape is refused by a class v5 rule ((c''') the + // raised floor, or the acceptance reading a redrawn shadow), never silently + let same = candidate_class(seed, seed.as_bytes(), v4.attempt, v5.class); + assert!(crate::accept::check(&same).is_err(), "{seed}: the v5 draw skipped attempt {} without a reason", v4.attempt); + } + // the draw equality above is the claim; sub-version 3's own freshness rule (dataflow, with the last resort after + // the 256 cap) is what the chain applies, so the sub-version 1 scan below is informational for v5 + let _ = scan(&v5); + assert_eq!(v5.generator, GENERATOR_VERSION_V5); + assert!(v5.class.state && !v4.class.state); + assert_ne!(v5.program_id(), v4.program_id()); + let r1 = generate_from_seed_bytes_program_class_shadow(seed, seed.as_bytes(), ProgramClass::V5, Some(&era), 35); + let r1v4 = generate_from_seed_bytes_program_class_shadow(seed, seed.as_bytes(), ProgramClass::V4, Some(&era), 35); + if r1.shadow == r1v4.shadow { + assert_eq!(r1.instrs, r1v4.instrs, "{seed}: rung 1 too"); + } + let _ = scan(&r1); + assert_eq!(r1.class, v5_class_at(35).with_era_of(&r1v4.class)); + } + // the known-failed case: the unamended v3 stream of the same seeds is lossy-sourced somewhere in three seeds and is + // not the v5 draw (a v5 that drew without the rule would equal it) + let seeds = ["igneum-genesis", "igneum-epoch-7", "igneum-epoch-99"]; + let lossy_v3: usize = seeds.iter().map(|seed| scan(&generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, Some(&era)))).sum(); + assert!(lossy_v3 > 0, "the known-failed case: the unamended stream carries lossy-sourced loads"); + assert!(seeds.iter().any(|seed| generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, Some(&era)).instrs != generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V5, Some(&era)).instrs), "v5 is not the unamended stream"); } /// Counter ASIC 3.0 (6 October 2026): class v4 is class v3 with the latency-shadow block `sh256x27`, drawn after diff --git a/igneum-pow/src/lib.rs b/igneum-pow/src/lib.rs index 244c4ab17..690ddb0bc 100644 --- a/igneum-pow/src/lib.rs +++ b/igneum-pow/src/lib.rs @@ -24,17 +24,20 @@ pub mod accept; pub mod bind; +pub mod blake2b; pub mod derive; pub mod emit; pub mod generator; pub mod memhard; pub mod packcheck; pub mod seed; +pub mod state; pub mod verify; pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256}; pub use accept::{check as accept_program, AcceptReport, Reject}; -pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS}; +pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, v5_class_at, v5_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, GENERATOR_VERSION_V5, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS, V5_CLASS, PROGRAM_SUBVERSION_V4}; pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape}; pub use seed::{fnv1a64, seed_words, SplitMix64}; +pub use state::{StateLeaves, StateStream}; pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch}; diff --git a/igneum-pow/src/main.rs b/igneum-pow/src/main.rs index 29bb77d80..95795cd5d 100644 --- a/igneum-pow/src/main.rs +++ b/igneum-pow/src/main.rs @@ -39,6 +39,9 @@ struct Args { prehash: String, epoch_hex: Option, day_hex: Option, + /// Class v5: the window's state stream file (`--state `, the IGSD1 format of `igneum_pow::state`), whose + /// leaves every item of the dataset is keyed by. + state: Option, class: LoadClass, /// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache). days: u64, @@ -101,7 +104,8 @@ fn usage() -> ! { \x20 --class C load class: v2 (default), mx4, mx8 (class v3: mixer x8, cache growth), dr (Counter ASIC 3.0 item 2: the per-day derivation program, dr736 = the x8-equivalent), w4, w16, w64, w64x4, p4,p16,p64[xN], m[g]\n\ \x20 also: w4, w16, w64, w64x4, p4,p16,p64[xN], m[g], +shx (latency-shadow block of S ALU instructions x R passes per iteration, Counter ASIC 3.0 item 8)\n\ \x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\ - \x20 --program-class v2|v3|v4 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, the chain's own derivation; --era-hex records the era seed)\n\ + \x20 --program-class v2|v3|v4|v5 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, v5 = generator 5 on V5_CLASS = mx8+sh256x27+state, the chain's own derivation; --era-hex records the era seed)\n\ + \x20 --state class v5 (or any --class ...+state): the window's state stream (IGSD1 file, igneum-day-stream --out), whose leaves key every item\n\ \x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\ \x20 --era E era layout over --class: igneum-era-test/ or :<64 hex> (the 32-byte era seed E_n)\n\ \x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)" @@ -116,6 +120,7 @@ fn parse() -> Args { day: "2026-10-03".into(), out: None, closed_form: false, + state: None, dataset_log2: DEFAULT_DATASET_LOG2, warps: 20, nonce: 0, @@ -150,6 +155,7 @@ fn parse() -> Args { "--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()), "--days" => a.days = val().parse().unwrap_or_else(|_| usage()), "--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())), + "--state" => a.state = Some(val()), "--era-hex" => a.era_hex = Some(val()), "--shadow-reps" => a.shadow_reps = val().parse().unwrap_or_else(|_| usage()), "--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())), @@ -216,6 +222,33 @@ fn main() { fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) { let (mut e, label) = epoch_of_class(a, mode); stamp_era(&mut e, a); + // class v5: the leaves of --state, built for the dataset's size; a state class without --state is refused here + // rather than at the first derivation + if e.program.class.state { + let Some(path) = &a.state else { + eprintln!("class {} keys every item by the window's state: give --state (igneum-day-stream --out)", e.program.class.name()); + std::process::exit(2); + }; + let stream = igneum_pow::state::StateStream::read_file(std::path::Path::new(path)).unwrap_or_else(|err| { + eprintln!("{err}"); + std::process::exit(2) + }); + let leaves = igneum_pow::state::StateLeaves::from_stream(&stream, e.dataset.log2_words); + eprintln!( + "state stream {}: chain block {} {}, root {}, {} records, {} leaves{}", + path, + stream.number, + igneum_pow::emit::hex_bytes(&stream.block), + igneum_pow::emit::hex_bytes(&stream.root), + stream.records.len(), + leaves.n(), + if leaves.sampled { " (sampled)" } else { "" } + ); + e.dataset = e.dataset.with_leaves(std::sync::Arc::new(leaves)); + } else if a.state.is_some() { + eprintln!("--state given for a class without state leaves ({}); use --program-class v5 or --class +state", e.program.class.name()); + std::process::exit(2); + } (e, label) } diff --git a/igneum-pow/src/memhard.rs b/igneum-pow/src/memhard.rs index ad42470b3..63a35ee50 100644 --- a/igneum-pow/src/memhard.rs +++ b/igneum-pow/src/memhard.rs @@ -12,6 +12,8 @@ use crate::derive::{run_round, DeriveProgram, SoaState, DERIVE_REGS, SOA_LANES}; use crate::generator::LoadClass; use crate::seed::{day_key, fnv1a64_words, SplitMix64}; +use crate::state::StateLeaves; +use std::sync::Arc; pub const CACHE_LOG2_WORDS: usize = 26; pub const CACHE_SEGMENT_LOG2_LINES: usize = 6; @@ -50,11 +52,14 @@ pub struct Shape { /// program, which replaces the `mixer_mult` applications of `M_r` in every mixer slot when non-zero. 0 for /// version 2 and class v3 (the fixed mixer). pub derive_len: u32, + /// Class v5 (`docs/design/class-v5-stored-state.md`, 7 October 2026): the item derivation XORs the window's state + /// leaf `leaf(t)` into the 16 initial words before the first mixer (`crate::state`). `false` for every other class. + pub state: bool, } impl Shape { /// Version 2: one mixer application per round, a 2^26-word cache. - pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32, derive_len: 0 }; + pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32, derive_len: 0, state: false }; /// The shape of a load class on day 0 of the chain (and on every day for a class without the growth rule). pub fn for_class(class: &LoadClass) -> Shape { @@ -68,6 +73,7 @@ impl Shape { mixer_mult: class.mixer_mult(), cache_log2_words: if class.growth { cache_log2_words(days_since_genesis) } else { CACHE_LOG2_WORDS as u32 }, derive_len: class.derive_len as u32, + state: class.state, } } @@ -200,6 +206,46 @@ pub struct MixParams { pub shape: Shape, /// The per-day derivation program when `shape.derive_len != 0`, else `None`. pub derive: Option, + /// Class v5: how many mixer blocks the AP-F4-1 rule redrew before this one (0 on every other class, and on most days). + pub redraws: u32, +} + +/// Class v5's mixer-draw rule (AP-F4-1): the NAF sum of the 16 multipliers at least this. +pub const MIXER_NAF_SUM_MIN: u32 = 163; +/// Class v5's mixer-draw rule: every multiplier's NAF weight at least this. +pub const MIXER_NAF_WORD_MIN: u32 = 4; +/// Class v5's mixer-draw rule: at least this many distinct rotation amounts among the eight. +pub const MIXER_DISTINCT_ROT_MIN: usize = 4; +/// Class v5's mixer-draw rule: redraws before the last block stands as drawn (never reached at 6.1e-4 per try). +pub const MIXER_REDRAW_CAP: u32 = 64; + +/// The non-adjacent-form weight of a 32-bit word: the number of non-zero digits of its NAF, the adders a +/// shift-and-add multiplier by that constant needs (the M1 metric of the weak-day census). +pub fn naf_weight(mut x: u64) -> u32 { + let mut w = 0; + while x != 0 { + if x & 1 == 1 { + w += 1; + // the digit is +1 or -1: take x to the nearest multiple of 4 + if x & 3 == 3 { + x += 1; + } else { + x -= 1; + } + } + x >>= 1; + } + w +} + +/// Whether a mixer block passes class v5's draw rule (AP-F4-1). +pub fn mixer_block_admissible(rot: &[u32; 8], mul: &[u32; 16]) -> bool { + let sum: u32 = mul.iter().map(|&m| naf_weight(m as u64)).sum(); + let words = mul.iter().all(|&m| naf_weight(m as u64) >= MIXER_NAF_WORD_MIN); + let mut distinct = rot.to_vec(); + distinct.sort_unstable(); + distinct.dedup(); + sum >= MIXER_NAF_SUM_MIN && words && distinct.len() >= MIXER_DISTINCT_ROT_MIN } impl MixParams { @@ -221,8 +267,28 @@ impl MixParams { for c in rc.iter_mut() { *c = rng.next() as u32; } + let mut redraws = 0u32; + if shape.state { + // Class v5 (docs/design/class-v5-stored-state.md section 11, AP-F4-1, the attack-pass lane's weak-day census): + // a mixer block whose multipliers are cheap on an adder datapath (NAF sum under 163, a word under NAF weight 4) + // or whose rotations repeat (under 4 distinct amounts) is redrawn from the next stream values, so no day is a + // weak day for a per-day LUT-recompute FPGA (the worst calendar day of the census, chain day 29,337, was 1.121x). + // About 6.1e-4 of days redraw. The derive program's draws (none under v5) come after, as before. + while !mixer_block_admissible(&rot, &mul) && redraws < MIXER_REDRAW_CAP { + for r in rot.iter_mut() { + *r = 1 + rng.below(31) as u32; + } + for m in mul.iter_mut() { + *m = (rng.next() as u32) | 1; + } + for c in rc.iter_mut() { + *c = rng.next() as u32; + } + redraws += 1; + } + } let derive = if shape.is_derived() { Some(DeriveProgram::draw(&mut rng, shape.derive_len)) } else { None }; - Self { key, rot, mul, rc, shape, derive } + Self { key, rot, mul, rc, shape, derive, redraws } } /// Parameters for a day string: the key is `seed_words("day/" + day)`. pub fn for_day(day: &str) -> Self { @@ -374,7 +440,7 @@ impl Cache { /// are the smaller cache's segments word for word. pub fn fill_log2(key: [u32; 8], log2_words: u32) -> Cache { assert!((10..=30).contains(&log2_words), "cache log2 words must be in 10..=30"); - let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words, derive_len: 0 }; + let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words, derive_len: 0, state: false }; let mut words = vec![0u32; shape.cache_words()]; for seg in 0..shape.cache_segments() { Self::fill_segment(&mut words, seg, &key); @@ -505,8 +571,21 @@ impl HotTable { /// (`mp.shape.mixer_mult`) round `r` applies `M` with keys `round_key(r m + j)` for `j = 0 .. m - 1` before its /// one cache read; the final mixer applies `M` with keys `round_key(8 m + j)`. `m = 1` is version 2. pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) { + derive_items_leaves(ts, mp, cache, None, out) +} + +/// [`derive_items`] with the state leaves of class v5 (`docs/design/class-v5-stored-state.md` section 2): under a +/// shape with `state`, `leaf(t)` is XORed into the 16 initial words of item `t` before the first mixer, and the leaves +/// are required; under any other shape they must be absent. A mismatch is a programming error and panics: a dataset +/// built without the state it needs would be wrong on every item, which is the class's point. +pub fn derive_items_leaves(ts: &[u32], mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { + match (mp.shape.state, leaves) { + (true, None) => panic!("class v5 item derivation needs the window's state leaves and was given none"), + (false, Some(_)) => panic!("state leaves given to an item derivation whose shape has no state"), + _ => {} + } if let Some(prog) = &mp.derive { - return derive_items_program(ts, mp, prog, cache, out); + return derive_items_program(ts, mp, prog, cache, leaves, out); } // The item loop lives in its own function, one instance per cache size the growth rule can reach with the line // mask a constant, never inlined into the callers. Inlined into `MemhardCpu::fetch` it ran at 1.33 ms per unit @@ -515,19 +594,19 @@ pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; // constant with the loop still inlined, all stayed at 1.33; the out-of-line instances read 0.60 to 0.62). Any // other cache size (tests) takes the instance with the run-time mask. match cache.log2_words { - 26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, out), - 27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, out), - 28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, out), - 29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, out), - 30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, out), - _ => derive_items_mask::<0>(ts, mp, cache, out), + 26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, leaves, out), + 27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, leaves, out), + 28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, leaves, out), + 29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, leaves, out), + 30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, leaves, out), + _ => derive_items_mask::<0>(ts, mp, cache, leaves, out), } } /// [`derive_items`] with the cache line mask as a constant (`LINE_MASK = 0`: the cache's own run-time mask). Kept /// out of line on purpose (see [`derive_items`]). #[inline(never)] -fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) { +fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { let n = ts.len(); debug_assert!(out.len() >= n); debug_assert!(LINE_MASK == 0 || LINE_MASK == cache.line_mask); @@ -539,6 +618,13 @@ fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &C for i in 0..8 { s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); } + if let Some(l) = leaves { + // class v5: the window's state leaf of item t, before the first mixer + let leaf = l.leaf(t); + for i in 0..16 { + s[i] ^= leaf[i]; + } + } } for r in 0..ITEM_ROUNDS { for j in 0..m { @@ -570,7 +656,7 @@ fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &C /// cache reads of the batch are issued together, as in the fixed-mixer loop, so the 8 dependent misses of /// independent items overlap in the memory system. #[inline(never)] -pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, cache: &Cache, out: &mut [[u32; 16]]) { +pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { let n = ts.len(); debug_assert!(out.len() >= n && n <= SOA_LANES); assert_eq!(prog.rounds.len(), ITEM_ROUNDS + 1); @@ -581,6 +667,12 @@ pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, ca st[i][k] = mp.key[i]; st[8 + i][k] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); } + if let Some(l) = leaves { + let leaf = l.leaf(t); + for i in 0..16 { + st[i][k] ^= leaf[i]; + } + } } let mask = cache.line_mask(); for r in 0..ITEM_ROUNDS { @@ -602,15 +694,22 @@ pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, ca /// One dataset item, 16 words. pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] { + derive_item_leaves(t, mp, cache, None) +} + +/// [`derive_item`] with the state leaves of class v5. +pub fn derive_item_leaves(t: u32, mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>) -> [u32; 16] { let mut out = [[0u32; 16]; 1]; - derive_items(&[t], mp, cache, &mut out); + derive_items_leaves(&[t], mp, cache, leaves, &mut out); out[0] } -/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape) and the cache. +/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape), the cache (shared, so +/// a class v5 window refresh keeps the day's 256 MiB and swaps the leaves) and, under class v5, the window's leaves. pub struct MemhardCpu { pub params: MixParams, - pub cache: Cache, + pub cache: Arc, + pub leaves: Option>, } /// Largest batch `MemhardCpu::fetch` accepts (two warps). @@ -622,7 +721,7 @@ impl MemhardCpu { Self::with_shape(key, Shape::V2) } pub fn with_shape(key: [u32; 8], shape: Shape) -> Self { - Self { params: MixParams::with_shape(key, shape), cache: Cache::fill_log2(key, shape.cache_log2_words) } + Self { params: MixParams::with_shape(key, shape), cache: Arc::new(Cache::fill_log2(key, shape.cache_log2_words)), leaves: None } } pub fn for_day(day: &str) -> Self { Self::new(day_key(day)) @@ -630,6 +729,17 @@ impl MemhardCpu { pub fn shape(&self) -> Shape { self.params.shape } + /// This view with the window's state leaves (class v5). The shape must have `state`. + pub fn with_leaves(mut self, leaves: Arc) -> Self { + assert!(self.params.shape.state, "state leaves on a shape without state"); + self.leaves = Some(leaves); + self + } + /// A view of the same day (the same cache, shared) with other leaves: the class v5 window refresh. + pub fn refreshed(&self, leaves: Arc) -> Self { + assert!(self.params.shape.state, "state leaves on a shape without state"); + Self { params: self.params.clone(), cache: self.cache.clone(), leaves: Some(leaves) } + } /// `dataset[w] = item(w >> 4)[w & 15]` (the linear layout). pub fn word(&self, w: u32) -> u32 { self.word_at(Layout::LINEAR, w) @@ -638,7 +748,7 @@ impl MemhardCpu { /// day's, so one cache serves every era of a day). pub fn word_at(&self, layout: Layout, w: u32) -> u32 { let (t, j) = layout.split(w); - derive_item(t, &self.params, &self.cache)[j as usize] + derive_item_leaves(t, &self.params, &self.cache, self.leaves.as_deref())[j as usize] } /// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once. /// Returns the number of distinct items derived. @@ -664,7 +774,7 @@ impl MemhardCpu { slot[k] = j as u8; } let mut items = [[0u32; 16]; FETCH_MAX]; - derive_items(&uniq[..u], &self.params, &self.cache, &mut items); + derive_items_leaves(&uniq[..u], &self.params, &self.cache, self.leaves.as_deref(), &mut items); for k in 0..n { out[k] = items[slot[k] as usize][word[k] as usize]; } @@ -695,7 +805,7 @@ impl MemhardCpu { slot[k] = j as u8; } let mut items = [[0u32; 16]; FETCH_MAX]; - derive_items(&uniq[..u], &self.params, &self.cache, &mut items); + derive_items_leaves(&uniq[..u], &self.params, &self.cache, self.leaves.as_deref(), &mut items); for k in 0..n { let o = word[k] as usize; out[k][..width].copy_from_slice(&items[slot[k] as usize][o..o + width]); @@ -708,6 +818,50 @@ impl MemhardCpu { mod tests { use super::*; + /// Class v5's mixer-draw rule (AP-F4-1), the known-failed case first: a block of cheap multipliers (NAF sum under + /// 163) or repeated rotations is inadmissible; a scan of day keys finds days the rule redraws (the census's 6.1e-4), + /// every v5 block passes after the draw, and the v4 constants of the same keys never move. + #[test] + fn class_v5_mixer_draw_rule() { + assert_eq!(naf_weight(0), 0); + assert_eq!(naf_weight(1), 1); + assert_eq!(naf_weight(3), 2, "11 = 100 - 1"); + assert_eq!(naf_weight(7), 2, "111 = 1000 - 1"); + assert_eq!(naf_weight(0xffff_ffff), 2); + assert_eq!(naf_weight(0b1010_1010), 4); + let good_rot = [1u32, 5, 9, 13, 17, 21, 25, 29]; + let cheap = [0x8000_0001u32; 16]; + assert!(!mixer_block_admissible(&good_rot, &cheap), "the known-failed case: 16 two-adder multipliers"); + let dense = [0xaaaa_aaabu32; 16]; + assert!(mixer_block_admissible(&good_rot, &dense)); + assert!(!mixer_block_admissible(&[7u32; 8], &dense), "one rotation amount"); + assert!(!mixer_block_admissible(&[1u32, 2, 3, 3, 3, 3, 3, 3], &dense), "three distinct amounts"); + let v5 = Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: true }; + let v4 = Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false }; + let mut redrawn = 0; + let mut scanned = 0; + for d in 0..60_000u64 { + let key = crate::seed::seed_words_from_bytes(&crate::bind::day_bytes(20_000 + d)); + let a = MixParams::with_shape(key, v5); + assert!(mixer_block_admissible(&a.rot, &a.mul), "day {d}: a v5 block fails the rule after the draw"); + if a.redraws > 0 { + redrawn += 1; + let b = MixParams::with_shape(key, v4); + assert_eq!(b.redraws, 0, "v4 never redraws"); + assert_ne!((a.rot, a.mul), (b.rot, b.mul), "day {d}: v5 redrew, v4 kept the block"); + assert!(!mixer_block_admissible(&b.rot, &b.mul), "day {d}: the v4 block was the inadmissible one"); + } else { + let b = MixParams::with_shape(key, v4); + assert_eq!((a.rot, a.mul, a.rc), (b.rot, b.mul, b.rc), "day {d}: an admissible day is byte for byte v4's"); + } + scanned += 1; + if redrawn >= 3 && scanned >= 2_000 { + break; + } + } + assert!(redrawn >= 1, "no redraw in {scanned} days (the census says about 6.1e-4 per day)"); + } + #[test] fn mix_params_for_day() { // MEMHARD.md section 1.4 and the igneum-genesis-mh pack. @@ -851,7 +1005,7 @@ mod tests { let v2 = Shape::for_class_day(&LoadClass::V2, 100_000); assert_eq!(v2, Shape::V2); let v3 = Shape::for_class_day(&LoadClass::MX4, 0); - assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0 }); + assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false }); assert_eq!(Shape::for_class_day(&LoadClass::MX4, 1_460).cache_log2_words, 27); assert_eq!(v3.mixers_per_item(), 36); assert_eq!(Shape::V2.mixers_per_item(), 9); @@ -872,7 +1026,7 @@ mod tests { assert_eq!(small.segments(), 64); assert_eq!(small.line_mask(), 4095); for m in [1u32, 2, 4] { - let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16, derive_len: 0 }); + let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16, derive_len: 0, state: false }); for t in [0u32, 1, 12_345, u32::MAX] { let got = derive_item(t, &mp, &small); let mut s = [0u32; 16]; @@ -895,8 +1049,8 @@ mod tests { assert_eq!(got, s, "m {m} t {t}"); } } - let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0 }); - let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16, derive_len: 0 }); + let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false }); + let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16, derive_len: 0, state: false }); assert_ne!(derive_item(0, &v2, &small), derive_item(0, &v3, &small)); assert_eq!(round_key_mult(0, 0, 1), round_key(0)); assert_eq!(round_key_mult(8, 0, 1), round_key(8)); diff --git a/igneum-pow/src/packcheck.rs b/igneum-pow/src/packcheck.rs index ee0490662..a7cbea7f5 100644 --- a/igneum-pow/src/packcheck.rs +++ b/igneum-pow/src/packcheck.rs @@ -29,6 +29,8 @@ pub struct PackIdentity { pub class: ProgramClass, /// `IGNEUM_ERA_SEED_HEX` when the pack carries one (class v3 chain packs). pub era_hex: Option, + /// `IGNEUM_STATE_ROOT_HEX` of a class v5 pack (the window's state root the leaves derive from). + pub state_root_hex: Option, } /// Why a pack is not the one a worker should mine with. `Display` is the plain-words line the logs carry. @@ -193,8 +195,18 @@ pub fn verify_pack_texts_chain( // without the block is no v4 pack. A generator 2 pack with a shadow (the measurement ladder of // proto-cuda/packs-ca3-shadow) carries a class-bearing id and stays loadable. let shadow = define_u32(program_h, "IGNEUM_SHADOW_INSTRS").unwrap_or(0); + // Class v5 (docs/design/class-v5-stored-state.md): the state lines are the mark of class v5, so a generator 5 pack + // carries IGNEUM_STATE_ROOT_HEX (and the shadow block of v4) and no other generator does. + let state_root_hex = define_str(program_h, "IGNEUM_STATE_ROOT_HEX"); + match (class, state_root_hex.is_some()) { + (ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_STATE_ROOT_HEX: not a class v5 pack".into())), + (ProgramClass::V5, true) => {} + (_, true) => return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR {generator} with class v5 state lines: a program over state leaves is generator 5 (export the pack as class v5)"))), + _ => {} + } match (class, shadow > 0) { (ProgramClass::V4, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 4 (class v4) without IGNEUM_SHADOW_INSTRS: not a class v4 pack".into())), + (ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_SHADOW_INSTRS: not a class v5 pack".into())), // the class v4 stream sub-version (AP-F8-1 amendment, 7 October 2026): a generator 4 pack from before the // load-source rule carries no IGNEUM_PROGRAM_SUBVERSION and its program id is another stream's; refused (ProgramClass::V4, true) if define_u32(program_h, "IGNEUM_PROGRAM_SUBVERSION") != Some(u32::from(crate::generator::PROGRAM_SUBVERSION_V4)) => { @@ -265,7 +277,7 @@ pub fn verify_pack_texts_chain( if epoch_hex != want_epoch_hex || day_hex != want_day_hex { return Err(PackFault::OutOfDate { pack_epoch: epoch_hex, pack_day: day_hex, want_epoch: want_epoch_hex, want_day: want_day_hex }); } - Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex }) + Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex, state_root_hex }) } /// [`verify_pack_texts`] over a pack directory. diff --git a/igneum-pow/src/state.rs b/igneum-pow/src/state.rs new file mode 100644 index 000000000..4298b7fde --- /dev/null +++ b/igneum-pow/src/state.rs @@ -0,0 +1,287 @@ +//! Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026): the +//! leaves the item derivation XORs in (section 2 of the page), built from the canonical state stream of the +//! window's reference block. +//! +//! `D[i] = Blake2b-512("igneum-sd1/" || root || i_le32 || record_i)` for the `n` records of the stream, and item +//! `t` takes `leaf(t) = D[t mod n]`: every item is keyed by the state, so a hasher without it is wrong on every +//! item (the known-failed case, the first test). When the stream has more records than the dataset has items, the +//! records are ordered by `Blake2b-256("igneum-sd1-sample/" || root || record)` and the first `items` are taken, a +//! sample nobody can choose without the whole state and the root. +//! +//! The stream file (`StateStream`): the plain format every side reads without a serialisation library, `IGSD1\0`, +//! the chain block number (le64) and hash (32), the state root (32), the record count (le32), then each record as +//! its length (le32) and bytes. The node's executor writes it (`igneum/exec/src/day_stream.rs`), the miner fetches +//! it, the CLI's `--state` reads it, and a pack carries the leaves it yields as `leaves.bin`. + +use crate::blake2b::{blake2b_256, blake2b_512}; +use crate::seed::fnv1a64_words; + +pub const LEAF_TAG: &[u8] = b"igneum-sd1/"; +pub const SAMPLE_TAG: &[u8] = b"igneum-sd1-sample/"; +pub const STREAM_MAGIC: &[u8; 6] = b"IGSD1\0"; + +/// The canonical state stream at one chain block: what the executor serialises and what the leaves derive from. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct StateStream { + pub number: u64, + pub block: [u8; 32], + pub root: [u8; 32], + pub records: Vec>, +} + +impl StateStream { + pub fn encode(&self) -> Vec { + let mut b = Vec::with_capacity(6 + 8 + 32 + 32 + 4 + self.records.iter().map(|r| 4 + r.len()).sum::()); + b.extend_from_slice(STREAM_MAGIC); + b.extend_from_slice(&self.number.to_le_bytes()); + b.extend_from_slice(&self.block); + b.extend_from_slice(&self.root); + b.extend_from_slice(&(self.records.len() as u32).to_le_bytes()); + for r in &self.records { + b.extend_from_slice(&(r.len() as u32).to_le_bytes()); + b.extend_from_slice(r); + } + b + } + + pub fn decode(bytes: &[u8]) -> Result { + if bytes.len() < 6 + 8 + 32 + 32 + 4 || &bytes[..6] != STREAM_MAGIC { + return Err("not a state stream file (magic IGSD1)".into()); + } + let mut at = 6; + let number = u64::from_le_bytes(bytes[at..at + 8].try_into().unwrap()); + at += 8; + let block: [u8; 32] = bytes[at..at + 32].try_into().unwrap(); + at += 32; + let root: [u8; 32] = bytes[at..at + 32].try_into().unwrap(); + at += 32; + let n = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize; + at += 4; + let mut records = Vec::with_capacity(n.min(1 << 20)); + for i in 0..n { + if at + 4 > bytes.len() { + return Err(format!("state stream truncated at record {i} of {n}")); + } + let len = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize; + at += 4; + if at + len > bytes.len() { + return Err(format!("state stream truncated inside record {i} of {n}")); + } + records.push(bytes[at..at + len].to_vec()); + at += len; + } + if at != bytes.len() { + return Err(format!("state stream has {} trailing bytes", bytes.len() - at)); + } + Ok(StateStream { number, block, root, records }) + } + + pub fn read_file(path: &std::path::Path) -> Result { + let bytes = std::fs::read(path).map_err(|e| format!("read {}: {e}", path.display()))?; + Self::decode(&bytes) + } +} + +/// `D[i]`: the 64-byte digest of record `i` under `root`, as 16 little-endian words. +pub fn leaf_digest(root: &[u8; 32], i: u32, record: &[u8]) -> [u32; 16] { + let d = blake2b_512(&[LEAF_TAG, root, &i.to_le_bytes(), record]); + let mut w = [0u32; 16]; + for (k, x) in w.iter_mut().enumerate() { + *x = u32::from_le_bytes(d[k * 4..k * 4 + 4].try_into().unwrap()); + } + w +} + +/// The sample order key of a record under `root`. +pub fn sample_key(root: &[u8; 32], record: &[u8]) -> [u8; 32] { + blake2b_256(&[SAMPLE_TAG, root, record]) +} + +/// The leaves of one window (or day) of class v5: `n` digests of 64 bytes, `leaf(t) = D[t mod n]`. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct StateLeaves { + pub root: [u8; 32], + pub block: [u8; 32], + pub number: u64, + /// Records in the stream before any sample. + pub records_total: u64, + /// Whether the stream had more records than the dataset has items (the sample rule applied). + pub sampled: bool, + leaves: Vec<[u32; 16]>, +} + +impl StateLeaves { + /// The items a dataset of `2^log2_words` words has: `2^(log2_words - 4)`. + pub fn items_of(log2_words: u32) -> u64 { + 1u64 << log2_words.saturating_sub(4) + } + + /// The leaves of `records` (canonical order) under `root` for a dataset of `2^log2_words` words. An empty stream + /// yields one leaf, the digest of the empty record, so `n` is never 0. + pub fn build(root: [u8; 32], block: [u8; 32], number: u64, records: &[Vec], log2_words: u32) -> StateLeaves { + let items = Self::items_of(log2_words); + let records_total = records.len() as u64; + let empty: Vec> = vec![Vec::new()]; + let records = if records.is_empty() { &empty[..] } else { records }; + let sampled = records.len() as u64 > items; + let chosen: Vec<&Vec> = if sampled { + let mut keyed: Vec<([u8; 32], &Vec)> = records.iter().map(|r| (sample_key(&root, r), r)).collect(); + keyed.sort_unstable_by(|a, b| a.0.cmp(&b.0).then_with(|| a.1.cmp(b.1))); + keyed.into_iter().take(items as usize).map(|(_, r)| r).collect() + } else { + records.iter().collect() + }; + let leaves = chosen.iter().enumerate().map(|(i, r)| leaf_digest(&root, i as u32, r)).collect(); + StateLeaves { root, block, number, records_total, sampled, leaves } + } + + pub fn from_stream(s: &StateStream, log2_words: u32) -> StateLeaves { + Self::build(s.root, s.block, s.number, &s.records, log2_words) + } + + /// Leaves from the raw words of a `leaves.bin` (16 words per leaf), for a worker or a test that holds no stream. + pub fn from_words(root: [u8; 32], block: [u8; 32], number: u64, words: &[u32]) -> StateLeaves { + assert!(!words.is_empty() && words.len() % 16 == 0, "leaves are 16 words each"); + let leaves = words.chunks_exact(16).map(|c| c.try_into().unwrap()).collect::>(); + StateLeaves { root, block, number, records_total: leaves.len() as u64, sampled: false, leaves } + } + + #[inline(always)] + pub fn n(&self) -> u32 { + self.leaves.len() as u32 + } + + /// `leaf(t) = D[t mod n]`. + #[inline(always)] + pub fn leaf(&self, t: u32) -> &[u32; 16] { + &self.leaves[(t % self.n()) as usize] + } + + pub fn leaves(&self) -> &[[u32; 16]] { + &self.leaves + } + + /// The flat words of `leaves.bin`. + pub fn words(&self) -> Vec { + self.leaves.iter().flat_map(|l| l.iter().copied()).collect() + } + + /// The bytes of `leaves.bin` (little-endian words). + pub fn bytes(&self) -> Vec { + self.words().iter().flat_map(|w| w.to_le_bytes()).collect() + } + + /// FNV-1a 64 over the leaves as little-endian bytes (the pack's `IGNEUM_STATE_LEAVES_FNV64`). + pub fn fnv1a64(&self) -> u64 { + fnv1a64_words(&self.words()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::generator::{generate_class, V5_CLASS}; + use crate::memhard::{derive_item_leaves, Cache, MixParams, Shape}; + use crate::seed::day_key; + use crate::verify::{hash_warp, DatasetMode, DatasetSource}; + use std::sync::Arc; + + fn records(n: usize, salt: u8) -> Vec> { + (0..n).map(|i| vec![salt, i as u8, (i >> 8) as u8, 7]).collect() + } + + fn leaves(root: u8, n: usize, log2_words: u32) -> Arc { + Arc::new(StateLeaves::build([root; 32], [0x22; 32], 5, &records(n, root), log2_words)) + } + + /// The known-failed case, first: a hasher without the state (no leaves, the leaves of another root, the leaves + /// of a stream one record short, the previous window's leaves) is wrong on every item and every lane. + #[test] + fn a_stateless_hasher_is_wrong_on_every_item() { + let key = day_key("2026-10-03"); + let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true }; + let cache = Arc::new(Cache::fill_log2(key, 16)); + let mp = MixParams::with_shape(key, shape); + let good = leaves(0x11, 93, 20); + let other_root = leaves(0x12, 93, 20); + let one_short = Arc::new(StateLeaves::build([0x13; 32], [0x22; 32], 5, &records(92, 0x11), 20)); // a record short means another root + let previous_window = leaves(0x10, 93, 20); + for (name, bad) in [("another root", other_root.clone()), ("one record short", one_short.clone()), ("the previous window", previous_window.clone())] { + let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp, &cache, Some(&bad))).count(); + assert_eq!(equal, 0, "{name}: {equal} of 64 items equal"); + } + let stateless = Shape { state: false, ..shape }; + let mp_stateless = MixParams::with_shape(key, stateless); + let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp_stateless, &cache, None)).count(); + assert_eq!(equal, 0, "no leaves at all: {equal} of 64 items equal"); + // the warp: a class v5 program over a small dataset, the same program and cache, other leaves + let program = generate_class("igneum-genesis", V5_CLASS); + let ds = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(good.clone()); + let ds_other = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(previous_window.clone()); + let a = hash_warp(&program, 0, &ds); + let b = hash_warp(&program, 0, &ds_other); + assert_eq!(a.iter().zip(b.iter()).filter(|(x, y)| x == y).count(), 0, "0 of 32 lanes agree"); + assert_eq!(hash_warp(&program, 0, &ds), a, "the same leaves hash the same"); + } + + /// Every item takes a leaf: `leaf(t) = D[t mod n]`, so items `t` and `t + n` share a leaf and still differ. + #[test] + fn every_item_is_keyed_and_the_leaf_wraps() { + let l = leaves(0x11, 93, 28); + assert_eq!(l.n(), 93); + assert!(!l.sampled); + assert_eq!(l.records_total, 93); + for t in [0u32, 1, 92, 93, 94, 1_000_000, u32::MAX] { + assert_eq!(l.leaf(t), l.leaf(t % 93)); + assert_eq!(*l.leaf(t), leaf_digest(&[0x11; 32], t % 93, &records(93, 0x11)[(t % 93) as usize])); + } + let key = day_key("2026-10-03"); + let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true }; + let cache = Cache::fill_log2(key, 16); + let mp = MixParams::with_shape(key, shape); + assert_ne!(derive_item_leaves(5, &mp, &cache, Some(&l)), derive_item_leaves(5 + 93, &mp, &cache, Some(&l))); + // an empty stream yields one leaf (the digest of the empty record), never a division by zero + let empty = StateLeaves::build([0x11; 32], [0; 32], 0, &[], 28); + assert_eq!(empty.n(), 1); + assert_eq!(empty.records_total, 0); + assert_eq!(*empty.leaf(12_345), leaf_digest(&[0x11; 32], 0, &[])); + } + + /// Above the dataset size the records are sampled in the keyed order: a different root picks a different set, + /// and the set cannot be the first `items` records of the stream. + #[test] + fn the_sample_above_the_dataset_size_is_keyed_by_the_root() { + let recs = records(40, 0x33); + let a = StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8); + let b = StateLeaves::build([0x12; 32], [0; 32], 0, &recs, 8); + assert_eq!(StateLeaves::items_of(8), 16); + assert_eq!((a.n(), a.sampled, a.records_total), (16, true, 40)); + assert_ne!(a.leaves(), b.leaves(), "another root, another sample"); + // the positional first 16 are not the sample (with overwhelming probability for 40 choose 16) + let positional = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8); + assert_ne!(a.leaves(), positional.leaves()); + // the same inputs sample the same + assert_eq!(StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8), a); + // at the dataset size exactly, no sample + let c = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8); + assert!(!c.sampled && c.n() == 16); + } + + #[test] + fn stream_file_round_trip_and_refusals() { + let s = StateStream { number: 159_357, block: [0xaf; 32], root: [0x1c; 32], records: records(93, 1) }; + let bytes = s.encode(); + assert_eq!(&bytes[..6], STREAM_MAGIC); + assert_eq!(StateStream::decode(&bytes).unwrap(), s); + assert!(StateStream::decode(&bytes[..bytes.len() - 1]).is_err(), "truncated"); + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(StateStream::decode(&trailing).is_err(), "trailing bytes"); + assert!(StateStream::decode(b"IGSD0\0").is_err(), "wrong magic"); + let l = StateLeaves::from_stream(&s, 28); + let back = StateLeaves::from_words(s.root, s.block, s.number, &l.words()); + assert_eq!(back.leaves(), l.leaves()); + assert_eq!(l.bytes().len(), 93 * 64); + assert_eq!(l.fnv1a64(), back.fnv1a64()); + } +} diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index 34692b684..9d1281aeb 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -227,6 +227,42 @@ impl DatasetSource { Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None } } + /// This source with the window's state leaves (class v5, `docs/design/class-v5-stored-state.md`): memory-hard mode + /// under a shape with `state` only. + pub fn with_leaves(mut self, leaves: std::sync::Arc) -> Self { + match &mut self.dataset { + Dataset::MemoryHard(m) => { + assert!(m.params.shape.state, "state leaves on a dataset whose shape has no state"); + m.leaves = Some(leaves); + } + Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"), + } + self + } + + /// A source of the same day with other leaves, the 256 MiB cache shared (the class v5 window refresh). + pub fn refreshed(&self, leaves: std::sync::Arc) -> Self { + let dataset = match &self.dataset { + Dataset::MemoryHard(m) => Dataset::MemoryHard(m.refreshed(leaves)), + Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"), + }; + Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + } + + /// A copy of this source sharing its cache (and leaves), for a caller that needs an owned source from a shared one. + pub fn refreshed_or_clone(&self) -> Self { + let dataset = match &self.dataset { + Dataset::MemoryHard(m) => Dataset::MemoryHard(crate::memhard::MemhardCpu { params: m.params.clone(), cache: m.cache.clone(), leaves: m.leaves.clone() }), + Dataset::ClosedForm { d0, d1 } => Dataset::ClosedForm { d0: *d0, d1: *d1 }, + }; + Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + } + + /// The window's state leaves, when the source carries them. + pub fn leaves(&self) -> Option<&std::sync::Arc> { + self.memhard().and_then(|m| m.leaves.as_ref()) + } + /// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB). pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self { self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb)); diff --git a/igneum-pow/tests/derive.rs b/igneum-pow/tests/derive.rs index 66b58df74..a64af0292 100644 --- a/igneum-pow/tests/derive.rs +++ b/igneum-pow/tests/derive.rs @@ -43,7 +43,7 @@ fn item_by_hand(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] { fn derived_item_by_hand_and_in_batches() { let key = day_key(DAY); let cache = Cache::fill_log2(key, 16); - let shape = Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 }; + let shape = Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }; let mp = MixParams::with_shape(key, shape); let prog = mp.derive.as_ref().unwrap(); assert_eq!(prog.rounds.len(), DERIVE_PROGRAMS); @@ -64,7 +64,7 @@ fn derived_item_by_hand_and_in_batches() { derive_items(&ts[..5], &mp, &cache, &mut out5); assert_eq!(&out5[..], &out[..5]); // the fixed mixer of the same key gives other items - let v3 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0 }); + let v3 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: false }); assert!(v3.derive.is_none()); assert_ne!(derive_item(0, &v3, &cache), derive_item(0, &mp, &cache)); } @@ -79,7 +79,7 @@ fn v2_and_v3_are_untouched() { let key = day_key(DAY); let cache = Cache::fill_log2(key, 16); // the version 2 item restated by hand (the mixer_mult_by_hand test of memhard.rs, m = 1) - let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0 }); + let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false }); let t = 12_345u32; let mut s = [0u32; 16]; s[..8].copy_from_slice(&key); @@ -96,7 +96,7 @@ fn v2_and_v3_are_untouched() { mixer(&mut s, round_key(8), &v2); assert_eq!(derive_item(t, &v2, &cache), s); // the mixer constants of the derivation class are the v2 draws (the stream continues after them) - let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 }); + let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }); assert_eq!((dr.rot, dr.mul, dr.rc), (v2.rot, v2.mul, v2.rc)); } @@ -109,12 +109,12 @@ fn stream_class_name_and_id() { rng.next(); } let expect = DeriveProgram::draw(&mut rng, DERIVE_LEN_X8); - let mp = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8 }); + let mp = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false }); assert_eq!(mp.derive.as_ref().unwrap(), &expect); // another day, another program; another length, another program - let other = MixParams::with_shape(day_key("2026-10-04"), Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8 }); + let other = MixParams::with_shape(day_key("2026-10-04"), Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false }); assert_ne!(other.derive.as_ref().unwrap().fingerprint(), expect.fingerprint()); - let short = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: 368 }); + let short = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: 368, state: false }); assert_eq!(short.derive.as_ref().unwrap().instr_count(), 9 * 368); // the class: name, parse, id, and the v2 program stream (v2 loads, no width roll) let c = LoadClass::DR736; @@ -179,8 +179,8 @@ fn determinism_and_pack_text() { fn stats_beside_x8() { let key = day_key(DAY); let cache = Cache::fill_log2(key, 18); - let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 18, derive_len: DERIVE_LEN_X8 }); - let x8 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 18, derive_len: 0 }); + let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 18, derive_len: DERIVE_LEN_X8, state: false }); + let x8 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 18, derive_len: 0, state: false }); for (label, mp) in [("dr736", &dr), ("x8", &x8)] { let n = 2048u32; let mut ones = [0u32; 512]; @@ -265,7 +265,7 @@ fn text_forms_match_scalar_reference() { /// The dataset source of the class on a day: the verifier's `word` path derives through the program. #[test] fn dataset_source_word_path() { - let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 }); + let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }); let m = ds.memhard().unwrap(); let item = derive_item(3, &m.params, &m.cache); for j in 0..16u32 { diff --git a/igneum-pow/tests/mixer.rs b/igneum-pow/tests/mixer.rs index 61e32aa05..f4644ae8d 100644 --- a/igneum-pow/tests/mixer.rs +++ b/igneum-pow/tests/mixer.rs @@ -275,7 +275,7 @@ fn edge_items_every_multiplier() { let key = day_key(DAY); let cache = Cache::fill_log2(key, 14); for m in [1u32, 2, 4, 8] { - let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14, derive_len: 0 }); + let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14, derive_len: 0, state: false }); let by_hand = |t: u32| -> [u32; 16] { let mut s = [0u32; 16]; s[..8].copy_from_slice(&key); diff --git a/igneum-pow/tests/packs.rs b/igneum-pow/tests/packs.rs index 98446833d..f66d91536 100644 --- a/igneum-pow/tests/packs.rs +++ b/igneum-pow/tests/packs.rs @@ -369,7 +369,7 @@ fn v3_packs_are_the_v2_seeds_under_mixer_x8() { assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt)); let m3 = e3.dataset.memhard().unwrap(); let m2 = e2.dataset.memhard().unwrap(); - assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0 }); + assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false }); assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0"); assert_eq!(m3.params.rot, m2.params.rot); assert_eq!(e3.dataset.log2_words, 28); @@ -405,7 +405,7 @@ fn v3_packs_are_the_v2_seeds_under_mixer_x8() { assert_eq!(j["load_class"].as_str().unwrap(), "mx4"); assert_eq!(e4.program.class, LoadClass::MX4); assert_eq!(e4.program.instrs, epoch(v2).program.instrs); - assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0 }); + assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false }); assert!(read(x4, "memhard.h").contains("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))")); } } @@ -899,3 +899,101 @@ fn hot_packs_emitted_sources_and_load_forms() { assert!(ph.contains("igneum_launch_hot_fill(")); } } + +// --------------------------------------------------------------------------------------------------------------- +// Class v5 (docs/design/class-v5-stored-state.md, 7 October 2026): the pinned pack under proto-cuda/packs-ca3-v5/ +// --------------------------------------------------------------------------------------------------------------- + +fn v5_packs_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca3-v5") +} + +fn v5_read(pack: &str, file: &str) -> String { + std::fs::read_to_string(v5_packs_dir().join(pack).join(file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")) +} + +fn v5_json(pack: &str, file: &str) -> Value { + serde_json::from_str(&v5_read(pack, file)).unwrap() +} + +/// The epoch of a pinned class v4 or v5 pack of the string seed and day: the program through the seam, the dataset at +/// the day-0 size, and for a v5 pack the leaves of its `state.igsd1` (the devnet's state stream of 7 October 2026, +/// node 1's exec snapshot at chain block 159,357: 93 records, root 0x1c583d35...). +fn v5_epoch(pack: &str) -> Epoch { + let j = v5_json(pack, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let class = ProgramClass::parse(j["program_class"].as_str().unwrap()).unwrap(); + let program = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), class, None); + let day = j["dataset"]["day"].as_str().unwrap(); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let shape = Shape::for_class(&program.class); + let mut dataset = DatasetSource::new_shape(day, DatasetMode::MemoryHard, log2, shape); + if class == ProgramClass::V5 { + let stream = igneum_pow::StateStream::read_file(&v5_packs_dir().join(pack).join("state.igsd1")).unwrap(); + dataset = dataset.with_leaves(std::sync::Arc::new(igneum_pow::StateLeaves::from_stream(&stream, log2))); + } + Epoch { program, dataset } +} + +/// The class v5 pack is the class v4 program of the same seed over the state leaves: generator 5 and +/// `program_id(5, seed, attempt)`, the base program, the shadow block and the dataset's cache equal to the v4 pack's, +/// every vector and dataset word different, every emitted file byte for byte what the crate exports (leaves.bin +/// included, its FNV in program.h), and the hash kernel text of kernel.cu equal to the v4 pack's but for the build +/// kernel and the class lines. The known-failed case first: the v4 control pack's vectors under the v5 epoch agree on +/// no lane. +#[test] +fn v5_pack_is_the_v4_program_over_the_state_leaves() { + let e5 = v5_epoch("v5-genesis"); + let e4 = v5_epoch("v4-genesis"); + let v4 = v5_json("v4-genesis", "vectors.json"); + // the known-failed case: the v4 pack's hashes are not the v5 epoch's on any lane + let out4: Vec = v4["warps"][0]["expected"].as_array().unwrap().iter().map(hex64).collect(); + let got5 = e5.hash_warp(0); + assert_eq!(out4.iter().zip(got5.iter()).filter(|(a, b)| a == b).count(), 0, "0 of 32 lanes of the v4 pack agree with the v5 epoch"); + assert_eq!(e4.hash_warp(0).to_vec(), out4, "the v4 control pack is the v4 epoch"); + // the program: v4's draw, generator 5, the state in the class and the id + assert_eq!(e5.program.generator, igneum_pow::GENERATOR_VERSION_V5); + assert_eq!(e5.program.class, igneum_pow::V5_CLASS); + assert_eq!(e5.program.class.name(), "mx8+sh256x27+state"); + assert_eq!(e5.program.instrs, e4.program.instrs); + assert_eq!(e5.program.shadow, e4.program.shadow); + assert_eq!((e5.program.seed, e5.program.attempt), (e4.program.seed, e4.program.attempt)); + assert_eq!(e5.program.program_id(), igneum_pow::generator::program_id(igneum_pow::GENERATOR_VERSION_V5, &e5.program.seed, e5.program.attempt)); + assert_ne!(e5.program.program_id(), e4.program.program_id()); + // the dataset: the same cache, other items + let m5 = e5.dataset.memhard().unwrap(); + let m4 = e4.dataset.memhard().unwrap(); + assert_eq!(m5.cache.fnv1a64(), m4.cache.fnv1a64(), "one day cache"); + assert!(m5.shape().state && !m4.shape().state); + let leaves = e5.dataset.leaves().unwrap(); + assert_eq!((leaves.n(), leaves.records_total, leaves.sampled), (93, 93, false)); + assert_ne!(e5.dataset_word(0), e4.dataset_word(0)); + // every file as the crate exports it, leaves.bin included + for pack in ["v4-genesis", "v5-genesis"] { + let e = if pack == "v5-genesis" { &e5 } else { &e4 }; + let v = v5_json(pack, "vectors.json"); + let out = export_pack(e, v["day"].as_str().unwrap(), v["source"].as_str().unwrap()); + for (name, text) in &out.files { + assert_eq!(v5_read(pack, name), *text, "{pack}/{name} differs from the export"); + } + for (name, bytes) in &out.binaries { + assert_eq!(std::fs::read(v5_packs_dir().join(pack).join(name)).unwrap(), *bytes, "{pack}/{name}"); + } + assert_eq!(out.binaries.len(), (pack == "v5-genesis") as usize); + } + let h5 = v5_read("v5-genesis", "program.h"); + assert!(h5.contains("#define IGNEUM_PROGRAM_CLASS \"v5\"") && h5.contains("#define IGNEUM_STATE_LEAVES 93") && h5.contains(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}", igneum_pow::emit::hex64(leaves.fnv1a64())))); + assert!(h5.contains("igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems)")); + // the hash kernel text is class v4's byte for byte; the build kernel and the class lines are what differ + let hash_text = |s: &str| { + let a = s.find("__global__ void igneum_hash(").unwrap(); + let b = s.find("// Host-side launch wrappers").unwrap(); + s[a..b].to_string() + }; + let k5 = v5_read("v5-genesis", "kernel.cu"); + let k4 = v5_read("v4-genesis", "kernel.cu"); + assert_eq!(hash_text(&k5), hash_text(&k4), "the hash kernel is class v4's"); + assert!(k5.contains("mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s)") && !k4.contains("mh_leaf")); + let mh5 = v5_read("v5-genesis", "memhard.h"); + assert!(mh5.contains("s[i] ^= leaf[i]"), "the leaf XOR before the first mixer"); +}