diff --git a/igneum-pow/src/accept.rs b/igneum-pow/src/accept.rs index a3a59d7a4..198871a28 100644 --- a/igneum-pow/src/accept.rs +++ b/igneum-pow/src/accept.rs @@ -11,12 +11,14 @@ //! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) | //! //! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and -//! costs about a millisecond on one core. The census (section 7.3) checked on 100,000 programs that the +//! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by +//! seed words 2 and 3 at its multiply-shift index, a second pure table beside the dataset stand-in (words 0 and 1). The census (section 7.3) checked on 100,000 programs that the //! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases. use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES}; use crate::seed::{fnv1a64, SplitMix64}; -use crate::verify::{dataset_elem, splitmix32}; +use crate::memhard::hot_index; +use crate::verify::{dataset_elem, fold_words, load_index, splitmix32, ScratchModel}; /// Units (32-lane warps) the dynamic test interprets. pub const ACCEPT_UNITS: usize = 64; @@ -33,6 +35,14 @@ pub const BIAS_TOLERANCE: u32 = 136; /// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120). pub const MIN_DISTINCT_SUM: u64 = 245_760; +/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so +/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts. +/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later +/// read-modify-write sees an earlier write), so they are neither counted nor bounded here. +pub fn min_distinct_sum(loads: usize) -> u64 { + loads as u64 * ACCEPT_HASHES as u64 * 120 / 128 +} + /// Why a candidate was rejected. The verdict (accept or reject) is what consensus depends on; the reason is the /// first failing test in the order of the module table. #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -67,7 +77,7 @@ impl std::fmt::Display for Reject { Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"), Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"), Reject::DistinctAddresses { sum } => { - write!(f, "(c) distinct addresses {sum} over 2048 hashes (mean {:.2}, needs above 120)", *sum as f64 / 2048.0) + write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0) } } } @@ -170,6 +180,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut let seed = &p.seed; let mask: u32 = (1u32 << ACCEPT_DATASET_LOG2) - 1; let (d0, d1) = (seed[0], seed[1]); + let (h0, h1) = (seed[2], seed[3]); + let hot_words = p.hot_words(); let loads = p.loads_per_hash(); let mut r = [[0u32; LANES]; 8]; for lane in 0..LANES { @@ -183,12 +195,30 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut } let mut idx = [0u32; LANES]; let mut nload = 0usize; + let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None }; + let slot_mask = p.class.scratch_slot_mask(); + let era = p.class.era; for it in 0..ITERATIONS { let sel = r[0]; for (k, ins) in p.instrs.iter().enumerate() { let d = ins.dst as usize; let a = ins.src as usize; match ins.op { + Op::Scratch => { + // Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word). + let m = scratch.as_mut().expect("a scratch op needs a scratch class"); + for lane in 0..LANES { + idx[lane] = r[a][lane] & slot_mask; + } + if idx.iter().all(|&x| x == idx[0]) { + return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); + } + for lane in 0..LANES { + r[d][lane] = m.rmw(&p.seed, base, lane, idx[lane], r[d][lane]); + lane_addrs[lane * loads + nload] = 0x8000_0000 | idx[lane]; + } + nload += 1; + } Op::Add => { let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32); let src = r[a]; @@ -255,18 +285,45 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut } } Op::Load => { + // Read-width experiment: a load of `width` words reads from the aligned address and folds every + // word (verify::fold_words); width 1 is the lottery hash's xor of one word. + let width = ins.width as usize; + let align = !(ins.width as u32 - 1); for lane in 0..LANES { - idx[lane] = r[a][lane] & mask; + idx[lane] = load_index(era.as_ref(), ins, r[a][lane], mask, ACCEPT_DATASET_LOG2) & align; } if idx.iter().all(|&x| x == idx[0]) { return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); } for lane in 0..LANES { - r[d][lane] ^= dataset_elem(idx[lane], d0, d1); + if width == 1 { + r[d][lane] ^= dataset_elem(idx[lane], d0, d1); + } else { + let mut w = [0u32; 16]; + for j in 0..width { + w[j] = dataset_elem(idx[lane] + j as u32, d0, d1); + } + r[d][lane] = fold_words(r[d][lane], &w[..width]); + } lane_addrs[lane * loads + nload] = idx[lane]; } nload += 1; } + Op::Hot => { + // Hot table: the stand-in is dataset_elem keyed by seed words 2 and 3; the address is tagged with + // bit 30 so a hot word and a dataset word at one index count as two addresses. + for lane in 0..LANES { + idx[lane] = hot_index(r[a][lane], hot_words); + } + if idx.iter().all(|&x| x == idx[0]) { + return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); + } + for lane in 0..LANES { + r[d][lane] ^= dataset_elem(idx[lane], h0, h1); + lane_addrs[lane * loads + nload] = 0x4000_0000 | idx[lane]; + } + nload += 1; + } Op::WLoad => { let b = (r[a][0] & mask) & !31; for lane in 0..LANES { @@ -298,7 +355,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut sl.sort_unstable(); let mut distinct = 0u64; for k in 0..loads { - if k == 0 || sl[k] != sl[k - 1] { + // scratch slots carry bit 31 (variant 5) and are not dataset addresses + if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) { distinct += 1; } } @@ -333,7 +391,7 @@ pub fn check_dynamic(p: &Program) -> Result { } bias_max = bias_max.max(d); } - if acc.distinct_sum <= MIN_DISTINCT_SUM { + if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) { return Err(Reject::DistinctAddresses { sum: acc.distinct_sum }); } Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max }) @@ -348,9 +406,50 @@ pub fn check(p: &Program) -> Result { #[cfg(test)] mod tests { use super::*; - use crate::generator::{candidate, generate, GeneratorConfig, generate_v1}; + use crate::generator::{candidate, candidate_class, generate, generate_class, GeneratorConfig, generate_v1, LoadClass}; use crate::verify::{DatasetMode, DatasetSource}; + #[test] + fn distinct_bound_scales_with_the_load_count() { + assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM); + assert_eq!(min_distinct_sum(32), 61_440); + } + + /// The read-width classes pass the rule at about the version 2 rate, and the instrumented interpreter agrees + /// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way). + #[test] + fn classes_pass_and_match_verify() { + for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] { + let c = LoadClass::parse(name).unwrap(); + let p = generate_class("igneum-genesis", c); + assert!(check(&p).is_ok(), "{name}"); + let mut rejected = 0; + for i in 0..60u32 { + let s = format!("igneum-rw-accept/{i}"); + let q = candidate_class(&s, s.as_bytes(), 0, c); + if check(&q).is_err() { + rejected += 1; + } + } + assert!(rejected < 15, "{name}: {rejected} of 60 rejected"); + let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2); + let bases = accept_base_nonces(&p.seed); + let loads = p.loads_per_hash(); + let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 }; + let mut la = vec![0u32; LANES * loads]; + let mut ones = [0u32; 64]; + for (u, &b) in bases.iter().enumerate() { + run_unit(&p, u, b, &mut acc, &mut la).unwrap(); + for h in crate::verify::hash_warp(&p, b, &ds) { + for j in 0..64 { + ones[j] += ((h >> j) & 1) as u32; + } + } + } + assert_eq!(acc.bit_ones, ones, "{name}: bit counts match the reference interpreter"); + } + } + /// The instrumented interpreter agrees with `verify.rs` on the closed-form dataset keyed by the seed words. #[test] fn instrumented_interpreter_matches_verify() { @@ -381,6 +480,52 @@ mod tests { } } + /// Hot-table experiment: the hot classes pass the rule at about the version 2 rate, and the hot addresses are + /// uniform over the table (16 buckets of the index over 64 units x 32 lanes x 32 hot loads). + #[test] + fn hot_classes_pass_and_hot_loads_are_uniform() { + for name in ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "scr4k32+hot64k4", "hot32k4a", "hot64k4a", "hot96k4a"] { + let c = LoadClass::parse(name).unwrap(); + let p = generate_class("igneum-genesis", c); + assert!(check(&p).is_ok(), "{name}"); + let mut rejected = 0; + for i in 0..60u32 { + let s = format!("igneum-hot-accept/{i}"); + let q = candidate_class(&s, s.as_bytes(), 0, c); + if check(&q).is_err() { + rejected += 1; + } + } + assert!(rejected < 15, "{name}: {rejected} of 60 rejected"); + } + let p = generate_class("igneum-genesis", LoadClass::hot(96, 4)); + let words = p.hot_words(); + let loads = p.loads_per_hash(); + let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 }; + let mut la = vec![0u32; LANES * loads]; + let mut buckets = [0u64; 16]; + let mut hot_count = 0u64; + for (u, &b) in accept_base_nonces(&p.seed).iter().enumerate() { + run_unit(&p, u, b, &mut acc, &mut la).unwrap(); + for &a in &la { + if a & 0xC000_0000 == 0x4000_0000 { + let idx = a & 0x3FFF_FFFF; + assert!(idx < words); + buckets[(idx as u64 * 16 / words as u64) as usize] += 1; + hot_count += 1; + } + } + } + assert_eq!(hot_count, 64 * 32 * 32, "32 hot loads per hash over 2,048 hashes"); + let mean = hot_count as f64 / 16.0; + for (i, &b) in buckets.iter().enumerate() { + assert!((b as f64 - mean).abs() < 0.15 * mean, "bucket {i}: {b} against a mean of {mean}"); + } + // the dataset distinct count still holds for the dataset loads alone + let r = check(&p).unwrap(); + assert!(r.distinct_mean() > 120.0); + } + #[test] fn base_nonces_are_aligned_and_seed_dependent() { let a = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]); diff --git a/igneum-pow/src/emit.rs b/igneum-pow/src/emit.rs index 717c4d452..7cb1ceed1 100644 --- a/igneum-pow/src/emit.rs +++ b/igneum-pow/src/emit.rs @@ -9,13 +9,383 @@ //! One deliberate difference from the Swift: `program_json` writes the cache line mask inside the `"item"` string //! as a bare `0x003fffff`. The Swift writes it quoted (`jhex`), which is not valid JSON. -use crate::generator::{Op, Program, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LOAD_SLOTS}; +use crate::generator::{EraParams, Instr, Op, Program, ProgramClass, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LOAD_SLOTS}; use crate::memhard::{ - MixParams, CACHE_LINES_PER_SEGMENT, CACHE_LINE_MASK, CACHE_LOG2_WORDS, CACHE_SEGMENTS, CACHE_SEGMENT_LOG2_LINES, - CACHE_TAG, CACHE_WORDS, CHACHA_ROUNDS, CHACHA_SIGMA, ITEM_ROUNDS, + hot_key, hot_segments, hot_words, Layout, MixParams, Shape, CACHE_LINES_PER_SEGMENT, CACHE_SEGMENT_LOG2_LINES, CACHE_TAG, + CHACHA_ROUNDS, CHACHA_SIGMA, HOT_TAG, ITEM_ROUNDS, }; use crate::seed::SplitMix64; -use crate::verify::{DatasetMode, DatasetSource, Epoch}; +use crate::verify::{window, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT}; + +/// The index expression of a dataset load (era layout, `docs/plans/era-layout.md` section 1.3). For every class +/// without an era it is the lottery hash's `rN & MASK`; for an era program it is the one form +/// `((rotl_imm(rN * M, R) & WM) | OFF) & MASK` with the site's window constants at the pack's dataset size. +fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, dataset_log2: u32) -> String { + let mask_name = match dialect { + CoreDialect::Metal => "MASK", + _ => "mask", + }; + match era { + None => format!("{a} & {mask_name}"), + Some(e) => { + let (wm, off) = window(ins, mask_for(dataset_log2), dataset_log2); + format!("((rotl_imm({a} * {}, {}u) & {}) | {}) & {mask_name}", hex(e.stride_mul), e.stride_rot, hex(wm), hex(off)) + } + } +} + +/// The era lines of program.h (empty without an era). +fn era_header_lines(p: &Program) -> String { + let Some(e) = p.class.era else { return String::new() }; + let mut s = String::new(); + s.push_str("// Era layout (5 October 2026, docs/plans/era-layout.md): NOT the lottery hash. Every dataset load reads\n"); + s.push_str("// idx = ((rotl(src * STRIDE_MUL, STRIDE_ROT) & window mask) | window offset) & MASK; the window of a load site is the\n"); + s.push_str("// dataset, a half or a quarter of it (IGNEUM_ERA_WINDOWS: site:shrink:offset); dataset word w holds word j(w) of item\n"); + s.push_str("// t(w) with j's bits at the INTERLEAVE positions (memhard.h: mh_t, mh_j, mh_addr).\n"); + s.push_str(&format!("#define IGNEUM_ERA_LABEL {}\n", jstr(&e.label()))); + s.push_str(&format!("#define IGNEUM_ERA_SEED_WORDS {{ {} }}\n", join_hex(&e.words))); + s.push_str(&format!("#define IGNEUM_ERA_ALLOWED_WIDTHS {{ {}, {}, {} }} // words, ascending, 0 = unused; one entry pins the width\n", e.allowed[0], e.allowed[1], e.allowed[2])); + s.push_str(&format!("#define IGNEUM_ERA_WIDTH_WORDS {}\n", e.width_words)); + s.push_str(&format!("#define IGNEUM_ERA_STRIDE_MUL {}\n", hex(e.stride_mul))); + s.push_str(&format!("#define IGNEUM_ERA_STRIDE_ROT {}\n", e.stride_rot)); + s.push_str(&format!("#define IGNEUM_ERA_INTERLEAVE {{ {}, {}, {}, {} }}\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3])); + s.push_str(&format!("#define IGNEUM_ERA_WINDOWS {}\n", jstr(&era_windows(p)))); + s +} + +/// "site:shrink:offset" for every load site of an era program, space separated. +fn era_windows(p: &Program) -> String { + p.instrs + .iter() + .enumerate() + .filter(|(_, i)| i.op == Op::Load) + .map(|(k, i)| format!("{k}:{}:{}", i.win, i.off)) + .collect::>() + .join(" ") +} + +/// The layout helpers of the memory-hard core for a non-linear layout: `mh_j(w)`, `mh_t(w)` and `mh_addr(t, j)` +/// (`Layout::split` and `Layout::join` as text). Empty for the linear layout, so the pinned packs do not change. +fn layout_helpers(layout: Layout, u: &str, fn_: &str) -> String { + if layout.is_linear() { + return String::new(); + } + let p = layout.pos; + let low = |q: u8| hex(((1u64 << q) - 1) as u32); + let mut s = String::new(); + s.push_str(&format!( + "// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions {} {} {} {} of w.\n", + p[0], p[1], p[2], p[3] + )); + s.push_str(&format!( + "{fn_} {u} mh_j({u} w) {{ return ((w >> {}u) & 1u) | (((w >> {}u) & 1u) << 1) | (((w >> {}u) & 1u) << 2) | (((w >> {}u) & 1u) << 3); }}\n", + p[0], p[1], p[2], p[3] + )); + s.push_str(&format!("{fn_} {u} mh_t({u} w) {{")); + for &q in p.iter().rev() { + s.push_str(&format!(" w = (w & {}) | ((w >> {}u) << {}u);", low(q), q + 1, q)); + } + s.push_str(" return w; }\n"); + s.push_str(&format!("{fn_} {u} mh_addr({u} t, {u} j) {{ {u} w = t;")); + for (i, &q) in p.iter().enumerate() { + s.push_str(&format!(" w = ((w >> {}u) << {}u) | (w & {}) | (((j >> {}u) & 1u) << {}u);", q, q + 1, low(q), i, q)); + } + s.push_str(" return w; }\n"); + s +} + +/// Where the words of a wide load come from (read-width experiment). +#[derive(Clone, Copy, PartialEq, Eq)] +enum WideSource { + /// `dataset`/`ds`: vector loads from the stored dataset. + Stored, + /// the closed form per word (Metal inline shortcut kernel). + InlineClosed, + /// one `mh_item` derivation per load, words taken from it (Metal inline memory-hard kernel). + InlineMemhard, +} + +/// One wide `load` as a single statement block (read-width experiment, 5 October 2026): `width` words from the +/// address aligned down to `width` words, folded into `dst` as `verify::fold_words`. The emitted text is the +/// same shape in the three dialects: the vector loads differ (`uint4` pointer on Metal and CUDA, `vload4` on +/// OpenCL C 1.2). For `width == 1` the caller emits the lottery hash's one-word form instead. +fn wide_load_stmt(dialect: CoreDialect, d: &str, idx: &str, width: u8, src: WideSource, closed: Option<(u32, u32)>) -> String { + debug_assert!(width == 4 || width == 16); + let (u, base_ptr) = match dialect { + CoreDialect::Metal => ("uint", "dataset"), + CoreDialect::Cuda => ("uint32_t", "ds"), + CoreDialect::OpenCl => ("uint", "ds"), + }; + let vectors = width as usize / 4; + let mut s = String::with_capacity(400); + s.push_str(&format!("{{ {u} b_ = ({idx}) & ~{}u; ", width as u32 - 1)); + match src { + WideSource::Stored => match dialect { + CoreDialect::Metal => s.push_str(&format!("device const uint4* l_ = (device const uint4*)({base_ptr} + b_); ")), + CoreDialect::Cuda => s.push_str(&format!("const uint4* l_ = (const uint4*)({base_ptr} + b_); ")), + CoreDialect::OpenCl => {} + }, + WideSource::InlineClosed => {} + WideSource::InlineMemhard => s.push_str("uint s_[16]; mh_item(cache, mh_t(b_), s_); "), + } + let word = |j: usize| -> String { + match src { + WideSource::Stored => format!("v{}_.{}", j / 4, ["x", "y", "z", "w"][j % 4]), + WideSource::InlineClosed => { + let (d0, d1) = closed.expect("closed-form words need d0, d1"); + format!("ds_elem(b_ + {j}u, {}, {})", hex(d0), hex(d1)) + } + WideSource::InlineMemhard => format!("s_[mh_j(b_) + {j}u]"), + } + }; + if src == WideSource::Stored { + for v in 0..vectors { + match dialect { + CoreDialect::OpenCl => s.push_str(&format!("uint4 v{v}_ = vload4({v}u, {base_ptr} + b_); ")), + _ => s.push_str(&format!("uint4 v{v}_ = l_[{v}]; ")), + } + } + } + s.push_str(&format!("{u} x_ = {d} ^ {}; ", word(0))); + for j in 1..width as usize { + s.push_str(&format!("x_ = (rotl_imm(x_, {FOLD_ROT}u) * {}) ^ {}; ", hex(FOLD_MUL), word(j))); + } + s.push_str(&format!("{d} = x_; }}")); + s +} + +/// The program class lines of program.h (Counter ASIC 2.0): `IGNEUM_PROGRAM_CLASS` and, when the program was drawn +/// on the chain, `IGNEUM_ERA_SEED_HEX`. Empty for every version 2 program, so the pinned packs do not change; a +/// worker reads an absent line as class v2. The generator version is `IGNEUM_GENERATOR` as before (3 for class v3). +fn program_class_header_lines(p: &Program) -> String { + if p.program_class() == ProgramClass::V2 { + return String::new(); + } + let mut s = String::new(); + s.push_str("// Program class v3 (Counter ASIC 2.0, docs/plans/counter-asic-2-rollout.md): generator version 3; a worker that\n"); + s.push_str("// runs another class refuses this pack, and a job line names the class it wants (class=v3 era=).\n"); + s.push_str(&format!("#define IGNEUM_PROGRAM_CLASS {}\n", jstr(p.program_class().name()))); + if let Some(era) = &p.era_bytes { + s.push_str(&format!("#define IGNEUM_ERA_SEED_HEX {}\n", jstr(&hex_bytes(era)))); + } + s +} + +/// The load class lines of program.h (empty for the lottery hash, so the pinned packs do not change). +fn class_header_lines(p: &Program) -> String { + if p.class.is_v2() { + return String::new(); + } + let mut s = String::new(); + if p.class.v2_loads() { + s.push_str("// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +"); + s.push_str("// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +"); + } else { + s.push_str("// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads +"); + s.push_str("// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. +"); + } + s.push_str(&format!("#define IGNEUM_LOAD_CLASS {} +", jstr(&p.class.name()))); + if p.class.mixer_mult != 1 || p.class.growth { + s.push_str(&format!("#define IGNEUM_CLASS_MIXER_MULT {} +", p.class.mixer_mult)); + s.push_str(&format!("#define IGNEUM_CACHE_GROWTH {} // 1: cache words = 2^(26 + doublings(day)), doublings = floor(log2(1 + day / 1460)) +", p.class.growth as u8)); + } + s.push_str(&format!("#define IGNEUM_LOAD_SLOTS {} +", p.class.load_slots)); + s.push_str(&format!("#define IGNEUM_LOAD_MIX {{ {}, {}, {} }} +", p.class.mix[0], p.class.mix[1], p.class.mix[2])); + let c = p.width_counts(); + s.push_str(&format!("#define IGNEUM_LOAD_WIDTH_COUNTS {{ {}, {}, {} }} // loads of 4, 16, 64 bytes per program +", c[0], c[1], c[2])); + s.push_str(&format!("#define IGNEUM_BYTES_PER_HASH {} +", p.bytes_per_hash())); + s.push_str(&format!("#define IGNEUM_FOLD_ROT {FOLD_ROT} +")); + s.push_str(&format!("#define IGNEUM_FOLD_MUL {} +", hex(FOLD_MUL))); + s +} + +/// The width of a load instruction's statement, for the emitters (1 for every non-load op). +fn load_width(ins: &Instr) -> u8 { + if ins.op == Op::Load { + ins.width + } else { + 1 + } +} + +/// Variant 5 prelude: `scr_fill(gbase, lane, slot, j)`, the fill word of a scratch slot (`verify::scratch_fill`), +/// with the program's seed words as literals. +fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String { + if !p.has_scratch() { + return String::new(); + } + let (u, fn_) = match dialect { + CoreDialect::Metal => ("uint", "inline"), + CoreDialect::Cuda => ("uint32_t", "__device__ __forceinline__"), + CoreDialect::OpenCl => ("uint", "static inline"), + }; + let mut s = String::new(); + s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a {} KiB scratch per warp, {} slots of\n", p.class.scratch_kb, p.class.scratch_slots_per_lane())); + s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not\n"); + s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.\n"); + s.push_str(&format!( + "{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }}\n", + hex(p.seed[0]), + hex(p.seed[1]), + hex(p.seed[2]) + )); + s +} + +/// Variant 5: one scratch read-modify-write as a statement block. `arena`, `tag`, `gbase` and `lane` are in scope +/// (the persistent prologue). Reads 16 bytes, folds the three data words into dst, rewrites the slot behind the tag. +fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str, slot_mask: u32) -> String { + let (u, load, store) = match dialect { + CoreDialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"), + CoreDialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"), + CoreDialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena);"), + }; + format!( + "{{ {u} s_ = {a} & {slot_mask}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}", + k = hex(FOLD_MUL) + ) +} + +/// Variant 5: the persistent-warp prologue. The kernel is launched with N warps (the resident count, the host's +/// choice); warp `w` owns arena `w` and runs the units `w, w + N, w + 2N, ...` of the launch. Inside the loop the +/// lottery hash's text is unchanged: `gid` is the unit's first output index plus the lane. The host MUST launch +/// `groups` as a multiple of N (a uniform trip count: the OpenCL local-memory exchange carries a barrier). +fn persistent_prologue(dialect: CoreDialect, words_per_lane: usize) -> String { + let (a, b) = persistent_prologue_parts(dialect, words_per_lane); + a + &b +} + +/// The prologue in two parts: the warp's identity and arena, then the unit loop. OpenCL C requires a `__local` +/// variable at the outermost scope of the kernel (AMD's compiler enforces it, 5 October 2026, round 3 on the +/// 9070 XT), so the OpenCL kernels declare the exchange buffer between the two parts. +fn persistent_prologue_parts(dialect: CoreDialect, words_per_lane: usize) -> (String, String) { + let (u, tid, nthreads, ptr) = match dialect { + CoreDialect::Metal => ("uint", "tid", "nthreads", "device uint*"), + CoreDialect::Cuda => ("uint32_t", "(blockIdx.x * blockDim.x + threadIdx.x)", "(gridDim.x * blockDim.x)", "uint32_t*"), + CoreDialect::OpenCl => ("uint", "(uint)get_global_id(0)", "(uint)get_global_size(0)", "__global uint*"), + }; + let mut s = String::new(); + s.push_str(&format!(" {u} lane = {tid} & 31u;\n")); + s.push_str(&format!(" {u} warp_ = {tid} >> 5;\n")); + s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5;\n")); + s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {words_per_lane}u;\n")); + let mut l = String::new(); + l.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{\n")); + l.push_str(&format!(" {u} gid = g_ * 32u + lane;\n")); + l.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u;\n")); + l.push_str(&format!(" {u} tag = salt + g_;\n")); + (s, l) +} + +/// The scratch lines of program.h (variant 5). +fn scratch_header_lines(p: &Program) -> String { + if !p.has_scratch() { + return String::new(); + } + let mut s = String::new(); + s.push_str(&format!("// Variant 5: persistent warps, a {} KiB scratch per launched warp (the host launches N warps and passes scratch,\n", p.class.scratch_kb)); + s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).\n"); + s.push_str("#define IGNEUM_PERSISTENT_WARPS 1\n"); + s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash)\n", p.class.scratch_slots(), p.scratch_ops_per_hash())); + s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {}u\n", p.class.scratch_slots_per_lane())); + s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {}u\n", p.class.scratch_words_per_lane())); + s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {}u\n", p.class.scratch_bytes_per_warp())); + s +} + +/// Hot-table experiment (`docs/plans/hot-table.md`): the `HOT_WORDS` literal of a hot pack's hash kernels (empty +/// for every other class, so the pinned packs do not change). +fn hot_define(p: &Program) -> String { + match p.class.hot { + Some(h) => format!("// Hot table ({} MiB, {} of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS {} +", h.mb, h.k, hex(hot_words(h.mb as u32))), + None => String::new(), + } +} + +/// The hot load as one statement per dialect: the high 32 bits of `src x HOT_WORDS` index the table. +fn hot_stmt(dialect: CoreDialect, d: &str, a: &str) -> String { + match dialect { + CoreDialect::Metal => format!("{d} = {d} ^ hot[mulhi({a}, HOT_WORDS)];"), + CoreDialect::Cuda => format!("{d} = {d} ^ hot[__umulhi({a}, HOT_WORDS)];"), + CoreDialect::OpenCl => format!("{d} = {d} ^ hot[mul_hi({a}, HOT_WORDS)];"), + } +} + +/// The hot table's fill core: `ht_segment(hot, seg)`, the cache chain of `mh_cache_segment` under the hot key and +/// the hot tag (`mh_chacha_block` and `MH_SEGMENT_LINES` must be in scope: the memory-hard core comes first). +fn emit_hot_core(p: &Program, dialect: CoreDialect) -> String { + let Some(h) = p.class.hot else { return String::new() }; + let (u, fn_, wptr) = match dialect { + CoreDialect::Metal => ("uint", "inline", "device uint*"), + CoreDialect::Cuda => ("uint32_t", "IGNEUM_HD", "uint32_t*"), + CoreDialect::OpenCl => ("uint", "static inline", "__global uint*"), + }; + let k = hot_key(&p.seed_bytes); + let mut s = String::with_capacity(1500); + s.push_str(&format!( + "// Hot table (docs/plans/hot-table.md): {} MiB = {} segments of {} chained ChaCha{} lines under the hot key KH = seed_words(\"igneum-hot/\" || epoch seed bytes), tag \"Igne\" \"umHT\". The cache chain with another key and tag.\n", + h.mb, + hot_segments(h.mb as u32), + CACHE_LINES_PER_SEGMENT, + CHACHA_ROUNDS + )); + s.push_str(&format!("{fn_} void ht_segment({wptr} hot, {u} seg) {{\n")); + s.push_str(&format!(" {u} prev[16]; {u} x[16]; {u} y[16];\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) prev[i] = 0u;\n")); + s.push_str(&format!(" for ({u} j = 0u; j < MH_SEGMENT_LINES; ++j) {{\n")); + s.push_str(&format!( + " x[0] = {} ^ prev[0]; x[1] = {} ^ prev[1]; x[2] = {} ^ prev[2]; x[3] = {} ^ prev[3];\n", + hex(CHACHA_SIGMA[0]), + hex(CHACHA_SIGMA[1]), + hex(CHACHA_SIGMA[2]), + hex(CHACHA_SIGMA[3]) + )); + for i in 0..8 { + s.push_str(&format!(" x[{}] = {} ^ prev[{}];\n", 4 + i, hex(k[i]), 4 + i)); + } + s.push_str(&format!( + " x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = {} ^ prev[14]; x[15] = {} ^ prev[15];\n", + hex(HOT_TAG[0]), + hex(HOT_TAG[1]) + )); + s.push_str(" mh_chacha_block(x, y);\n"); + s.push_str(&format!(" {wptr} line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u);\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) {{ line[i] = y[i]; prev[i] = y[i]; }}\n")); + s.push_str(" }\n"); + s.push_str("}\n"); + s +} + +/// The hot lines of program.h. +fn hot_header_lines(p: &Program) -> String { + let Some(h) = p.class.hot else { return String::new() }; + let mut s = String::new(); + s.push_str("// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the\n"); + s.push_str("// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under\n"); + s.push_str("// KH = seed_words(\"igneum-hot/\" || epoch seed bytes) and the tag \"Igne\" \"umHT\"; filled by igneum_hot_fill once per epoch.\n"); + s.push_str(&format!("#define IGNEUM_HOT_MB {}\n", h.mb)); + s.push_str(&format!("#define IGNEUM_HOT_WORDS {}\n", hex(hot_words(h.mb as u32)))); + s.push_str(&format!("#define IGNEUM_HOT_SEGMENTS {}u\n", hot_segments(h.mb as u32))); + s.push_str(&format!("#define IGNEUM_HOT_SLOTS {} // hot loads per program ({} per hash), {} the dataset loads ({} of them)\n", h.k, p.hot_loads_per_hash(), if h.added { "added beside" } else { "replacing" }, p.class.dataset_slots())); + s.push_str(&format!("#define IGNEUM_HOT_ADDED {}\n", h.added as u8)); + s.push_str(&format!("#define IGNEUM_HOT_KEY_INIT {{ {} }}\n", join_hex(&hot_key(&p.seed_bytes)))); + s +} pub fn hex(v: u32) -> String { format!("0x{v:08x}u") @@ -67,12 +437,22 @@ pub enum LoadSource<'a> { InlineMemhard(&'a MixParams), } -fn log2_segments() -> usize { - CACHE_SEGMENTS.trailing_zeros() as usize +fn log2_segments(shape: &Shape) -> usize { + shape.log2_segments() as usize } -/// The memory-hard core as source text (`emitMemhardCore`). Every parameter is a literal. +/// The memory-hard core as source text (`emitMemhardCore`). Every parameter is a literal. Linear layout. pub fn emit_memhard_core(mp: &MixParams, dialect: CoreDialect) -> String { + emit_memhard_core_layout(mp, dialect, Layout::LINEAR) +} + +/// [`emit_memhard_core`] with the dataset layout: with a non-linear layout `mh_word` and the build kernels go +/// through `mh_t`, `mh_j` and `mh_addr` (era layout); with the linear layout the text is unchanged. +pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: Layout) -> String { + let shape = &mp.shape; + let m = shape.mixer_mult; + let cache_log2_words = shape.cache_log2_words; + let cache_line_mask = shape.cache_line_mask(); let (u, fn_, cptr, wptr, lptr, lcptr) = match dialect { CoreDialect::Metal => { ("uint", "inline", "device const uint*", "device uint*", "thread uint*", "const thread uint*") @@ -84,18 +464,23 @@ pub fn emit_memhard_core(mp: &MixParams, dialect: CoreDialect) -> String { }; let k = &mp.key; let r = &mp.rot; - let m = &mp.mul; + let mul = &mp.mul; let c = &mp.rc; let mut s = String::with_capacity(6000); s.push_str(&format!( "// Memory-hard dataset core (MEMHARD.md). Cache: 2^{} words in 2^{} segments of {} chained ChaCha{} lines.\n", - CACHE_LOG2_WORDS, - log2_segments(), + cache_log2_words, + log2_segments(shape), CACHE_LINES_PER_SEGMENT, CHACHA_ROUNDS )); - s.push_str("// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.\n"); - s.push_str(&format!("#define MH_CACHE_LINE_MASK {}\n", hex(CACHE_LINE_MASK))); + if m == 1 { + s.push_str("// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.\n"); + } else { + s.push_str(&format!("// Item: 8 rounds of {m} x seed-parameterised mixer + one 64-byte cache read, then {m} x final mixer (class v3, mixer multiplier {m},\n")); + s.push_str("// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.\n"); + } + s.push_str(&format!("#define MH_CACHE_LINE_MASK {}\n", hex(cache_line_mask))); s.push_str(&format!("#define MH_SEGMENT_LINES {}u\n", CACHE_LINES_PER_SEGMENT)); s.push_str("#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }\n"); s.push_str(&format!( @@ -155,7 +540,7 @@ pub fn emit_memhard_core(mp: &MixParams, dialect: CoreDialect) -> String { s.push_str("// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.\n"); s.push_str(&format!("{fn_} void mh_mixer({lptr} s, {u} rk) {{\n")); for i in 0..16 { - s.push_str(&format!(" s[{i}] = (s[{i}] ^ ({} + rk)) * {};\n", hex(c[i]), hex(m[i]))); + s.push_str(&format!(" s[{i}] = (s[{i}] ^ ({} + rk)) * {};\n", hex(c[i]), hex(mul[i]))); } let col = (0..4).map(|i| format!("{}u", r[i])).collect::>().join(", "); let dia = (4..8).map(|i| format!("{}u", r[i])).collect::>().join(", "); @@ -165,38 +550,103 @@ pub fn emit_memhard_core(mp: &MixParams, dialect: CoreDialect) -> String { s.push_str(&format!(" MH_QR(s[2], s[7], s[8], s[13], {dia}) MH_QR(s[3], s[4], s[9], s[14], {dia})\n")); s.push_str("}\n"); s.push('\n'); - s.push_str(&format!( - "// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of mixer + cache line s[0] & mask; final mixer.\n" - )); + if m == 1 { + s.push_str(&format!( + "// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of mixer + cache line s[0] & mask; final mixer.\n" + )); + } else { + s.push_str(&format!( + "// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of {m} x mixer + cache line s[0] & mask; {m} x final mixer.\n" + )); + } s.push_str(&format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n")); for i in 0..8 { s.push_str(&format!(" s[{i}] = {};\n", hex(k[i]))); } for i in 0..8 { - s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(m[i]), hex(c[i]))); + s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i]))); } s.push_str(&format!(" for ({u} r = 0u; r < {ITEM_ROUNDS}u; ++r) {{\n")); - s.push_str(" mh_mixer(s, 0x9E3779B9u * (r + 1u));\n"); + if m == 1 { + s.push_str(" mh_mixer(s, 0x9E3779B9u * (r + 1u));\n"); + } else { + s.push_str(&format!(" for ({u} j = 0u; j < {m}u; ++j) mh_mixer(s, 0x9E3779B9u * (r * {m}u + j + 1u));\n")); + } s.push_str(&format!(" {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);\n")); s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i];\n")); s.push_str(" }\n"); - s.push_str(&format!(" mh_mixer(s, 0x9E3779B9u * {}u);\n", ITEM_ROUNDS + 1)); + if m == 1 { + s.push_str(&format!(" mh_mixer(s, 0x9E3779B9u * {}u);\n", ITEM_ROUNDS + 1)); + } else { + s.push_str(&format!( + " for ({u} j = 0u; j < {m}u; ++j) mh_mixer(s, 0x9E3779B9u * ({}u + j + 1u));\n", + ITEM_ROUNDS as u32 * m + )); + } s.push_str("}\n"); - s.push_str("// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.\n"); - s.push_str(&format!( - "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n" - )); + if layout.is_linear() { + s.push_str("// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.\n"); + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n" + )); + } else { + s.push_str(&layout_helpers(layout, u, fn_)); + s.push_str("// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).\n"); + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n" + )); + } + s +} + +/// The dataset store of one item in a build kernel: `d[i] = s[i]` at `ds + t * 16` for the linear layout, else the +/// scatter `ds[mh_addr(t, i)] = s[i]` (era layout). +fn build_store(layout: Layout, dialect: CoreDialect, ds: &str, t: &str) -> String { + let (u, wptr, cast) = match dialect { + CoreDialect::Metal => ("uint", "device uint*", ""), + CoreDialect::Cuda => ("uint32_t", "uint32_t*", "(size_t)"), + CoreDialect::OpenCl => ("uint", "__global uint*", "(ulong)"), + }; + if layout.is_linear() { + match dialect { + CoreDialect::Metal => format!(" {wptr} d = {ds} + {t} * 16u;\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"), + CoreDialect::Cuda => format!(" {wptr} d = {ds} + (size_t){t} * 16u;\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"), + CoreDialect::OpenCl => format!(" {wptr} d = {ds} + ((ulong){t} * 16u);\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"), + } + } else { + let indent = if dialect == CoreDialect::Metal { " " } else { " " }; + format!("{indent}for ({u} i = 0u; i < 16u; ++i) {ds}[{cast}mh_addr({t}, i)] = s[i];\n") + } +} + +/// [`metal_memhard`] plus, for a hot pack, the hot table's `ht_segment` and `igneum_hot_fill` kernel (one thread per +/// segment, `IGNEUM_HOT_SEGMENTS` threads). Byte-identical to [`metal_memhard`] for every other class. +pub fn metal_memhard_for(p: &Program, mp: &MixParams) -> String { + let mut s = metal_memhard_layout(mp, p.class.layout()); + if p.has_hot() { + s.push('\n'); + s.push_str(&emit_hot_core(p, CoreDialect::Metal)); + s.push_str("// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads).\n"); + s.push_str("kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" ht_segment(hot, gid);\n"); + s.push_str("}\n"); + } s } /// Metal library with the cache fill and dataset build kernels for one day key (`memhardMSL`, memhard.metal). pub fn metal_memhard(mp: &MixParams) -> String { + metal_memhard_layout(mp, Layout::LINEAR) +} + +/// [`metal_memhard`] with the dataset layout (era layout). +pub fn metal_memhard_layout(mp: &MixParams, layout: Layout) -> String { let mut s = String::new(); s.push_str("#include \n"); s.push_str("using namespace metal;\n"); - s.push_str(&emit_memhard_core(mp, CoreDialect::Metal)); + s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Metal, layout)); s.push('\n'); - s.push_str(&format!("// One thread per segment (2^{} threads).\n", log2_segments())); + s.push_str(&format!("// One thread per segment (2^{} threads).\n", log2_segments(&mp.shape))); s.push_str( "kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {\n", ); @@ -209,8 +659,7 @@ pub fn metal_memhard(mp: &MixParams) -> String { s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); s.push_str(" uint s[16];\n"); s.push_str(" mh_item(cache, gid, s);\n"); - s.push_str(" device uint* d = dataset + gid * 16u;\n"); - s.push_str(" for (uint i = 0u; i < 16u; ++i) d[i] = s[i];\n"); + s.push_str(&build_store(layout, CoreDialect::Metal, "dataset", "gid")); s.push_str("}\n"); s } @@ -236,6 +685,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: s.push_str("using namespace metal;\n"); s.push('\n'); s.push_str(&format!("#define MASK {}\n", hex(mask))); + s.push_str(&hot_define(p)); s.push_str(&format!("constant uint SEEDW[8] = {{ {} }};\n", join_hex(&p.seed))); s.push('\n'); s.push_str("inline uint splitmix32(uint x) {\n"); @@ -255,10 +705,11 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: } let mut buffer0 = "device const uint* dataset [[buffer(0)]]"; if let LoadSource::InlineMemhard(mp) = &source { - s.push_str(&emit_memhard_core(mp, CoreDialect::Metal)); + s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Metal, p.class.layout())); s.push('\n'); buffer0 = "device const uint* cache [[buffer(0)]]"; } + s.push_str(&scratch_prelude(p, CoreDialect::Metal)); if bound { s.push_str("// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.\n"); s.push_str(&format!("kernel void igneum_hash_bound({buffer0},\n")); @@ -270,7 +721,20 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: if bound { s.push_str(" constant uint* initw [[buffer(3)]],\n"); } - s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + if p.has_hot() { + s.push_str(&format!(" device const uint* hot [[buffer({})]],\n", if bound { 4 } else { 3 })); + } + if p.has_scratch() { + let b = (if bound { 4 } else { 3 }) + p.has_hot() as usize; + s.push_str(&format!(" device uint* scratch [[buffer({b})]],\n")); + s.push_str(&format!(" constant uint& groups [[buffer({})]],\n", b + 1)); + s.push_str(&format!(" constant uint& salt [[buffer({})]],\n", b + 2)); + s.push_str(" uint tid [[thread_position_in_grid]],\n"); + s.push_str(" uint nthreads [[threads_per_grid]]) {\n"); + s.push_str(&persistent_prologue(CoreDialect::Metal, p.class.scratch_words_per_lane())); + } else { + s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + } s.push_str(" uint nonce = baseNonce + gid;\n"); s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); if p.has_wide() { @@ -285,11 +749,12 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: )); } s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); - let word_index = |a: &str, wide: bool| -> String { + let era = p.class.era; + let word_index = |a: &str, wide: bool, ins: &Instr| -> String { if wide { format!("(simd_broadcast({a}, 0) & WMASK) + lane") } else { - format!("{a} & MASK") + load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, dataset_log2) } }; let fetch = |idx: String| -> String { @@ -319,8 +784,18 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: Op::Rotr => format!("{d} = rotr_var({d}, {a});"), Op::Mad => format!("{d} = {a} * {b} + {d};"), Op::Shfl => format!("{d} = {d} ^ simd_shuffle_xor({a}, (ushort){});", ins.mask), - Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false))), - Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true))), + Op::Load if load_width(ins) > 1 => { + let (src, closed) = match &source { + LoadSource::Stored => (WideSource::Stored, None), + LoadSource::InlineClosed(d0, d1) => (WideSource::InlineClosed, Some((*d0, *d1))), + LoadSource::InlineMemhard(_) => (WideSource::InlineMemhard, None), + }; + wide_load_stmt(CoreDialect::Metal, &d, &word_index(&a, false, ins), ins.width, src, closed) + } + Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false, ins))), + Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true, ins))), + Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::Metal, &d, &a), }; s.push_str(&format!(" {line} // {k}\n")); } @@ -328,6 +803,9 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } s.push_str("}\n"); s } @@ -356,8 +834,9 @@ fn init_line(p: &Program, u: &str, i: usize) -> String { } /// The instruction lines of the CUDA hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`). -fn cuda_instr_lines(p: &Program) -> String { +fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String { let mut s = String::with_capacity(6000); + let era = p.class.era; for (k, ins) in p.instrs.iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); @@ -379,8 +858,13 @@ fn cuda_instr_lines(p: &Program) -> String { Op::Rotr => format!("{d} = rotr_var({d}, {a});"), Op::Mad => format!("{d} = {a} * {b} + {d};"), Op::Shfl => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask), - Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"), + Op::Load if load_width(ins) > 1 => { + wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None) + } + Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2)), Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"), + Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::Cuda, &d, &a), }; s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); } @@ -389,6 +873,13 @@ fn cuda_instr_lines(p: &Program) -> String { /// The CUDA kernel (`generateCUDA`, kernel.cu). `memhard` is `None` for a closed-form pack. pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { + cuda_kernel_at(p, memhard, DEFAULT_DATASET_LOG2) +} + +/// [`cuda_kernel`] at a dataset size (an era program's window constants are literals of the pack's size; every +/// other class ignores it). +pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + let layout = p.class.layout(); let mut s = String::with_capacity(9000); s.push_str(&generated_by(&p.seed_string)); s.push_str( @@ -402,6 +893,7 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str("#include \"memhard.h\"\n"); } s.push('\n'); + s.push_str(&hot_define(p)); s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n"); s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); @@ -439,17 +931,33 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str(" if (t < nItems) {\n"); s.push_str(" uint32_t s[16];\n"); s.push_str(" mh_item(cache, t, s);\n"); - s.push_str(" uint32_t* d = ds + (size_t)t * 16u;\n"); - s.push_str(" for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];\n"); + s.push_str(&build_store(layout, CoreDialect::Cuda, "ds", "t")); s.push_str(" }\n"); s.push_str("}\n"); + if p.has_hot() { + s.push_str("// Hot table (ht_segment is in memhard.h): one thread per segment.\n"); + s.push_str("__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) {\n"); + s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n"); + s.push_str(" if (seg < nSegments) ht_segment(hot, seg);\n"); + s.push_str("}\n"); + } s.push('\n'); } + if memhard.is_none() && p.has_hot() { + panic!("a hot-table pack needs the memory-hard dataset (the hot fill shares its ChaCha core)"); + } s.push_str("// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every\n"); s.push_str("// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a\n"); s.push_str("// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.\n"); - s.push_str("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {\n"); - s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); + s.push_str(&scratch_prelude(p, CoreDialect::Cuda)); + let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" }; + let hot_args = if p.has_hot() { ", const uint32_t* hot" } else { "" }; + s.push_str(&format!("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask{hot_args}{scratch_args}) {{\n")); + if p.has_scratch() { + s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane())); + } else { + s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); + } s.push_str(" uint32_t nonce = baseNonce + gid;\n"); s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n"); if p.has_wide() { @@ -459,11 +967,14 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str(&init_line(p, "uint32_t", i)); } s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n")); - s.push_str(&cuda_instr_lines(p)); + s.push_str(&cuda_instr_lines(p, dataset_log2)); s.push_str(" }\n"); s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } s.push_str("}\n"); s.push('\n'); s.push_str("// Host-side launch wrappers. Declared in program.h, called from host.cu.\n"); @@ -493,17 +1004,40 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); s.push('\n'); + if p.has_hot() { + s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) {\n"); + s.push_str(" if (nSegments == 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 256u;\n"); + s.push_str(" uint32_t grid = (nSegments + block - 1u) / block;\n"); + s.push_str(" igneum_hot_fill<<>>(hot, nSegments);\n"); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + s.push('\n'); + } + } + let (hot_decl, hot_pass) = if p.has_hot() { (" const uint32_t* hot,", " hot,") } else { ("", "") }; + if p.has_scratch() { + s.push_str("// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).\n"); + s.push_str(&format!("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n")); + s.push_str(" uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {\n"); + s.push_str(" if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 32u * blockWarps;\n"); + s.push_str(" if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(&format!(" igneum_hash<<>>(ds, out, baseNonce, mask,{hot_pass} scratch, nonces / 32u, salt);\n")); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + } else { + s.push_str(&format!( + "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n", + )); + s.push_str(" uint32_t nonces, uint32_t blockWarps) {\n"); + s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 32u * blockWarps;\n"); + s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(&format!(" igneum_hash<<>>(ds, out, baseNonce, mask{});\n", if p.has_hot() { ", hot" } else { "" })); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); } - s.push_str( - "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n", - ); - s.push_str(" uint32_t nonces, uint32_t blockWarps) {\n"); - s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n"); - s.push_str(" uint32_t block = 32u * blockWarps;\n"); - s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n"); - s.push_str(" igneum_hash<<>>(ds, out, baseNonce, mask);\n"); - s.push_str(" return cudaGetLastError();\n"); - s.push_str("}\n"); s.push('\n'); s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {\n"); s.push_str(" cudaFuncAttributes attr;\n"); @@ -520,6 +1054,11 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { /// `I` arrive by value in `IgneumInitWords` (`bind::block_init_words`), the instruction text is that of /// `igneum_hash`. Declarations for the host are at the top of the file. pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { + cuda_kernel_bound_at(p, memhard, DEFAULT_DATASET_LOG2) +} + +/// [`cuda_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]). +pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { let mut s = String::with_capacity(9000); s.push_str(&generated_by(&p.seed_string)); s.push_str( @@ -538,6 +1077,7 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { s.push('\n'); s.push_str("struct IgneumInitWords { uint32_t w[8]; };\n"); s.push('\n'); + s.push_str(&hot_define(p)); s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n"); s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); @@ -548,8 +1088,15 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); s.push('\n'); let _ = memhard; // the bound kernel reads the stored dataset in both constructions - s.push_str("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {\n"); - s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); + s.push_str(&scratch_prelude(p, CoreDialect::Cuda)); + let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" }; + let hot_args = if p.has_hot() { ", const uint32_t* hot" } else { "" }; + s.push_str(&format!("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw{hot_args}{scratch_args}) {{\n")); + if p.has_scratch() { + s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane())); + } else { + s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); + } s.push_str(" uint32_t nonce = baseNonce + gid;\n"); s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n"); if p.has_wide() { @@ -563,23 +1110,39 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { )); } s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n")); - s.push_str(&cuda_instr_lines(p)); + s.push_str(&cuda_instr_lines(p, dataset_log2)); s.push_str(" }\n"); s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } s.push_str("}\n"); s.push('\n'); - s.push_str( - "cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n", - ); - s.push_str(" IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {\n"); - s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n"); - s.push_str(" uint32_t block = 32u * blockWarps;\n"); - s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n"); - s.push_str(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw);\n"); - s.push_str(" return cudaGetLastError();\n"); - s.push_str("}\n"); + let (hot_decl, hot_pass) = if p.has_hot() { (" const uint32_t* hot,", ", hot") } else { ("", "") }; + if p.has_scratch() { + s.push_str("// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).\n"); + s.push_str("cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n"); + s.push_str(&format!(" IgneumInitWords iw,{hot_decl} uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {{\n")); + s.push_str(" if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 32u * blockWarps;\n"); + s.push_str(" if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(&format!(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw{hot_pass}, scratch, nonces / 32u, salt);\n")); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + } else { + s.push_str( + "cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n", + ); + s.push_str(&format!(" IgneumInitWords iw,{hot_decl} uint32_t nonces, uint32_t blockWarps) {{\n")); + s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 32u * blockWarps;\n"); + s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(&format!(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw{hot_pass});\n")); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + } s.push('\n'); s.push_str("cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {\n"); s.push_str(" cudaFuncAttributes attr;\n"); @@ -592,8 +1155,9 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { } /// The instruction lines of the OpenCL hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`). -fn opencl_instr_lines(p: &Program) -> String { +fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String { let mut s = String::with_capacity(6000); + let era = p.class.era; for (k, ins) in p.instrs.iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); @@ -614,8 +1178,13 @@ fn opencl_instr_lines(p: &Program) -> String { Op::Rotr => format!("{d} = rotr_var({d}, {a});"), Op::Mad => format!("{d} = {a} * {b} + {d};"), Op::Shfl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask), - Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"), + Op::Load if load_width(ins) > 1 => { + wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None) + } + Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2)), Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"), + Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::OpenCl, &d, &a), }; s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); } @@ -626,23 +1195,49 @@ fn opencl_instr_lines(p: &Program) -> String { /// fifth argument (`__global const uint* initw`, 8 words, `bind::block_init_words`). One source file so the serve /// mode of proto-opencl/host.c builds cache fill, dataset build and the bound hash from it at runtime. pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { - let mut s = opencl_kernel(p, memhard); + opencl_kernel_bound_at(p, memhard, DEFAULT_DATASET_LOG2) +} + +/// [`opencl_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]). +pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + let mut s = opencl_kernel_at(p, memhard, dataset_log2); s.push('\n'); s.push_str( "// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.\n", ); - s.push_str("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {\n"); - s.push_str(" uint gid = (uint)get_global_id(0);\n"); + let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" }; + let hot_args = if p.has_hot() { ", __global const uint* hot" } else { "" }; + s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw{hot_args}{scratch_args}) {{\n")); + let (setup, unit_loop) = persistent_prologue_parts(CoreDialect::OpenCl, p.class.scratch_words_per_lane()); + if p.has_scratch() { + s.push_str(&setup); + } else { + s.push_str(" uint gid = (uint)get_global_id(0);\n"); + } s.push_str(" uint lid = (uint)get_local_id(0);\n"); - s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); - s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); - s.push_str("#if IGNEUM_EXCHANGE == 0\n"); - s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); - s.push_str(" uint xk = 0u;\n"); - s.push_str("#else\n"); - s.push_str(" (void)lid;\n"); - s.push_str("#endif\n"); + if p.has_scratch() { + // the __local exchange buffer must sit at the kernel's outermost scope: declare it, then open the unit loop + s.push_str("#if IGNEUM_EXCHANGE == 0\n"); + s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); + s.push_str(" uint xk = 0u;\n"); + s.push_str("#else\n"); + s.push_str(" (void)lid;\n"); + s.push_str("#endif\n"); + s.push_str(&unit_loop); + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); + } else { + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); + s.push_str("#if IGNEUM_EXCHANGE == 0\n"); + s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); + s.push_str(" uint xk = 0u;\n"); + s.push_str("#else\n"); + s.push_str(" (void)lid;\n"); + s.push_str("#endif\n"); + } if p.has_wide() { s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n"); } @@ -654,17 +1249,26 @@ pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { )); } s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); - s.push_str(&opencl_instr_lines(p)); + s.push_str(&opencl_instr_lines(p, dataset_log2)); s.push_str(" }\n"); s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } s.push_str("}\n"); s } /// The OpenCL C 1.2 kernel (`generateOpenCL`, kernel.cl). pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { + opencl_kernel_at(p, memhard, DEFAULT_DATASET_LOG2) +} + +/// [`opencl_kernel`] at a dataset size (see [`cuda_kernel_at`]). +pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + let layout = p.class.layout(); let mut s = String::with_capacity(14000); s.push_str(&generated_by(&p.seed_string)); s.push_str("// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).\n"); @@ -686,6 +1290,9 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str("#ifdef cl_khr_subgroups\n#pragma OPENCL EXTENSION cl_khr_subgroups : enable\n#endif\n"); s.push_str("#ifdef cl_khr_subgroup_shuffle\n#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable\n#endif\n"); s.push_str("#elif IGNEUM_EXCHANGE == 2\n#pragma OPENCL EXTENSION cl_intel_subgroups : enable\n#endif\n"); + if p.has_scratch() { + s.push_str("#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))\n"); + } s.push_str("#else\n"); s.push_str("// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.\n"); s.push_str("#include \"emu_opencl.h\"\n#endif\n"); @@ -707,6 +1314,7 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str("#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }\n"); s.push_str("#endif\n"); s.push('\n'); + s.push_str(&hot_define(p)); s.push_str("static inline uint splitmix32(uint x) {\n"); s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); @@ -722,7 +1330,7 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str(DS_ELEM_BODY); s.push('\n'); if let Some(mp) = memhard { - s.push_str(&emit_memhard_core(mp, CoreDialect::OpenCl)); + s.push_str(&emit_memhard_core_layout(mp, CoreDialect::OpenCl, layout)); s.push('\n'); s.push_str("// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.\n"); s.push_str("// The same constants as memhard.h in this pack (one emitter, three dialects).\n"); @@ -735,11 +1343,20 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str(" if (t < nItems) {\n"); s.push_str(" uint s[16];\n"); s.push_str(" mh_item(cache, t, s);\n"); - s.push_str(" __global uint* d = ds + ((ulong)t * 16u);\n"); - s.push_str(" for (uint i = 0u; i < 16u; ++i) d[i] = s[i];\n"); + s.push_str(&build_store(layout, CoreDialect::OpenCl, "ds", "t")); s.push_str(" }\n"); s.push_str("}\n"); + if p.has_hot() { + s.push_str(&emit_hot_core(p, CoreDialect::OpenCl)); + s.push_str("// Hot table: one work-item per segment.\n"); + s.push_str("__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) {\n"); + s.push_str(" uint seg = (uint)get_global_id(0);\n"); + s.push_str(" if (seg < nSegments) ht_segment(hot, seg);\n"); + s.push_str("}\n"); + } s.push('\n'); + } else if p.has_hot() { + panic!("a hot-table pack needs the memory-hard dataset (the hot fill shares its ChaCha core)"); } else { s.push_str("// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.\n"); s.push_str("__kernel void igneum_fill(__global uint* ds, uint n, uint d0, uint d1) {\n"); @@ -751,17 +1368,37 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str("// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the\n"); s.push_str("// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and\n"); s.push_str("// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).\n"); - s.push_str("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {\n"); - s.push_str(" uint gid = (uint)get_global_id(0);\n"); + s.push_str(&scratch_prelude(p, CoreDialect::OpenCl)); + let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" }; + let hot_args = if p.has_hot() { ", __global const uint* hot" } else { "" }; + s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask{hot_args}{scratch_args}) {{\n")); + let (setup, unit_loop) = persistent_prologue_parts(CoreDialect::OpenCl, p.class.scratch_words_per_lane()); + if p.has_scratch() { + s.push_str(&setup); + } else { + s.push_str(" uint gid = (uint)get_global_id(0);\n"); + } s.push_str(" uint lid = (uint)get_local_id(0);\n"); - s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); - s.push_str("#if IGNEUM_EXCHANGE == 0\n"); - s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); - s.push_str(" uint xk = 0u;\n"); - s.push_str("#else\n"); - s.push_str(" (void)lid;\n"); - s.push_str("#endif\n"); + if p.has_scratch() { + s.push_str("#if IGNEUM_EXCHANGE == 0\n"); + s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); + s.push_str(" uint xk = 0u;\n"); + s.push_str("#else\n"); + s.push_str(" (void)lid;\n"); + s.push_str("#endif\n"); + s.push_str(&unit_loop); + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + } else { + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str("#if IGNEUM_EXCHANGE == 0\n"); + s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); + s.push_str(" uint xk = 0u;\n"); + s.push_str("#else\n"); + s.push_str(" (void)lid;\n"); + s.push_str("#endif\n"); + } if p.has_wide() { s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n"); } @@ -769,11 +1406,14 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { s.push_str(&init_line(p, "uint", i)); } s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); - s.push_str(&opencl_instr_lines(p)); + s.push_str(&opencl_instr_lines(p, dataset_log2)); s.push_str(" }\n"); s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } s.push_str("}\n"); s.push('\n'); s.push_str("#if IGNEUM_EXCHANGE != 0\n"); @@ -828,16 +1468,24 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!("#define IGNEUM_LOADS_PER_HASH {}\n", p.loads_per_hash())); s.push_str(&format!("#define IGNEUM_WIDE_LOADS_PER_HASH {}\n", p.wide_loads_per_hash())); s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix()))); + s.push_str(&program_class_header_lines(p)); + s.push_str(&class_header_lines(p)); + s.push_str(&scratch_header_lines(p)); + s.push_str(&era_header_lines(p)); + s.push_str(&hot_header_lines(p)); s.push_str("// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)\n"); s.push_str(&format!("#define IGNEUM_DATASET_MODE {}\n", if memhard.is_some() { 1 } else { 0 })); s.push('\n'); s.push_str(&format!("#define IGNEUM_SEEDW_INIT {{ {} }}\n", join_hex(&p.seed))); if let Some(mp) = memhard { s.push_str(&format!("#define IGNEUM_KEY_INIT {{ {} }}\n", join_hex(&mp.key))); - s.push_str(&format!("#define IGNEUM_CACHE_LOG2_WORDS {CACHE_LOG2_WORDS}\n")); + s.push_str(&format!("#define IGNEUM_CACHE_LOG2_WORDS {}\n", mp.shape.cache_log2_words)); s.push_str(&format!("#define IGNEUM_CACHE_SEGMENT_LOG2_LINES {CACHE_SEGMENT_LOG2_LINES}\n")); - s.push_str(&format!("#define IGNEUM_CACHE_SEGMENTS {CACHE_SEGMENTS}u\n")); + s.push_str(&format!("#define IGNEUM_CACHE_SEGMENTS {}u\n", mp.shape.cache_segments())); s.push_str(&format!("#define IGNEUM_ITEM_ROUNDS {ITEM_ROUNDS}\n")); + if mp.shape.mixer_mult != 1 { + s.push_str(&format!("#define IGNEUM_MIXER_MULT {} // mixer applications per round and after the last read (class v3, docs/plans/mixer-x4.md)\n", mp.shape.mixer_mult)); + } s.push_str(&format!( "#define IGNEUM_MIX_ROT_INIT {{ {} }}\n", mp.rot.iter().map(|r| format!("{r}u")).collect::>().join(", ") @@ -849,15 +1497,24 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str("// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().\n"); s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);\n"); s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n"); + if p.has_hot() { + s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments);\n"); + } } else { s.push_str("#ifndef IGNEUM_NO_CUDA\n"); s.push_str("// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().\n"); s.push_str("cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);\n"); } - s.push_str( - "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n", - ); - s.push_str(" uint32_t nonces, uint32_t blockWarps);\n"); + let hot_decl = if p.has_hot() { " const uint32_t* hot," } else { "" }; + if p.has_scratch() { + s.push_str(&format!("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n")); + s.push_str(" uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);\n"); + } else { + s.push_str(&format!( + "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n", + )); + s.push_str(" uint32_t nonces, uint32_t blockWarps);\n"); + } s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);\n"); s.push_str("#endif\n"); s @@ -881,7 +1538,11 @@ pub fn cuda_memhard_header(p: &Program, mp: &MixParams) -> String { s.push_str("#else\n"); s.push_str("#define IGNEUM_HD static inline\n"); s.push_str("#endif\n"); - s.push_str(&emit_memhard_core(mp, CoreDialect::Cuda)); + s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Cuda, p.class.layout())); + if p.has_hot() { + s.push('\n'); + s.push_str(&emit_hot_core(p, CoreDialect::Cuda)); + } s } @@ -901,6 +1562,13 @@ pub struct PackVectors { pub cache_last: Vec, /// FNV-1a 64 over the whole cache (memory-hard only) pub cache_fnv: u64, + /// The cache is 2^cache_log2_words words (memory-hard only; 26 under version 2) + pub cache_log2_words: u32, + /// Hot table (hot packs only): head line, last line, FNV-1a 64 over the whole table + pub has_hot: bool, + pub hot_head: Vec, + pub hot_last: Vec, + pub hot_fnv: u64, } /// The base nonces of the three vector warps every pack carries. @@ -962,7 +1630,8 @@ pub fn vectors_header( s.push_str("};\n"); if memhard { s.push_str(&format!( - "// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^{CACHE_LOG2_WORDS} words.\n" + "// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^{} words.\n", + v.cache_log2_words )); s.push_str("static const uint32_t IGNEUM_CACHE_HEAD[16] = {\n"); s.push_str(&format!(" {},\n", join_hex(&v.cache_head[..8]))); @@ -974,6 +1643,18 @@ pub fn vectors_header( s.push_str("};\n"); s.push_str(&format!("static const uint64_t IGNEUM_CACHE_FNV64 = {};\n", hex64(v.cache_fnv))); } + if v.has_hot { + s.push_str("// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words.\n"); + s.push_str("static const uint32_t IGNEUM_HOT_HEAD[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.hot_head[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.hot_head[8..16]))); + s.push_str("};\n"); + s.push_str("static const uint32_t IGNEUM_HOT_LAST[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.hot_last[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.hot_last[8..16]))); + s.push_str("};\n"); + s.push_str(&format!("static const uint64_t IGNEUM_HOT_FNV64 = {};\n", hex64(v.hot_fnv))); + } s } @@ -1004,6 +1685,62 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!(" \"iterations\": {ITERATIONS},\n")); s.push_str(&format!(" \"instruction_count\": {INSTR_COUNT},\n")); s.push_str(&format!(" \"loads_per_hash\": {},\n", p.loads_per_hash())); + if p.program_class() != ProgramClass::V2 { + s.push_str(&format!(" \"program_class\": {},\n", jstr(p.program_class().name()))); + if let Some(era) = &p.era_bytes { + s.push_str(&format!(" \"era_seed_bytes\": {},\n", jstr(&hex_bytes(era)))); + } + } + if !p.class.is_v2() { + let c = p.width_counts(); + s.push_str(&format!(" \"load_class\": {},\n", jstr(&p.class.name()))); + if p.class.mixer_mult != 1 || p.class.growth { + s.push_str(&format!(" \"mixer_mult\": {},\n", p.class.mixer_mult)); + s.push_str(&format!(" \"cache_growth\": {},\n", p.class.growth)); + s.push_str(&format!(" \"mixer\": \"class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is {} applications with round keys (r * {} + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis\",\n", p.class.mixer_mult, p.class.mixer_mult)); + } + s.push_str(&format!(" \"load_slots\": {},\n", p.class.load_slots)); + s.push_str(&format!(" \"load_mix_percent_4_16_64\": [{}, {}, {}],\n", p.class.mix[0], p.class.mix[1], p.class.mix[2])); + s.push_str(&format!(" \"load_width_counts_4_16_64\": [{}, {}, {}],\n", c[0], c[1], c[2])); + s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash())); + if p.has_scratch() { + s.push_str(&format!(" \"scratch_ops_per_hash\": {},\n", p.scratch_ops_per_hash())); + s.push_str(&format!(" \"scratch_kib_per_warp\": {},\n", p.class.scratch_kb)); + s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a {kb} KiB scratch per warp of {slots} 16-byte slots per lane (lane-major); slot = src & 0x{smask:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n", kb = p.class.scratch_kb, slots = p.class.scratch_slots_per_lane(), smask = p.class.scratch_slot_mask())); + } + s.push_str(&format!(" \"wide_load\": \"read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, {FOLD_ROT}) * 0x{FOLD_MUL:08x}) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots\",\n")); + if let Some(e) = p.class.era { + s.push_str(" \"era\": {\n"); + s.push_str(&format!(" \"label\": {},\n", jstr(&e.label()))); + s.push_str(&format!(" \"seed_words\": [{}],\n", join_jhex(&e.words))); + s.push_str(" \"draw\": \"docs/plans/era-layout.md 1.1: SplitMix64 seeded with seed_words[0] | seed_words[1] << 32 of seed_words_from_bytes('igneum-era/' || n_le64 || E_n); width = allowed[below(|allowed|)], stride_mul = low32(next()) | 1, stride_rot = 1 + below(31), then four next() draws for a partial Fisher-Yates over positions log2(W)..15 of which 4 - log2(W) are used\",\n"); + s.push_str(&format!(" \"allowed_widths\": [{}],\n", e.allowed_set().iter().map(|w| w.to_string()).collect::>().join(", "))); + s.push_str(&format!(" \"width_words\": {},\n", e.width_words)); + s.push_str(&format!(" \"stride_mul\": {},\n", jhex(e.stride_mul))); + s.push_str(&format!(" \"stride_rot\": {},\n", e.stride_rot)); + s.push_str(&format!(" \"interleave\": [{}, {}, {}, {}],\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3])); + s.push_str(" \"address\": \"y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n"); + s.push_str(" \"windows\": \"per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)\",\n"); + s.push_str(" \"dataset_word\": \"dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed\",\n"); + s.push_str(" \"program_id_suffix\": \"'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]\"\n"); + s.push_str(" },\n"); + } + if let Some(h) = p.class.hot { + let hk = hot_key(&p.seed_bytes); + s.push_str(&format!( + " \"hot_table\": {{\"mb\": {}, \"words\": {}, \"segments\": {}, \"slots\": {}, \"form\": {}, \"dataset_slots\": {}, \"hot_loads_per_hash\": {}, \"key\": [{}], \"key_derivation\": \"seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch\", \"tag\": [{}], \"chain\": \"the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0\", \"load\": \"dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)\", \"slots_rule\": \"the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected\", \"acceptance_stand_in\": \"dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])\", \"program_id\": \"the read-width id with 'hot/' || mb || k appended\", \"spec\": \"docs/plans/hot-table.md\"}},\n", + h.mb, + hot_words(h.mb as u32), + hot_segments(h.mb as u32), + h.k, + jstr(if h.added { "added: k load slots added beside the class's, the dataset loads unchanged" } else { "replaced: k of the class's load slots read the table" }), + p.class.dataset_slots(), + p.hot_loads_per_hash(), + join_jhex(&hk), + join_jhex(&HOT_TAG) + )); + } + } s.push_str(&format!( " \"op_mix\": {{{}}},\n", p.histogram().iter().map(|(n, c)| format!("{}: {c}", jstr(n))).collect::>().join(", ") @@ -1028,7 +1765,11 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n", ); s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"); - s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"\n"); + s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); + if p.has_hot() { + s.push_str(",\n \"hot\": \"dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)\""); + } + s.push('\n'); s.push_str(" },\n"); s.push_str(" \"dataset\": {\n"); s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); @@ -1044,9 +1785,12 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(" \"spec\": \"proto-metal/MEMHARD.md\",\n"); s.push_str(&format!(" \"key\": [{}],\n", join_jhex(&mp.key))); s.push_str(" \"key_derivation\": \"the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]\",\n"); + let shape = &mp.shape; s.push_str(&format!( - " \"cache\": {{\"log2_words\": {CACHE_LOG2_WORDS}, \"bytes\": {}, \"line_words\": 16, \"segment_lines\": {CACHE_LINES_PER_SEGMENT}, \"segments\": {CACHE_SEGMENTS}, \"block\": \"ChaCha{CHACHA_ROUNDS} core + feed-forward, rotations 16 12 8 7\", \"sigma\": [{}], \"tag\": [{}], \"chain\": \"in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0\"}},\n", - CACHE_WORDS as u64 * 4, + " \"cache\": {{\"log2_words\": {}, \"bytes\": {}, \"line_words\": 16, \"segment_lines\": {CACHE_LINES_PER_SEGMENT}, \"segments\": {}, \"block\": \"ChaCha{CHACHA_ROUNDS} core + feed-forward, rotations 16 12 8 7\", \"sigma\": [{}], \"tag\": [{}], \"chain\": \"in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0\"}},\n", + shape.cache_log2_words, + shape.cache_words() as u64 * 4, + shape.cache_segments(), join_jhex(&CHACHA_SIGMA), join_jhex(&CACHE_TAG) )); @@ -1057,11 +1801,23 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { join_jhex(&mp.rc) )); // The Swift writes jhex(cacheLineMask) here, which breaks the JSON. We write the bare literal. - s.push_str(&format!( - " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: s = M_r(s); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then s = M_{ITEM_ROUNDS}(s); item(t) = s\",\n", - ITEM_ROUNDS - 1, - CACHE_LINE_MASK - )); + if shape.mixer_mult == 1 { + s.push_str(&format!( + " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: s = M_r(s); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then s = M_{ITEM_ROUNDS}(s); item(t) = s\",\n", + ITEM_ROUNDS - 1, + shape.cache_line_mask() + )); + } else { + let m = shape.mixer_mult; + s.push_str(&format!( + " \"mixer_mult\": {m},\n \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: for j in 0..{}: s = M(s, rk = (r * {m} + j + 1) * 0x9E3779B9); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then for j in 0..{}: s = M(s, rk = ({} + j + 1) * 0x9E3779B9); item(t) = s\",\n", + ITEM_ROUNDS - 1, + m - 1, + shape.cache_line_mask(), + m - 1, + ITEM_ROUNDS as u32 * m + )); + } s.push_str(" \"word\": \"dataset[w] = item(w >> 4)[w & 15]\"\n"); } else { s.push_str(" \"mode\": \"closed-form\",\n"); @@ -1071,6 +1827,24 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(" \"instructions\": [\n"); let n = p.instrs.len(); for (k, ins) in p.instrs.iter().enumerate() { + if !p.class.is_v2() { + let era_fields = if p.class.era.is_some() { format!(", \"win\": {}, \"off\": {}", ins.win, ins.off) } else { String::new() }; + s.push_str(&format!( + " {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}, \"width\": {}{era_fields}}}", + jstr(ins.op.name()), + ins.dst, + ins.src, + ins.src2, + jhex(ins.imm), + jhex(ins.imm2), + ins.rot, + ins.bit, + ins.mask, + ins.width + )); + s.push_str(if k + 1 < n { ",\n" } else { "\n" }); + continue; + } s.push_str(&format!( " {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}}}", jstr(ins.op.name()), @@ -1136,7 +1910,13 @@ pub fn vectors_json( if memhard { s.push_str(&format!(",\n \"cache_head\": [{}],\n", join_jhex(&v.cache_head))); s.push_str(&format!(" \"cache_last_line\": [{}],\n", join_jhex(&v.cache_last))); - s.push_str(&format!(" \"cache_fnv1a64\": {}\n", jhex64(v.cache_fnv))); + s.push_str(&format!(" \"cache_fnv1a64\": {}", jhex64(v.cache_fnv))); + if v.has_hot { + s.push_str(&format!(",\n \"hot_head\": [{}],\n", join_jhex(&v.hot_head))); + s.push_str(&format!(" \"hot_last_line\": [{}],\n", join_jhex(&v.hot_last))); + s.push_str(&format!(" \"hot_fnv1a64\": {}", jhex64(v.hot_fnv))); + } + s.push('\n'); } else { s.push('\n'); } @@ -1171,36 +1951,45 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { let memhard = ds.memhard().map(|m| &m.params); let bases = PACK_VECTOR_BASES.to_vec(); let outs: Vec<[u64; 32]> = bases.iter().map(|&b| epoch.hash_warp(b)).collect(); + // the self-test words under the program's layout (era layout; linear for every other class) let mut v = PackVectors { - head: (0..16).map(|i| ds.word(i)).collect(), - last: ds.word(mask), + head: (0..16).map(|i| epoch.dataset_word(i)).collect(), + last: epoch.dataset_word(mask), sample_idx: sample_indices(mask), ..Default::default() }; - v.sample_val = v.sample_idx.iter().map(|&i| ds.word(i)).collect(); + v.sample_val = v.sample_idx.iter().map(|&i| epoch.dataset_word(i)).collect(); if let Some(m) = ds.memhard() { let w = m.cache.words(); v.cache_head = w[..16].to_vec(); v.cache_last = w[w.len() - 16..].to_vec(); v.cache_fnv = m.cache.fnv1a64(); + v.cache_log2_words = m.shape().cache_log2_words; + } + if let Some(h) = &ds.hot { + let w = h.words(); + v.has_hot = true; + v.hot_head = w[..16].to_vec(); + v.hot_last = w[w.len() - 16..].to_vec(); + v.hot_fnv = h.fnv1a64(); } let is_mh = memhard.is_some(); let mut files = vec![ ("program.json".to_string(), program_json(p, day, ds)), ("vectors.json".to_string(), vectors_json(p, day, ds.log2_words, &bases, &outs, &v, mask, source, is_mh)), - ("kernel.cu".to_string(), cuda_kernel(p, memhard)), - ("kernel.cl".to_string(), opencl_kernel(p, memhard)), + ("kernel.cu".to_string(), cuda_kernel_at(p, memhard, ds.log2_words)), + ("kernel.cl".to_string(), opencl_kernel_at(p, memhard, ds.log2_words)), ("program.h".to_string(), program_header(p, day, ds)), ("vectors.h".to_string(), vectors_header(p, &bases, &outs, &v, mask, source, is_mh)), ("program.metal".to_string(), metal_program(p, ds.log2_words, LoadSource::Stored)), // Header-bound kernels (3 October 2026, bind.rs): new files, the seven above are unchanged. ("program_bound.metal".to_string(), metal_program_bound(p, ds.log2_words)), - ("kernel_bound.cu".to_string(), cuda_kernel_bound(p, memhard)), - ("kernel_bound.cl".to_string(), opencl_kernel_bound(p, memhard)), + ("kernel_bound.cu".to_string(), cuda_kernel_bound_at(p, memhard, ds.log2_words)), + ("kernel_bound.cl".to_string(), opencl_kernel_bound_at(p, memhard, ds.log2_words)), ]; if let Some(mp) = memhard { files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp))); - files.push(("memhard.metal".to_string(), metal_memhard(mp))); + files.push(("memhard.metal".to_string(), metal_memhard_for(p, mp))); } Pack { files, bases, outs, vectors: v } } diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index c3d812307..5201d950b 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -18,9 +18,28 @@ //! The retired version 1 generator (op rolled per instruction with a 25 percent load weight, no acceptance) is //! kept as [`generate_v1`] for the census tool and the lever measurements of `proto-metal/MEMHARD.md`. Its //! programs are not the lottery hash and no pack or vector of version 1 is current. +//! +//! Era layout (5 October 2026, Counter ASIC 2.0 layers 4 and 8, `docs/plans/era-layout.md`; NOT the lottery hash, +//! behind [`LoadClass::era`]): [`EraParams`] drawn from the era seed `E_n` by [`era_draw`] (the load width, a stride +//! multiplier and rotation, the interleave of item words over the dataset), and per load site two more draws (a +//! window of the dataset: a half, a quarter or all of it, at a drawn offset). The load address is +//! [`crate::verify::load_index`]. An era class takes 12 draws per instruction, so its stream differs from version 2. +//! +//! Read-width experiment (5 October 2026, gate 1, `docs/plans/read-width.md`; NOT the lottery hash, behind +//! [`LoadClass`]): a program class whose `load` reads `W` bytes (4, 16 or 64: 1, 4 or 16 words, aligned to `W`) +//! and folds every word into `dst` (`verify::fold_words`), with the width fixed per class or drawn per load from +//! an era-fixed mix. The default class [`LoadClass::V2`] is the generator above, draw for draw and byte for byte; +//! every other class takes one extra draw per instruction (the width roll), so its program stream differs from +//! version 2 and its program id carries the class. +//! +//! Hot-table experiment (5 October 2026, Counter ASIC 2.0 layer 5, `docs/plans/hot-table.md`; NOT the lottery hash, +//! behind [`LoadClass::hot`]): `k` of the load slots read a second table `H` of `S` MiB derived from the epoch seed +//! ([`crate::memhard::HotTable`]) at `H[mulhi(src, words)]` with the plain one-word fold. The hot slots are the +//! first `k` drawn load slots after the scratch slots (a uniform `k`-subset, no extra draw), so a class with the +//! version 2 widths and no scratch takes the version 2 stream exactly ([`LoadClass::takes_width_roll`]). use crate::accept::{check, Reject}; -use crate::seed::{fnv1a64, program_rng, seed_words_from_bytes}; +use crate::seed::{fnv1a64, program_rng, seed_words_from_bytes, SplitMix64}; /// Iterations of the instruction list per hash. pub const ITERATIONS: usize = 8; @@ -53,6 +72,12 @@ pub enum Op { Load, /// Warp-coalesced load (lever b of the version 1 generator). Never emitted by version 2. WLoad, + /// Scratch read-modify-write (read-width experiment, variant 5, 5 October 2026): a 16-byte slot of the lane's + /// own 32 KiB of the warp's 1 MiB scratch, read, folded into dst, rewritten. Never emitted by version 2. + Scratch, + /// Hot-table load (hot-table experiment, 5 October 2026): `dst = dst XOR H[mulhi(src, HOT_WORDS)]`, one word + /// of the epoch's `S` MiB table. Never emitted by version 2. + Hot, } impl Op { @@ -71,6 +96,8 @@ impl Op { Op::Shfl => "shfl", Op::Load => "load", Op::WLoad => "wload", + Op::Scratch => "scratch", + Op::Hot => "hot", } } @@ -88,6 +115,8 @@ impl Op { "shfl" => Op::Shfl, "load" => Op::Load, "wload" => Op::WLoad, + "scratch" => Op::Scratch, + "hot" => Op::Hot, _ => return None, }) } @@ -95,11 +124,13 @@ impl Op { /// An injecting op: bijective in `dst` and bringing another register (or the dataset) in. The acceptance /// rule's part (b) requires one such write per register. pub fn injects(self) -> bool { - matches!(self, Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl | Op::Load | Op::WLoad) + matches!(self, Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl | Op::Load | Op::WLoad | Op::Scratch | Op::Hot) } + /// A memory operation: the fresh-source rule, the acceptance tests and the load count treat the scratch + /// read-modify-write and the hot-table load as loads (each is one of the program's 128 memory operations). pub fn is_load(self) -> bool { - matches!(self, Op::Load | Op::WLoad) + matches!(self, Op::Load | Op::WLoad | Op::Scratch | Op::Hot) } } @@ -124,6 +155,14 @@ pub struct Instr { pub bit: u8, /// Shuffle xor mask: 1, 2, 4, 8 or 16. pub mask: u8, + /// Words read by a `load`: 1 (the lottery hash, 4 bytes), 4 or 16 (the read-width experiment). 1 on every + /// other op. + pub width: u8, + /// Era layout, layer 8: the window shrink of this load site, 0..2 (the dataset, a half, a quarter). 0 on every + /// op of every other class. + pub win: u8, + /// Era layout, layer 8: which aligned window, below `2^win`. 0 on every op of every other class. + pub off: u8, } #[derive(Clone, Debug, PartialEq, Eq)] @@ -139,9 +178,597 @@ pub struct Program { pub generator: u32, /// Attempt index: 0 for the bare seed, `k` for the k-th re-derivation after rejections. pub attempt: u32, + /// The load class: [`LoadClass::V2`] for the lottery hash, another for the read-width experiment. + pub class: LoadClass, + /// The era seed bytes a class v3 chain program was drawn under (`E_n` of spec 04 section 4.4, the devnet stand-in + /// of `docs/plans/era-layout.md` section 2), recorded in the pack so a worker can check it carries the era the + /// job names. `None` for every version 2 program and every string-seed pack. The placeholder [`V3_CLASS`] does + /// not read it; the era draw of the integration branch will. + pub era_bytes: Option>, pub instrs: Vec, } +/// The widths a `load` may read, in words: 4, 16 and 64 bytes. +pub const WIDTH_WORDS: [u8; 3] = [1, 4, 16]; + +/// The load class of a program (read-width experiment, 5 October 2026). `mix` holds the percent weights of the +/// three widths of [`WIDTH_WORDS`] (sum 100); `load_slots` the number of `load` instructions per program. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct LoadClass { + pub mix: [u8; 3], + pub load_slots: u8, + /// Variant 5: `Some(k)` gives the program a per-warp scratch (the kernels run persistent warps) and turns `k` + /// of the load slots into scratch read-modify-writes. `None` for every other class. + pub scratch: Option, + /// Variant 5: the scratch per warp in KiB (32 or 128; the whole working set of a card at full occupancy must + /// stay under 6 GB, coordinator's cap of 5 October 2026). 0 for every other class. + pub scratch_kb: u8, + /// Mixer cost multiplier `m` of the dataset item derivation (Counter ASIC 2.0, M16, decided 5 October 2026 for + /// class v3): every mixer application of spec 01 section 1.8.5 becomes `m` applications with distinct round + /// keys, the 8 dependent cache reads per item unchanged (`memhard::derive_items`). 1 for version 2, 4 for v3. + pub mixer_mult: u8, + /// Cache growth rule, option C (`memhard::growth_doublings`): the cache doubles when the dataset doubles. `false` + /// for version 2 (the cache is 2^26 words on every day), `true` for v3. + pub growth: bool, + /// Era layout (`docs/plans/era-layout.md`): `Some` turns on the strided, windowed load address and the + /// interleaved dataset mapping with the parameters drawn from the era seed. `None` for every other class. + pub era: Option, + /// Hot table (`docs/plans/hot-table.md`, measured 5 October 2026 and not adopted): `Some(HotClass { mb, k, added })` + /// turns `k` load slots into reads of an `mb` MiB epoch table. `None` for every other class, class v3 included. + pub hot: Option, +} + +/// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of +/// spec 01 section 1.13.1). `Copy` so the class stays `Copy`; the eight words of the era stream's seed and the era +/// index are carried so a pack can say where the draw came from. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct EraParams { + /// `seed_words_from_bytes("igneum-era/" || E_n)` (`E_n` commits to the era index through the VDF input of spec + /// 04 section 4.4 step 2, so the index does not enter the draw). + pub words: [u32; 8], + /// The genesis-fixed set the width is drawn from, ascending, zero-padded (`[1, 0, 0]` pins 4 bytes: the + /// read-width decision of 5 October 2026, v2's 128 x 4 B stays). + pub allowed: [u8; 3], + /// Layer 4, item size: the words one load folds (1, 4 or 16), drawn from the allowed set. + pub width_words: u8, + /// Layer 4, stride: `y = rotl(x * stride_mul, stride_rot)`; the multiplier is odd, the rotation in 1..31. + pub stride_mul: u32, + pub stride_rot: u32, + /// Layer 4, interleave: the four ascending bit positions (0..15) of the word-within-item bits in the word + /// index; the first `log2(width_words)` are `0..`, so one aligned load stays inside one item. + pub pos: [u8; 4], +} + +/// Domain tag of the era stream seed. +pub const ERA_TAG: &[u8] = b"igneum-era/"; + +impl EraParams { + /// `seed_words_from_bytes("igneum-era/" || era_bytes)`. + pub fn stream_words(era_bytes: &[u8]) -> [u32; 8] { + let mut b = Vec::with_capacity(ERA_TAG.len() + era_bytes.len()); + b.extend_from_slice(ERA_TAG); + b.extend_from_slice(era_bytes); + seed_words_from_bytes(&b) + } + + /// The short label of the era in class names and pack lines: the first stream word as hex. + pub fn label(&self) -> String { + format!("{:08x}", self.words[0]) + } + + /// The era bytes of a test seed string: the 32 bytes (little-endian words) of `seed_words_from_bytes(s)`. + pub fn test_era_bytes(s: &str) -> [u8; 32] { + let w = seed_words_from_bytes(s.as_bytes()); + let mut out = [0u8; 32]; + for (i, x) in w.iter().enumerate() { + out[i * 4..i * 4 + 4].copy_from_slice(&x.to_le_bytes()); + } + out + } + + /// The dataset layout this era's loads and build use. + pub fn layout(&self) -> crate::memhard::Layout { + crate::memhard::Layout { pos: self.pos } + } + + /// The allowed widths as a slice (the non-zero entries). + pub fn allowed_set(&self) -> Vec { + self.allowed.iter().copied().filter(|&w| w != 0).collect() + } + + /// The bytes that enter the program id after `"era/"`: the allowed set, width, multiplier, rotation, positions. + pub fn id_bytes(&self) -> Vec { + let mut b = Vec::with_capacity(3 + 1 + 4 + 4 + 4); + b.extend_from_slice(&self.allowed); + b.push(self.width_words); + b.extend_from_slice(&self.stride_mul.to_le_bytes()); + b.extend_from_slice(&self.stride_rot.to_le_bytes()); + b.extend_from_slice(&self.pos); + b + } +} + +/// The era draw (`docs/plans/era-layout.md` section 1.1): seven draws from one SplitMix64 stream seeded with words 0 +/// and 1 of [`EraParams::stream_words`], in this order: the width from `allowed` (ascending, a non-empty subset of +/// [`WIDTH_WORDS`]; one element pins it, the draw is still consumed), the odd stride multiplier, the stride rotation +/// in 1..31, then four draws for the interleave (a partial Fisher-Yates over the candidate positions `log2(W)..15`, +/// `4 - log2(W)` of them used, the rest consumed). +pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams { + assert!(!allowed.is_empty() && allowed.len() <= 3, "the allowed width set has 1 to 3 entries"); + for (i, &w) in allowed.iter().enumerate() { + assert!(WIDTH_WORDS.contains(&w), "allowed width {w} is not 1, 4 or 16 words"); + assert!(i == 0 || allowed[i - 1] < w, "the allowed width set is ascending"); + } + let words = EraParams::stream_words(era_bytes); + let mut s = SplitMix64::new(words[0] as u64 | ((words[1] as u64) << 32)); + let width_words = allowed[s.below(allowed.len() as u64) as usize]; + let stride_mul = (s.next() as u32) | 1; + let stride_rot = 1 + s.below(31) as u32; + let b = width_words.trailing_zeros() as usize; // 0, 2 or 4 + let free = 4 - b; + let mut c: Vec = (b as u8..16).collect(); + let mut r = [0u64; 4]; + for x in r.iter_mut() { + *x = s.next(); + } + for i in 0..free { + let n = c.len() - i; + let j = i + (r[i] % n as u64) as usize; + c.swap(i, j); + } + let mut chosen: Vec = c[..free].to_vec(); + chosen.sort_unstable(); + let mut pos = [0u8; 4]; + for i in 0..b { + pos[i] = i as u8; + } + for (i, p) in chosen.iter().enumerate() { + pos[b + i] = *p; + } + let mut al = [0u8; 3]; + al[..allowed.len()].copy_from_slice(allowed); + EraParams { words, allowed: al, width_words, stride_mul, stride_rot, pos } +} + +/// The hot table of a class: `mb` MiB (32, 64 or 96 in the experiment) and `k` hot slots. Two forms: `replaced` +/// (`added = false`): `k` of the 16 load slots read the table, 16 - k dataset loads; `added` (`added = true`, +/// coordinator's form of 5 October 2026 against the on-die-cache recompute chip): the program has 16 + k load slots, +/// the `k` hot ones drawn among them, so the 16 dataset loads and the 4,096-item verifier bound are unchanged and the +/// hot loads are extra work (a cache hit on a GPU, SRAM and a read on a chip). +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct HotClass { + pub mb: u8, + pub k: u8, + pub added: bool, +} + +/// Scratch geometry (variant 5): 16-byte slots, lane-major, 32 lanes per warp; `scratch_kb` KiB per warp gives +/// `scratch_kb x 2` slots per lane (32 KiB: 64 slots, 128 KiB: 256 slots). +pub const SCRATCH_SLOT_BYTES: usize = 16; + +impl LoadClass { + /// Slots per lane of the scratch (0 without one). + pub fn scratch_slots_per_lane(&self) -> usize { + self.scratch_kb as usize * 1024 / LANES / SCRATCH_SLOT_BYTES + } + pub fn scratch_slot_mask(&self) -> u32 { + self.scratch_slots_per_lane().saturating_sub(1) as u32 + } + pub fn scratch_words_per_lane(&self) -> usize { + self.scratch_slots_per_lane() * 4 + } + pub fn scratch_bytes_per_warp(&self) -> usize { + self.scratch_kb as usize * 1024 + } +} + +impl LoadClass { + /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash. + pub const V2: LoadClass = + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None }; + + /// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): + /// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the + /// mixer applied 4 times per round, and the cache growth rule. Name "mx4". + pub const MX4: LoadClass = + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None }; + + /// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed` + /// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base + /// class's mix stands (the width rule of the read-width branch, pinned) and the era's `width_words` is the widest + /// width that mix can draw. + pub fn era(base: LoadClass, era_bytes: &[u8], allowed: &[u8]) -> LoadClass { + let mut e = era_draw(era_bytes, allowed); + let mut c = base; + if allowed.len() > 1 { + let i = WIDTH_WORDS.iter().position(|&w| w == e.width_words).unwrap(); + c.mix = [0, 0, 0]; + c.mix[i] = 100; + } else { + let widest = (0..3).rev().find(|&i| c.mix[i] > 0).map(|i| WIDTH_WORDS[i]).unwrap_or(1); + if widest != e.width_words { + // the interleave must keep the widest load inside one item: redraw the positions for that width + e = era_draw(era_bytes, &[widest]); + } + } + c.era = Some(e); + c + } + + /// The dataset layout of this class ([`crate::memhard::Layout::LINEAR`] without an era). + pub fn layout(&self) -> crate::memhard::Layout { + self.era.map(|e| e.layout()).unwrap_or(crate::memhard::Layout::LINEAR) + } + + /// The x8 candidate beside [`LoadClass::MX4`] (coordinator's rule of 5 October 2026, 21:30 UTC: x8 enters v3 if + /// the per-warp verify stays under 10 ms on one Mac core and the daily 1 GiB build under 1 s on every card): + /// the same loads and growth rule, the mixer applied 8 times per round. Name "mx8". + pub const MX8: LoadClass = + LoadClass { mixer_mult: 8, ..LoadClass::MX4 }; + + /// A fixed width (1, 4 or 16 words) with `load_slots` loads per program. + pub fn fixed(width_words: u8, load_slots: u8) -> LoadClass { + let mut mix = [0u8; 3]; + let i = WIDTH_WORDS.iter().position(|&w| w == width_words).expect("width must be 1, 4 or 16 words"); + mix[i] = 100; + LoadClass { mix, load_slots, ..LoadClass::V2 } + } + + /// Per-load width drawn from `mix` (percent for 4, 16, 64 bytes), 16 loads per program. + pub fn mixed(mix: [u8; 3]) -> LoadClass { + assert_eq!(mix.iter().map(|&m| m as u32).sum::(), 100, "the mix must sum to 100"); + LoadClass { mix, ..LoadClass::V2 } + } + + /// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes into a + /// scratch of `kb` KiB per warp (a power of two, 1 to 128: at least one slot per lane, under the 6 GB cap). + pub fn scratch(k: u8, kb: u8) -> LoadClass { + assert!(k as usize <= LOAD_SLOTS); + assert!(kb.is_power_of_two() && kb <= 128, "scratch per warp must be a power of two up to 128 KiB"); + LoadClass { scratch: Some(k), scratch_kb: kb, ..LoadClass::V2 } + } + + /// Hot table, replaced form: version 2 widths, 16 load slots of which `k` read an `mb` MiB epoch table (`hot64k4`). + pub fn hot(mb: u8, k: u8) -> LoadClass { + LoadClass::V2.with_hot(mb, k) + } + + /// Hot table, added form: version 2 widths, 16 + `k` load slots of which `k` read the table (`hot64k4a`), the 16 + /// dataset loads unchanged. + pub fn hot_added(mb: u8, k: u8) -> LoadClass { + LoadClass::V2.with_hot_added(mb, k) + } + + /// This class with a hot table in the replaced form (composes with a width mix or a scratch: the hot slots are + /// drawn after the scratch slots, and `scratch + hot` must fit the slot count). + pub fn with_hot(mut self, mb: u8, k: u8) -> LoadClass { + assert!(mb >= 1, "a hot table needs at least 1 MiB"); + assert!(self.scratch_slots() + k as usize <= self.load_slots as usize, "scratch and hot slots exceed the load slots"); + self.hot = Some(HotClass { mb, k, added: false }); + self + } + + /// This class with a hot table in the added form: `k` load slots are added to the class's and read the table. + pub fn with_hot_added(mut self, mb: u8, k: u8) -> LoadClass { + assert!(mb >= 1, "a hot table needs at least 1 MiB"); + assert!((self.load_slots as usize) + (k as usize) < INSTR_COUNT, "added hot slots exceed the instruction count"); + self.load_slots += k; + self.hot = Some(HotClass { mb, k, added: true }); + self + } + + /// Hot slots per program (0 without a hot table). + pub fn hot_slots(&self) -> usize { + self.hot.map(|h| h.k as usize).unwrap_or(0) + } + + /// Hot slots that were added to the slot count (0 for the replaced form and without a hot table). + pub fn hot_added_slots(&self) -> usize { + self.hot.map(|h| if h.added { h.k as usize } else { 0 }).unwrap_or(0) + } + + /// Load slots that read the dataset: the slot count less the scratch and hot slots. + pub fn dataset_slots(&self) -> usize { + self.load_slots as usize - self.scratch_slots() - self.hot_slots() + } + + /// This class with the mixer multiplier `m` (1, 2, 4, 8 or 16) and the cache growth rule on or off. + pub fn with_mixer(self, mixer_mult: u8, growth: bool) -> LoadClass { + assert!(mixer_mult >= 1 && mixer_mult <= 16 && mixer_mult.is_power_of_two(), "mixer multiplier must be 1, 2, 4, 8 or 16"); + LoadClass { mixer_mult, growth, ..self } + } + + /// The mixer multiplier as the item derivation uses it. + pub fn mixer_mult(&self) -> u32 { + self.mixer_mult as u32 + } + + /// Whether the loads of this class are version 2's: 16 one-word loads, no scratch. Such a class takes no width + /// roll, so its program stream is the version 2 stream draw for draw (the mixer and the cache are properties of + /// the dataset, not of the program). + pub fn v2_loads(&self) -> bool { + self.mix == [100, 0, 0] && self.load_slots as usize == LOAD_SLOTS + self.hot_added_slots() && self.scratch.is_none() + } + + /// Whether every instruction takes the tenth draw (the width roll): every class whose loads are not version 2's. + pub fn takes_width_roll(&self) -> bool { + !self.v2_loads() + } + + /// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`] ("scr4k32": 4 scratch ops, 32 KiB per warp; + /// "mx4": the v3 construction; a trailing "m" and "g" set the mixer multiplier and the growth rule on any + /// load class, "w16m4g" for example). + pub fn parse(s: &str) -> Option { + if s == "mx4" { + return Some(LoadClass::MX4); + } + if s == "mx8" { + return Some(LoadClass::MX8); + } + // the mixer suffix: "...m" then an optional "g" + let (s, growth) = match s.strip_suffix('g') { + Some(base) if base.rsplit_once('m').map(|(_, d)| !d.is_empty() && d.bytes().all(|b| b.is_ascii_digit())).unwrap_or(false) => (base, true), + _ => (s, false), + }; + if let Some((base, digits)) = s.rsplit_once('m') { + if !digits.is_empty() && digits.bytes().all(|b| b.is_ascii_digit()) && !base.is_empty() && !base.ends_with(',') { + let mult: u8 = digits.parse().ok()?; + if mult == 0 || mult > 16 || !mult.is_power_of_two() { + return None; + } + return Some(LoadClass::parse_loads(base)?.with_mixer(mult, growth)); + } + } + if growth { + return None; + } + LoadClass::parse_loads(s) + } + + /// The load part of a class name (no mixer suffix). + fn parse_loads(s: &str) -> Option { + // "+hotk[a]" composes a hot table with any class; "hotk[a]" alone is the version 2 base; + // the "a" suffix is the added form (k slots added to the class's), without it the replaced form + if let Some((base, hot)) = s.split_once("+hot") { + let (added, hot) = match hot.strip_suffix('a') { Some(h) => (true, h), None => (false, hot) }; + let (mb, k) = hot.split_once('k')?; + let (mb, k): (u8, u8) = (mb.parse().ok()?, k.parse().ok()?); + let c = LoadClass::parse_loads(base)?; + if mb == 0 || k == 0 { + return None; + } + if added { + if c.load_slots as usize + k as usize >= INSTR_COUNT { + return None; + } + return Some(c.with_hot_added(mb, k)); + } + if c.scratch_slots() + k as usize > c.load_slots as usize { + return None; + } + return Some(c.with_hot(mb, k)); + } + if let Some(rest) = s.strip_prefix("hot") { + let (added, rest) = match rest.strip_suffix('a') { Some(h) => (true, h), None => (false, rest) }; + let (mb, k) = rest.split_once('k')?; + let (mb, k): (u8, u8) = (mb.parse().ok()?, k.parse().ok()?); + if mb == 0 || k == 0 || k as usize > LOAD_SLOTS { + return None; + } + return Some(if added { LoadClass::hot_added(mb, k) } else { LoadClass::hot(mb, k) }); + } + if let Some(rest) = s.strip_prefix("scr") { + let (k, kb) = rest.split_once('k')?; + let k: u8 = k.parse().ok()?; + let kb: u8 = kb.parse().ok()?; + if k as usize > LOAD_SLOTS || !kb.is_power_of_two() || kb > 128 { + return None; + } + return Some(LoadClass::scratch(k, kb)); + } + let (mix_s, slots) = match s.split_once("x") { + Some((m, n)) if !m.contains(',') => (m, n.parse::().ok()?), + _ => (s, LOAD_SLOTS as u8), + }; + let mix: [u8; 3] = match mix_s { + "v2" => return Some(LoadClass::V2), + "w4" => [100, 0, 0], + "w16" => [0, 100, 0], + "w64" => [0, 0, 100], + m => { + let v: Vec = m.split(',').map(|x| x.trim().parse::().ok()).collect::>>()?; + if v.len() != 3 || v.iter().map(|&x| x as u32).sum::() != 100 { + return None; + } + [v[0], v[1], v[2]] + } + }; + if slots == 0 || slots as usize >= INSTR_COUNT { + return None; + } + Some(LoadClass { mix, load_slots: slots, ..LoadClass::V2 }) + } + + /// Scratch read-modify-writes per program (0 without a scratch). + pub fn scratch_slots(&self) -> usize { + self.scratch.unwrap_or(0) as usize + } + + pub fn is_v2(&self) -> bool { + *self == LoadClass::V2 + } + + /// "v2", "w4", "w16", "w64", "w64x4", "mix50-35-15", "mix25-50-25x8", "scr4k32"; "mx4" for the v3 construction, "mx8" for its x8 candidate; + /// any other mixer setting appends "m" and, with the growth rule, "g" ("v2m2", "w16m4g"). + /// An era class is the base name with "-era" appended ("w4-era401998a5", "mx4-era..."). + /// A hot class appends "hotk[a]" ("hot64k4", "scr4k32+hot64k4a"; measured and not adopted). + pub fn name(&self) -> String { + if let Some(e) = self.era { + let base = LoadClass { era: None, ..*self }; + let base_name = if base.is_v2() { "w4".to_string() } else { base.name() }; + return format!("{base_name}-era{}", e.label()); + } + let base = LoadClass { hot: None, load_slots: self.load_slots - self.hot_added_slots() as u8, ..*self }.base_name(); + match self.hot { + None => base, + Some(h) if base == "v2" => format!("hot{}k{}{}", h.mb, h.k, if h.added { "a" } else { "" }), + Some(h) => format!("{base}+hot{}k{}{}", h.mb, h.k, if h.added { "a" } else { "" }), + } + } + + /// The name without the hot table. + fn base_name(&self) -> String { + if self.is_v2() { + return "v2".to_string(); + } + if *self == LoadClass::MX4 { + return "mx4".to_string(); + } + if *self == LoadClass::MX8 { + return "mx8".to_string(); + } + let loads = LoadClass { mixer_mult: 1, growth: false, ..*self }; + let base = if loads.is_v2() { + "v2".to_string() + } else if let Some(k) = self.scratch { + format!("scr{k}k{}", self.scratch_kb) + } else { + let base = match self.mix { + [100, 0, 0] => "w4".to_string(), + [0, 100, 0] => "w16".to_string(), + [0, 0, 100] => "w64".to_string(), + [a, b, c] => format!("mix{a}-{b}-{c}"), + }; + if self.load_slots as usize == LOAD_SLOTS { + base + } else { + format!("{base}x{}", self.load_slots) + } + }; + let mut s = base; + if self.mixer_mult != 1 || self.growth { + s.push_str(&format!("m{}", self.mixer_mult)); + } + if self.growth { + s.push('g'); + } + s + } + + /// The width in words of a load whose width roll (0..99) is `roll`: the first entry of the mix whose cumulative + /// weight exceeds the roll. + pub fn width_for_roll(&self, roll: u64) -> u8 { + let mut acc = 0u64; + for (i, &m) in self.mix.iter().enumerate() { + acc += m as u64; + if roll < acc { + return WIDTH_WORDS[i]; + } + } + WIDTH_WORDS[2] + } + + /// Expected dataset bytes read per hash: dataset loads per hash times the mean width (scratch traffic apart). + pub fn expected_bytes_per_hash(&self) -> f64 { + let mean = self.mix.iter().zip(WIDTH_WORDS.iter()).map(|(&m, &w)| m as f64 / 100.0 * w as f64 * 4.0).sum::(); + (self.load_slots as usize - self.scratch_slots()) as f64 * ITERATIONS as f64 * mean + } +} + +impl Default for LoadClass { + fn default() -> Self { + LoadClass::V2 + } +} + +/// Generator version of a class v3 program (spec 01 section 1.4.6: `program_id(3, seed, attempt)`). +pub const GENERATOR_VERSION_V3: u32 = 3; + +/// The program class of an epoch (Counter ASIC 2.0, 5 October 2026, `docs/plans/counter-asic-2-rollout.md`): one +/// height switch in the node, `program_class_v3_activation_daa`, rounded up to an epoch boundary, decides which +/// class an epoch's program is drawn from. V2 is the lottery hash as adopted on 4 October 2026, byte for byte. +/// V3 is generator version 3: its program id carries `generator = 3` and its load class is [`V3_CLASS`]. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Default)] +pub enum ProgramClass { + #[default] + V2, + V3, +} + +/// The load class of program class v3, decided 5 October 2026 (Counter ASIC 2.0, `docs/plans/counter-asic-2-status.md` +/// "22:00 decided", `docs/plans/mixer-x4.md`): [`LoadClass::MX4`], version 2 loads (the width stays 4 bytes, the +/// per-load mix and the scratch share are out), the mixer applied 4 times per round and the cache growth rule. The +/// placeholder of the seam (w16) is replaced here; nothing else in the seam names the class. +/// Composed on 5 October 2026 (branch ca2-era): the era layout of `docs/plans/era-layout.md` is drawn inside this class by +/// [`generate_from_seed_bytes_program_class`] (`LoadClass::era(V3_CLASS, era, &V3_ALLOWED)`); here `era` is `None`. +pub const V3_CLASS: LoadClass = LoadClass { era: None, hot: None, ..LoadClass::MX8 }; + +/// The width set class v3's era draw chooses from: 4 bytes only (the read-width decision of 5 October 2026; the +/// draw is consumed, so widening the set at genesis keeps the derivation). +pub const V3_ALLOWED: [u8; 1] = [1]; + +/// The program of an era class (generator version 3): `base` with the era parameters drawn from `era_bytes` +/// over the width set `allowed`, generator 3 stamped and the era bytes recorded (`docs/plans/era-layout.md`). +/// The chain's path is this with `base = V3_CLASS` and `allowed = V3_ALLOWED`. +pub fn generate_era(seed_string: &str, seed_bytes: &[u8], base: LoadClass, era_bytes: &[u8], allowed: &[u8]) -> Program { + let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, LoadClass::era(base, era_bytes, allowed)); + p.generator = GENERATOR_VERSION_V3; + p.era_bytes = Some(era_bytes.to_vec()); + p +} + +impl ProgramClass { + /// The load class this program class draws from. + pub fn load_class(&self) -> LoadClass { + match self { + ProgramClass::V2 => LoadClass::V2, + ProgramClass::V3 => V3_CLASS, + } + } + + /// The generator version written into every pack and program id of this class. + pub fn generator_version(&self) -> u32 { + match self { + ProgramClass::V2 => GENERATOR_VERSION, + ProgramClass::V3 => GENERATOR_VERSION_V3, + } + } + + /// The class of a generator version: 2 and 3 are the two classes, anything else is no class this crate runs. + pub fn from_generator(generator: u32) -> Option { + match generator { + GENERATOR_VERSION => Some(ProgramClass::V2), + GENERATOR_VERSION_V3 => Some(ProgramClass::V3), + _ => None, + } + } + + /// "v2" or "v3": the `IGNEUM_PROGRAM_CLASS` string of a pack and the `class=` token of a job line. + pub fn name(&self) -> &'static str { + match self { + ProgramClass::V2 => "v2", + ProgramClass::V3 => "v3", + } + } + + pub fn parse(s: &str) -> Option { + match s.trim() { + "v2" => Some(ProgramClass::V2), + "v3" => Some(ProgramClass::V3), + _ => None, + } + } + + /// The generator version as a byte, for wire formats that carry the class as a number. + pub fn as_u8(&self) -> u8 { + self.generator_version() as u8 + } + + pub fn from_u8(v: u8) -> Option { + Self::from_generator(v as u32) + } +} + impl Program { pub fn loads_per_hash(&self) -> usize { self.instrs.iter().filter(|i| i.op.is_load()).count() * ITERATIONS @@ -152,9 +779,43 @@ impl Program { pub fn has_wide(&self) -> bool { self.instrs.iter().any(|i| i.op == Op::WLoad) } - /// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load. + /// Dataset bytes read per hash: 4 per one-word load, 16 and 64 for the wider loads of the experiment. + pub fn bytes_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op == Op::Load).map(|i| i.width as usize * 4).sum::() * ITERATIONS + } + /// Scratch read-modify-writes per hash (variant 5): each reads 16 bytes and writes 16 bytes. + pub fn scratch_ops_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op == Op::Scratch).count() * ITERATIONS + } + pub fn has_scratch(&self) -> bool { + self.class.scratch.is_some() + } + /// Hot-table loads per hash (one 4-byte word each). + pub fn hot_loads_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op == Op::Hot).count() * ITERATIONS + } + pub fn has_hot(&self) -> bool { + self.class.hot.is_some() + } + /// The hot table's words (0 without one). + pub fn hot_words(&self) -> u32 { + self.class.hot.map(|h| crate::memhard::hot_words(h.mb as u32)).unwrap_or(0) + } + /// Width histogram of the loads, in words: (1, 4, 16) counts. + pub fn width_counts(&self) -> [usize; 3] { + let mut c = [0usize; 3]; + for i in self.instrs.iter().filter(|i| i.op == Op::Load) { + if let Some(k) = WIDTH_WORDS.iter().position(|&w| w == i.width) { + c[k] += 1; + } + } + c + } + /// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load; scratch and + /// hot loads touch none (so the added form keeps 4,096). pub fn items_per_warp(&self) -> usize { - (self.loads_per_hash() - self.wide_loads_per_hash()) * 32 + self.wide_loads_per_hash() * 2 + (self.loads_per_hash() - self.wide_loads_per_hash() - self.scratch_ops_per_hash() - self.hot_loads_per_hash()) * 32 + + self.wide_loads_per_hash() * 2 } /// Op histogram, count descending then name ascending. pub fn histogram(&self) -> Vec<(&'static str, usize)> { @@ -177,7 +838,18 @@ impl Program { /// || attempt_le32`. Written into every pack so a version 1 program, or another attempt of the same seed, /// can never be mistaken for this one. pub fn program_id(&self) -> u64 { - program_id(self.generator, &self.seed, self.attempt) + if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 { + // Spec 01 section 1.4.6: a class v3 program's id is `program_id(3, seed, attempt)`; the generator version + // in the preimage separates it from every version 2 program of the same seed + program_id(self.generator, &self.seed, self.attempt) + } else { + program_id_class(self.generator, &self.seed, self.attempt, &self.class) + } + } + + /// The program class of this program, from its generator version (3 = v3, everything else v2). + pub fn program_class(&self) -> ProgramClass { + if self.generator == GENERATOR_VERSION_V3 { ProgramClass::V3 } else { ProgramClass::V2 } } } @@ -192,6 +864,49 @@ pub fn program_id(generator: u32, seed: &[u32; 8], attempt: u32) -> u64 { fnv1a64(&b) } +/// Domain tag of the program id of a read-width class (never collides with [`PROGRAM_ID_TAG`]). +pub const PROGRAM_ID_TAG_RW: &[u8] = b"igneum-program-rw/"; + +/// The program id of a non-default class: the tag, then the same fields as [`program_id`], then the three mix +/// percentages and the slot count as bytes. +pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &LoadClass) -> u64 { + let mut b = Vec::with_capacity(PROGRAM_ID_TAG_RW.len() + 4 + 32 + 4 + 4); + b.extend_from_slice(PROGRAM_ID_TAG_RW); + b.extend_from_slice(&generator.to_le_bytes()); + for w in seed { + b.extend_from_slice(&w.to_le_bytes()); + } + b.extend_from_slice(&attempt.to_le_bytes()); + b.extend_from_slice(&class.mix); + b.push(class.load_slots); + if let Some(k) = class.scratch { + b.extend_from_slice(b"scratch/"); + b.push(k); + b.push(class.scratch_kb); + } + if class.mixer_mult != 1 || class.growth { + // Counter ASIC 2.0: the mixer multiplier and the growth rule are part of the construction, so a program of + // the same seed under a different mixer carries a different id (under the v3 seam the id is + // program_id(3, seed, attempt) and this branch is not taken) + b.extend_from_slice(b"mixer/"); + b.push(class.mixer_mult); + b.push(class.growth as u8); + } + if let Some(e) = class.era { + b.extend_from_slice(b"era/"); + b.extend_from_slice(&e.id_bytes()); + } + if let Some(h) = class.hot { + b.extend_from_slice(b"hot/"); + b.push(h.mb); + b.push(h.k); + if h.added { + b.extend_from_slice(b"added"); + } + } + fnv1a64(&b) +} + /// Weights of the ten non-load families under version 2, in draw order. Sum 75. The load family has no /// weight: its count is fixed by [`LOAD_SLOTS`]. pub const NONLOAD_WEIGHTS: [(Op, u64); 10] = [ @@ -237,20 +952,44 @@ pub fn attempt_words(seed_bytes: &[u8], attempt: u32) -> [u32; 8] { /// One version 2 candidate from its seed words, before the acceptance rule. Spec 01 section 1.4.3: 16 slot draws, /// then nine draws per instruction, 592 per program. pub fn candidate_from_words(seed_string: &str, seed_bytes: &[u8], seed: [u32; 8], attempt: u32) -> Program { + candidate_from_words_class(seed_string, seed_bytes, seed, attempt, LoadClass::V2) +} + +/// [`candidate_from_words`] for a load class. For [`LoadClass::V2`] this is the version 2 draw stream exactly; +/// for any other class the slot count is the class's and every instruction takes a tenth draw, `below(100)`, +/// the width roll (used only on a load slot, drawn on every slot so the stream stays uniform). +pub fn candidate_from_words_class( + seed_string: &str, + seed_bytes: &[u8], + seed: [u32; 8], + attempt: u32, + class: LoadClass, +) -> Program { let mut rng = program_rng(&seed); - // (1) Load slots: a uniform 16-subset of 1..63 by partial Fisher-Yates. Instruction 0 is never a load. + let slots = class.load_slots as usize; + // (1) Load slots: a uniform subset of 1..63 by partial Fisher-Yates. Instruction 0 is never a load. let mut p: [u8; INSTR_COUNT - 1] = [0; INSTR_COUNT - 1]; for (i, slot) in p.iter_mut().enumerate() { *slot = (i + 1) as u8; } - for i in 0..LOAD_SLOTS { + for i in 0..slots { let j = i + rng.below((INSTR_COUNT - 1 - i) as u64) as usize; p.swap(i, j); } let mut is_load = [false; INSTR_COUNT]; - for &slot in &p[..LOAD_SLOTS] { + for &slot in &p[..slots] { is_load[slot as usize] = true; } + // Variant 5: the first k drawn load slots (a uniform k-subset, the draw order is random) are scratch ops. + let mut is_scratch = [false; INSTR_COUNT]; + for &slot in &p[..class.scratch_slots()] { + is_scratch[slot as usize] = true; + } + // Hot table: the next k drawn load slots after the scratch slots are hot loads (again a uniform subset). + let mut is_hot = [false; INSTR_COUNT]; + for &slot in &p[class.scratch_slots()..class.scratch_slots() + class.hot_slots()] { + is_hot[slot as usize] = true; + } // (2) The instructions. `fresh[r]`: r was written by an earlier instruction and no load has read it since. let mut fresh = [false; 8]; let mut instrs = Vec::with_capacity(INSTR_COUNT); @@ -265,10 +1004,16 @@ pub fn candidate_from_words(seed_string: &str, seed_bytes: &[u8], seed: [u32; 8] roll -= w; } if is_load[k] { - op = Op::Load; + op = if is_scratch[k] { + Op::Scratch + } else if is_hot[k] { + Op::Hot + } else { + Op::Load + }; } let dst = rng.below(8); - let src = if op == Op::Load { + let src = if op.is_load() { let mut eligible = [0u64; 8]; let mut n = 0usize; for r in 0..8u64 { @@ -301,11 +1046,26 @@ pub fn candidate_from_words(seed_string: &str, seed_bytes: &[u8], seed: [u32; 8] let rot = 1 + rng.below(31) as u32; let bit = rng.below(32); let mask = 1u8 << rng.below(5); - if op == Op::Load { + // Version 2 loads take no width roll, so a mixer class with version 2 loads draws the version 2 program + let width = if class.takes_width_roll() { class.width_for_roll(rng.below(100)) } else { 1 }; + let width = if op == Op::Load { width } else { 1 }; + // Era layout, layer 8: two window draws per instruction (drawn on every slot, used on a load slot). + let (win, off) = if class.era.is_some() { + let k = rng.below(3) as u8; + let o = (rng.next() as u32 & ((1u32 << k) - 1)) as u8; + if op == Op::Load { + (k, o) + } else { + (0, 0) + } + } else { + (0, 0) + }; + if op.is_load() { fresh[src as usize] = false; } fresh[dst as usize] = true; - instrs.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask }); + instrs.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask, width, win, off }); } Program { seed_string: seed_string.to_string(), @@ -313,6 +1073,8 @@ pub fn candidate_from_words(seed_string: &str, seed_bytes: &[u8], seed: [u32; 8] seed, generator: GENERATOR_VERSION, attempt, + class, + era_bytes: None, instrs, } } @@ -322,6 +1084,11 @@ pub fn candidate(seed_string: &str, seed_bytes: &[u8], attempt: u32) -> Program candidate_from_words(seed_string, seed_bytes, attempt_words(seed_bytes, attempt), attempt) } +/// [`candidate`] for a load class. +pub fn candidate_class(seed_string: &str, seed_bytes: &[u8], attempt: u32, class: LoadClass) -> Program { + candidate_from_words_class(seed_string, seed_bytes, attempt_words(seed_bytes, attempt), attempt, class) +} + /// Why no program could be derived from a seed. #[derive(Clone, Debug, PartialEq, Eq)] pub struct Exhausted { @@ -342,9 +1109,14 @@ impl std::error::Error for Exhausted {} /// This is what the chain calls (`Epoch::from_seed_bytes`) with the 32-byte epoch seed, and what the packs call /// with the UTF-8 of a seed string. pub fn try_generate_from_seed_bytes(seed_string: &str, seed_bytes: &[u8]) -> Result { + try_generate_class(seed_string, seed_bytes, LoadClass::V2) +} + +/// [`try_generate_from_seed_bytes`] for a load class. +pub fn try_generate_class(seed_string: &str, seed_bytes: &[u8], class: LoadClass) -> Result { let mut last = None; for attempt in 0..MAX_ATTEMPTS { - let p = candidate(seed_string, seed_bytes, attempt); + let p = candidate_class(seed_string, seed_bytes, attempt, class); match check(&p) { Ok(_) => return Ok(p), Err(r) => last = Some(r), @@ -358,16 +1130,48 @@ pub fn generate_from_seed_bytes(seed_string: &str, seed_bytes: &[u8]) -> Program try_generate_from_seed_bytes(seed_string, seed_bytes).unwrap_or_else(|e| panic!("{e}")) } +/// [`generate_from_seed_bytes`] for a load class. +pub fn generate_from_seed_bytes_class(seed_string: &str, seed_bytes: &[u8], class: LoadClass) -> Program { + try_generate_class(seed_string, seed_bytes, class).unwrap_or_else(|e| panic!("{e}")) +} + +/// The program of a program class (what the chain calls through `Epoch::from_chain_seeds`): class v2 is +/// [`generate_from_seed_bytes`] exactly; class v3 draws from [`V3_CLASS`] and stamps generator version 3 on the +/// program, so its packs and its id say generator 3 (spec 01 sections 1.4.5 and 1.4.6). +pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u8], class: ProgramClass, era_bytes: Option<&[u8]>) -> Program { + // Class v3 with an era seed: the era layout (docs/plans/era-layout.md) drawn from the era bytes inside V3_CLASS. + // Without era bytes (a template before the era is known) the bare V3_CLASS stands. + if let (ProgramClass::V3, Some(era)) = (class, era_bytes) { + return generate_era(seed_string, seed_bytes, V3_CLASS, era, &V3_ALLOWED); + } + let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, class.load_class()); + p.generator = class.generator_version(); + // The era is a property of class v3 programs; a version 2 program never records one, so the pinned v2 packs + // and every v2 export stay byte-identical whatever the chain reports for the era + p.era_bytes = if class == ProgramClass::V3 { era_bytes.map(|b| b.to_vec()) } else { None }; + p +} + /// The program of a seed string (its UTF-8 bytes are the program seed). pub fn generate(seed_string: &str) -> Program { generate_from_seed_bytes(seed_string, seed_string.as_bytes()) } +/// [`generate`] for a load class. +pub fn generate_class(seed_string: &str, class: LoadClass) -> Program { + generate_from_seed_bytes_class(seed_string, seed_string.as_bytes(), class) +} + /// Every candidate of a seed up to and including the accepted one, with each rejection. For reports and tests. pub fn attempts(seed_string: &str, seed_bytes: &[u8]) -> Vec<(Program, Result<(), Reject>)> { + attempts_class(seed_string, seed_bytes, LoadClass::V2) +} + +/// [`attempts`] for a load class. +pub fn attempts_class(seed_string: &str, seed_bytes: &[u8], class: LoadClass) -> Vec<(Program, Result<(), Reject>)> { let mut out = Vec::new(); for attempt in 0..MAX_ATTEMPTS { - let p = candidate(seed_string, seed_bytes, attempt); + let p = candidate_class(seed_string, seed_bytes, attempt, class); let verdict = check(&p).map(|_| ()); let accepted = verdict.is_ok(); out.push((p, verdict)); @@ -455,7 +1259,7 @@ pub fn generate_v1_from_words(seed_string: &str, seed: [u32; 8], cfg: &Generator if op == Op::Load && bit * 100 < cfg.wide_frac * 32 { op = Op::WLoad; } - instrs.push(Instr { op, dst: dst as u8, src: a as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask }); + instrs.push(Instr { op, dst: dst as u8, src: a as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask, width: 1, win: 0, off: 0 }); } Program { seed_string: seed_string.to_string(), @@ -463,6 +1267,8 @@ pub fn generate_v1_from_words(seed_string: &str, seed: [u32; 8], cfg: &Generator seed, generator: 1, attempt: 0, + class: LoadClass::V2, + era_bytes: None, instrs, } } @@ -565,6 +1371,325 @@ mod tests { assert_ne!(program_id(2, &p.seed, 0), program_id(2, &p.seed, 1)); } + /// The read-width classes (5 October 2026): the default class is the version 2 stream exactly; a class + /// program has its slot count, widths from its mix only, and an id that separates it from version 2 and from + /// the other classes. + #[test] + fn load_classes() { + let v2 = candidate("igneum-genesis", b"igneum-genesis", 0); + let same = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::V2); + assert_eq!(v2, same); + assert!(v2.instrs.iter().all(|i| i.width == 1)); + assert_eq!(v2.bytes_per_hash(), 512); + assert_eq!(LoadClass::parse("w16"), Some(LoadClass::fixed(4, 16))); + assert_eq!(LoadClass::parse("w64x4"), Some(LoadClass::fixed(16, 4))); + assert_eq!(LoadClass::parse("50,35,15"), Some(LoadClass::mixed([50, 35, 15]))); + assert_eq!(LoadClass::parse("v2"), Some(LoadClass::V2)); + assert_eq!(LoadClass::parse("50,35,10"), None); + assert_eq!(LoadClass::fixed(16, 4).name(), "w64x4"); + assert_eq!(LoadClass::mixed([25, 50, 25]).name(), "mix25-50-25"); + // W = 4 with 16 slots IS the lottery hash: the w4 name parses to the default class + assert_eq!(LoadClass::fixed(1, 16), LoadClass::V2); + assert_eq!(LoadClass::parse("w4"), Some(LoadClass::V2)); + assert_eq!(LoadClass::fixed(1, 16).name(), "v2"); + assert_eq!(LoadClass::fixed(1, 8).name(), "w4x8"); + assert!((LoadClass::mixed([50, 35, 15]).expected_bytes_per_hash() - 2201.6).abs() < 1e-6); + assert!((LoadClass::fixed(16, 4).expected_bytes_per_hash() - 2048.0).abs() < 1e-9); + let mut ids = std::collections::HashSet::new(); + ids.insert(v2.program_id()); + for (name, slots, widths) in [("w4x8", 8, vec![1u8]), ("w16", 16, vec![4]), ("w64", 16, vec![16]), ("w64x4", 4, vec![16]), ("50,35,15", 16, vec![1, 4, 16]), ("25,50,25", 16, vec![1, 4, 16])] { + let c = LoadClass::parse(name).unwrap(); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.class, c); + assert_eq!(p.instrs.len(), INSTR_COUNT); + assert_eq!(p.loads_per_hash(), 8 * slots); + assert_ne!(p.instrs[0].op, Op::Load); + for ins in &p.instrs { + if ins.op == Op::Load { + assert!(widths.contains(&ins.width), "{name}: width {}", ins.width); + } else { + assert_eq!(ins.width, 1); + } + } + assert!(ids.insert(p.program_id()), "{name}: program id collides"); + } + // variant 5: k scratch ops among the 16 memory operations, the rest one-word loads + for (k, kb) in [(0u8, 32u8), (2, 32), (4, 128), (8, 128)] { + let c = LoadClass::parse(&format!("scr{k}k{kb}")).unwrap(); + assert_eq!(c, LoadClass::scratch(k, kb)); + assert_eq!(c.name(), format!("scr{k}k{kb}")); + assert_eq!(c.scratch_bytes_per_warp(), kb as usize * 1024); + assert_eq!(c.scratch_slots_per_lane(), kb as usize * 2); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.scratch_ops_per_hash(), 8 * k as usize); + assert_eq!(p.bytes_per_hash(), (16 - k as usize) * 8 * 4); + assert!(p.instrs.iter().all(|i| i.width == 1)); + assert!(ids.insert(p.program_id()), "scr{k}: program id collides"); + } + assert!(!LoadClass::scratch(0, 32).is_v2()); + assert_ne!(LoadClass::scratch(4, 32).name(), LoadClass::scratch(4, 128).name()); + assert_eq!(LoadClass::parse("scr4"), None); + // a class with the version 2 widths but another slot count takes the extra roll: a different stream + let w4x8 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(1, 8)); + assert_ne!(w4x8.instrs, v2.instrs); + // the mix draws every width over a population + let mut counts = [0usize; 3]; + for i in 0..200u32 { + let s = format!("igneum-rw-mix/{i}"); + let p = candidate_class(&s, s.as_bytes(), 0, LoadClass::mixed([50, 35, 15])); + let c = p.width_counts(); + for k in 0..3 { + counts[k] += c[k]; + } + } + let total = (counts[0] + counts[1] + counts[2]) as f64; + assert_eq!(total as usize, 200 * 16); + assert!((counts[0] as f64 / total - 0.50).abs() < 0.05, "{counts:?}"); + assert!((counts[1] as f64 / total - 0.35).abs() < 0.05, "{counts:?}"); + assert!((counts[2] as f64 / total - 0.15).abs() < 0.05, "{counts:?}"); + } + + /// Counter ASIC 2.0 seam: class v2 is the lottery hash exactly; class v3 is generator 3 on the placeholder load + /// class, with an id that no version 2 program of the seed can carry; the class round-trips through its name and + /// its generator number. + #[test] + fn program_classes() { + let v2 = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V2, Some(&[7u8; 32])); + let plain = generate("igneum-genesis"); + assert_eq!(v2, plain, "a v2 program never records an era"); + assert_eq!(v2.era_bytes, None); + assert_eq!(v2.program_class(), ProgramClass::V2); + assert_eq!(v2.generator, GENERATOR_VERSION); + let v3 = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&[7u8; 32])); + assert_eq!(v3.generator, GENERATOR_VERSION_V3); + assert_eq!(v3.era_bytes.as_deref(), Some(&[7u8; 32][..])); + let v3_no_era = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, None); + assert_ne!(v3_no_era.instrs, v3.instrs, "the era class takes two more draws per instruction"); + assert_eq!(v3_no_era.class, V3_CLASS); + assert_eq!(v3.program_class(), ProgramClass::V3); + assert_eq!(LoadClass { era: None, ..v3.class }, V3_CLASS, "the era rides inside V3_CLASS"); + assert_eq!(v3.class, LoadClass::era(V3_CLASS, &[7u8; 32], &V3_ALLOWED)); + assert_eq!(v3.class.era.unwrap().width_words, 1); + assert_eq!(v3.class.layout(), v3.class.era.unwrap().layout()); + assert_ne!(generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&[8u8; 32])).class, v3.class); + assert!(check(&v3).is_ok()); + assert_eq!(v3.program_id(), program_id(GENERATOR_VERSION_V3, &v3.seed, v3.attempt)); + assert_ne!(v3.program_id(), program_id(GENERATOR_VERSION, &v3.seed, v3.attempt)); + assert_ne!(v3.program_id(), v2.program_id()); + for c in [ProgramClass::V2, ProgramClass::V3] { + assert_eq!(ProgramClass::parse(c.name()), Some(c)); + assert_eq!(ProgramClass::from_generator(c.generator_version()), Some(c)); + assert_eq!(ProgramClass::from_u8(c.as_u8()), Some(c)); + } + assert_eq!(ProgramClass::from_generator(1), None); + assert_eq!(ProgramClass::from_generator(4), None); + assert_eq!(ProgramClass::parse("v4"), None); + assert_eq!(ProgramClass::default(), ProgramClass::V2); + assert_eq!(ProgramClass::V2.load_class(), LoadClass::V2); + } + + /// Era layout: the draw is deterministic, within bounds, and the six test eras are pinned; an era program has + /// 16 loads with window draws in bounds and nothing drawn on ALU slots; the class names and program ids separate + /// the eras from each other and from every other class. + #[test] + fn era_draw_deterministic_and_bounded() { + let all = [1u8, 4, 16]; + for n in 0..200u64 { + let eb = EraParams::test_era_bytes(&format!("igneum-era-test/{n}")); + let e = era_draw(&eb, &all); + assert_eq!(e, era_draw(&eb, &all)); + assert_eq!(e.words, EraParams::stream_words(&eb)); + assert!(all.contains(&e.width_words)); + assert_eq!(e.stride_mul & 1, 1); + assert!((1..=31).contains(&e.stride_rot)); + assert!(e.layout().is_valid(), "{:?}", e.pos); + let b = e.width_words.trailing_zeros() as usize; + for i in 0..b { + assert_eq!(e.pos[i], i as u8, "the low positions are the identity for width {}", e.width_words); + } + // a pinned set consumes the draw and keeps the stride draws in step + let pinned = era_draw(&eb, &[1]); + assert_eq!(pinned.width_words, 1); + assert_eq!((pinned.stride_mul, pinned.stride_rot), (e.stride_mul, e.stride_rot)); + assert_ne!(era_draw(&EraParams::test_era_bytes(&format!("igneum-era-test/{}", n + 1)), &all).words, e.words); + } + // the six test eras pinned at 4 bytes (docs/plans/era-layout.md section 5): ERA_VECTORS + for (n, mul, rot, pos) in ERA_VECTORS { + let e = era_draw(&EraParams::test_era_bytes(&format!("igneum-era-test/{n}")), &V3_ALLOWED); + assert_eq!((e.width_words, e.stride_mul, e.stride_rot, e.pos), (1, mul, rot, pos), "era test seed {n}"); + } + // era programs + let mut ids = std::collections::HashSet::new(); + ids.insert(candidate("igneum-genesis", b"igneum-genesis", 0).program_id()); + ids.insert(candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(16, 16)).program_id()); + for n in 0..6u64 { + let eb = EraParams::test_era_bytes(&format!("igneum-era-test/{n}")); + let c = LoadClass::era(LoadClass::V2, &eb, &all); + let e = c.era.unwrap(); + assert_eq!(c.mix.iter().position(|&m| m == 100).map(|i| WIDTH_WORDS[i]), Some(e.width_words)); + let base = match e.width_words { + 1 => "w4", + 4 => "w16", + _ => "w64", + }; + assert_eq!(c.name(), format!("{base}-era{}", e.label())); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.class, c); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.bytes_per_hash(), 128 * 4 * e.width_words as usize); + assert_ne!(p.instrs[0].op, Op::Load); + for ins in &p.instrs { + assert_ne!(ins.dst, ins.src); + assert!((1..=31).contains(&ins.rot)); + if ins.op == Op::Load { + assert_eq!(ins.width, e.width_words); + assert!(ins.win <= 2 && (ins.off as u32) < (1u32 << ins.win), "win {} off {}", ins.win, ins.off); + } else { + assert_eq!((ins.width, ins.win, ins.off), (1, 0, 0)); + } + } + assert!(ids.insert(p.program_id()), "era {n}: program id collides"); + // the pinned form keeps the base mix and redraws the interleave for the base's widest width + let pinned = LoadClass::era(LoadClass::mixed([50, 35, 15]), &eb, &[1]); + assert_eq!(pinned.mix, [50, 35, 15]); + assert_eq!(pinned.era.unwrap().width_words, 16); + assert_eq!(pinned.era.unwrap().pos, [0, 1, 2, 3]); + assert!(ids.insert(candidate_class("igneum-genesis", b"igneum-genesis", 0, pinned).program_id())); + } + // the window draws take two more draws per instruction: the stream differs from the read-width class + let w64 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(16, 16)); + let era0 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::era(LoadClass::V2, &EraParams::test_era_bytes("igneum-era-test/0"), &all)); + assert_ne!(w64.instrs, era0.instrs); + } + + /// The era draws of the six test seeds at the 4-byte width: (test seed, stride multiplier, rotation, positions), + /// as `igneum-pow show --era igneum-era-test/` printed them on 5 October 2026 (docs/plans/era-layout.md 5). + const ERA_VECTORS: [(u64, u32, u32, [u8; 4]); 6] = [ + (0, 0x625e5ab3, 19, [0, 2, 10, 15]), + (1, 0xb2a9d70d, 6, [1, 3, 8, 13]), + (2, 0x2b4a5b97, 28, [1, 3, 4, 8]), + (3, 0x27ea7eff, 30, [2, 3, 8, 13]), + (4, 0x4d38603d, 10, [2, 9, 13, 15]), + (5, 0x03ac37ad, 22, [0, 2, 10, 13]), + ]; + + /// The six test eras generate accepted programs through the chain's class v3 path (the acceptance rule with the + /// era address mirror); generator 3, the era bytes recorded, the class V3_CLASS with the era inside. + #[test] + fn era_programs_are_accepted() { + for n in 0..6u64 { + let eb = EraParams::test_era_bytes(&format!("igneum-era-test/{n}")); + let p = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&eb)); + assert!(check(&p).is_ok(), "era {n}"); + assert!(p.attempt < MAX_ATTEMPTS); + assert_eq!(p.generator, GENERATOR_VERSION_V3); + assert_eq!(p.era_bytes.as_deref(), Some(&eb[..])); + assert_eq!(p, generate_era("igneum-genesis", b"igneum-genesis", V3_CLASS, &eb, &V3_ALLOWED)); + } + } + + /// Hot-table experiment: names and ids; a hot class with version 2 widths and no scratch takes the version 2 + /// stream, so its program is the version 2 program with k of the load slots turned into hot loads; the hot + /// table composes with a scratch class. + #[test] + fn hot_classes() { + assert_eq!(LoadClass::parse("hot64k4"), Some(LoadClass::hot(64, 4))); + assert_eq!(LoadClass::hot(64, 4).name(), "hot64k4"); + assert_eq!(LoadClass::parse("hot96k4").unwrap().name(), "hot96k4"); + assert_eq!(LoadClass::parse("hot0k4"), None); + assert_eq!(LoadClass::parse("hot64k17"), None); + assert_eq!(LoadClass::parse("hot64"), None); + // the added form: 16 + k slots, 16 dataset loads, no width roll, its own name and id + for (mb, k) in [(32u8, 4u8), (64, 4), (96, 4)] { + let c = LoadClass::parse(&format!("hot{mb}k{k}a")).unwrap(); + assert_eq!(c, LoadClass::hot_added(mb, k)); + assert_eq!(c.name(), format!("hot{mb}k{k}a")); + assert_eq!(c.load_slots as usize, 16 + k as usize); + assert_eq!(c.dataset_slots(), 16); + assert!(!c.takes_width_roll()); + assert_ne!(c, LoadClass::hot(mb, k)); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.loads_per_hash(), 128 + 8 * k as usize); + assert_eq!(p.hot_loads_per_hash(), 8 * k as usize); + assert_eq!(p.bytes_per_hash(), 512); + assert_eq!(p.items_per_warp(), 4096); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Hot).count(), k as usize); + assert!(p.instrs.iter().all(|i| i.width == 1)); + assert_ne!(p.program_id(), candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(mb, k)).program_id()); + } + assert_eq!(LoadClass::parse("scr4k32+hot64k4a").unwrap().name(), "scr4k32+hot64k4a"); + assert_eq!(LoadClass::parse("scr4k32+hot64k4a").unwrap().dataset_slots(), 12); + assert_eq!(LoadClass::parse("hot64k48a"), None); + assert!(!LoadClass::hot(64, 4).is_v2()); + assert!(!LoadClass::hot(64, 4).takes_width_roll()); + assert!(LoadClass::V2.takes_width_roll() == false); + assert!(LoadClass::scratch(4, 32).takes_width_roll()); + assert!(LoadClass::fixed(4, 16).takes_width_roll()); + let v2 = candidate("igneum-genesis", b"igneum-genesis", 0); + let mut ids = std::collections::HashSet::new(); + ids.insert(v2.program_id()); + for (mb, k) in [(32u8, 4u8), (64, 4), (96, 4), (64, 2), (64, 8)] { + let c = LoadClass::hot(mb, k); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.class, c); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.hot_loads_per_hash(), 8 * k as usize); + assert_eq!(p.bytes_per_hash(), (16 - k as usize) * 8 * 4); + assert_eq!(p.hot_words(), crate::memhard::hot_words(mb as u32)); + assert!(ids.insert(p.program_id()), "hot{mb}k{k}: program id collides"); + // the version 2 program with k loads redirected: every other field identical, instruction by instruction + let mut hot = 0; + for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) { + if a.op == Op::Hot { + hot += 1; + assert_eq!(b.op, Op::Load, "a hot slot is one of the version 2 load slots"); + assert_eq!((a.dst, a.src, a.src2, a.imm, a.imm2, a.rot, a.bit, a.mask, a.width), (b.dst, b.src, b.src2, b.imm, b.imm2, b.rot, b.bit, b.mask, b.width)); + } else { + assert_eq!(a, b); + } + } + assert_eq!(hot, k as usize); + } + // the same k at two sizes: the same instructions, different ids and tables + let a = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(32, 4)); + let b = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(64, 4)); + assert_eq!(a.instrs, b.instrs); + assert_ne!(a.program_id(), b.program_id()); + // composition with a scratch class: scratch slots first, then hot, the rest dataset loads + let c = LoadClass::parse("scr4k32+hot64k4").unwrap(); + assert_eq!(c, LoadClass::scratch(4, 32).with_hot(64, 4)); + assert_eq!(c.name(), "scr4k32+hot64k4"); + assert!(c.takes_width_roll()); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.scratch_ops_per_hash(), 32); + assert_eq!(p.hot_loads_per_hash(), 32); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.bytes_per_hash(), 8 * 8 * 4); + assert!(ids.insert(p.program_id())); + assert_eq!(LoadClass::parse("scr12k32+hot8k8"), None, "scratch and hot slots exceed the 16"); + assert_eq!(LoadClass::parse("w16+hot64k4").unwrap().name(), "w16+hot64k4"); + // the hot slots are a uniform subset of the load slots over a population + let mut position_sum = 0usize; + let mut n = 0usize; + for i in 0..200u32 { + let s = format!("igneum-hot-slots/{i}"); + let p = candidate_class(&s, s.as_bytes(), 0, LoadClass::hot(64, 4)); + let loads: Vec = p.instrs.iter().enumerate().filter(|(_, x)| x.op.is_load()).map(|(i, _)| i).collect(); + assert_eq!(loads.len(), 16); + for (rank, &i) in loads.iter().enumerate() { + if p.instrs[i].op == Op::Hot { + position_sum += rank; + n += 1; + } + } + } + assert_eq!(n, 800); + let mean_rank = position_sum as f64 / n as f64; + assert!((mean_rank - 7.5).abs() < 0.6, "hot slots sit anywhere among the 16 loads: mean rank {mean_rank}"); + } + #[test] fn generate_returns_an_accepted_program() { let p = generate("igneum-genesis"); diff --git a/igneum-pow/src/lib.rs b/igneum-pow/src/lib.rs index a35f37a4c..7cb094d44 100644 --- a/igneum-pow/src/lib.rs +++ b/igneum-pow/src/lib.rs @@ -26,12 +26,13 @@ pub mod bind; pub mod emit; pub mod generator; pub mod memhard; +pub mod packcheck; pub mod seed; pub mod verify; pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256}; pub use accept::{check as accept_program, AcceptReport, Reject}; -pub use generator::{generate, generate_from_seed_bytes, Instr, Op, Program, GENERATOR_VERSION}; -pub use memhard::{Cache, MemhardCpu, MixParams}; +pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, V3_CLASS}; +pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape}; pub use seed::{fnv1a64, seed_words, SplitMix64}; pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch}; diff --git a/igneum-pow/src/main.rs b/igneum-pow/src/main.rs index f09697fa6..cf82ab766 100644 --- a/igneum-pow/src/main.rs +++ b/igneum-pow/src/main.rs @@ -7,11 +7,22 @@ //! [--epoch-hex <64 hex> --day-hex ] byte seeds instead of strings (Epoch::from_seed_bytes) //! igneum-pow accept --seed [--epoch-hex <64 hex>] every candidate of the seed with its verdict (spec 01 section 1.4.6) //! igneum-pow show --seed [--epoch-hex <64 hex>] the accepted program, one instruction per line +//! +//! Read-width experiment (5 October 2026, docs/plans/read-width.md): `--class v2|w4|w16|w64|w64x4|p4,p16,p64[xN]` +//! on every command selects the load class (default v2, the lottery hash). Nothing in a v2 run changes. +//! +//! Era layout (5 October 2026, docs/plans/era-layout.md): `--era igneum-era-test/` (a test era seed: the 32 bytes +//! of seed_words_from_bytes of the string, era index n) or `--era :<64 hex>` (the chain's 32-byte era seed E_n) +//! turns the chosen class into its era class; `--era-widths 4` (default: the read-width decision of 5 October 2026 +//! keeps v2's 4-byte load) is the allowed width set the era draws from; `4,16,64` lets the era draw the width. + +use igneum_pow::generator::{EraParams, GENERATOR_VERSION_V3}; use igneum_pow::emit::export_pack; -use igneum_pow::memhard::Cache; +use igneum_pow::generator::{LoadClass, ProgramClass}; +use igneum_pow::memhard::{Cache, Shape}; use igneum_pow::seed::day_key; -use igneum_pow::verify::{DatasetMode, Epoch, DEFAULT_DATASET_LOG2}; +use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2}; use std::time::Instant; struct Args { @@ -26,6 +37,51 @@ struct Args { prehash: String, epoch_hex: Option, day_hex: Option, + class: LoadClass, + /// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache). + days: u64, + /// The program class (Counter ASIC 2.0 seam): v2 (default) or v3, which draws from V3_CLASS with generator 3. + program_class: Option, + /// The era seed bytes a class v3 chain program records (`--era-hex`). + era_hex: Option, + /// Era layout: `--era igneum-era-test/` or `--era :<64 hex>` composes the era class over `--class` with + /// generator 3 and the era bytes recorded (the measurement packs: v2's mixer under the era layout). + era: Option<(u64, Vec, String)>, + era_widths: Vec, +} + +/// `igneum-era-test/` or `:<64 hex>` -> (index, 32 era bytes, label). +fn parse_era(s: &str) -> Option<(u64, Vec, String)> { + if let Some(n) = s.strip_prefix("igneum-era-test/") { + let index: u64 = n.parse().ok()?; + return Some((index, EraParams::test_era_bytes(s).to_vec(), s.to_string())); + } + let (n, hex) = s.split_once(':')?; + let index: u64 = n.parse().ok()?; + let bytes = igneum_pow::bind::unhex(hex)?; + if bytes.len() != 32 { + return None; + } + Some((index, bytes, format!("igneum-era/{index}/{hex}"))) +} + +/// "4,16,64" (bytes) -> ascending words. +fn parse_widths(s: &str) -> Option> { + let mut v: Vec = s + .split(',') + .map(|x| match x.trim() { + "4" => Some(1u8), + "16" => Some(4), + "64" => Some(16), + _ => None, + }) + .collect::>>()?; + v.sort_unstable(); + v.dedup(); + if v.is_empty() { + return None; + } + Some(v) } fn usage() -> ! { @@ -36,7 +92,12 @@ fn usage() -> ! { \x20 hash --nonce print the 64-bit hash of one nonce (pack form, init words = seed words)\n\ \x20 hash-bound --prehash <64 hex> --nonce print the header-bound hash (bind.rs) of one 64-bit nonce\n\ \x20 accept every candidate of the seed (or --epoch-hex) with its acceptance verdict\n\ - \x20 show the accepted program, one instruction per line" + \x20 show the accepted program, one instruction per line\n\ + \x20 --class C load class: v2 (default), mx4 (class v3: mixer x4, cache growth), w4, w16, w64, w64x4, p4,p16,p64[xN], m[g]\n\ + \x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\ + \x20 --program-class v2|v3 the program class of the seam (v3 = generator 3 on V3_CLASS, the chain's own derivation; --era-hex records the era seed)\n\ + \x20 --era E era layout over --class: igneum-era-test/ or :<64 hex> (the 32-byte era seed E_n)\n\ + \x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)" ); std::process::exit(2) } @@ -54,6 +115,12 @@ fn parse() -> Args { prehash: "00".repeat(32), epoch_hex: None, day_hex: None, + class: LoadClass::V2, + days: 0, + program_class: None, + era_hex: None, + era: None, + era_widths: vec![1], }; let mut it = std::env::args().skip(1); a.cmd = it.next().unwrap_or_else(|| usage()); @@ -70,12 +137,30 @@ fn parse() -> Args { "--prehash" => a.prehash = val(), "--epoch-hex" => a.epoch_hex = Some(val()), "--day-hex" => a.day_hex = Some(val()), + "--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()), + "--days" => a.days = val().parse().unwrap_or_else(|_| usage()), + "--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())), + "--era-hex" => a.era_hex = Some(val()), + "--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())), + "--era-widths" => a.era_widths = parse_widths(&val()).unwrap_or_else(|| usage()), _ => usage(), } } + if let Some((_, bytes, _)) = &a.era { + a.class = LoadClass::era(a.class, bytes, &a.era_widths); + } a } +/// An era program is a class v3 program: generator 3 and the era bytes recorded (what the chain's +/// `Epoch::from_chain_seeds` does); the pack then carries IGNEUM_PROGRAM_CLASS "v3" and IGNEUM_ERA_SEED_HEX. +fn stamp_era(e: &mut Epoch, a: &Args) { + if let Some((_, bytes, _)) = &a.era { + e.program.generator = GENERATOR_VERSION_V3; + e.program.era_bytes = Some(bytes.clone()); + } +} + fn main() { let a = parse(); let mode = if a.closed_form { DatasetMode::ClosedForm } else { DatasetMode::MemoryHard }; @@ -85,21 +170,14 @@ fn main() { "accept" => accept(&a), "show" => show(&a), "hash" => { - let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2); + let (e, _) = epoch_of(&a, mode); println!("{:016x}", e.hash(a.nonce as u32)); } "hash-bound" => { let bytes = igneum_pow::bind::unhex(&a.prehash).unwrap_or_else(|| usage()); let prehash: [u8; 32] = bytes.as_slice().try_into().unwrap_or_else(|_| usage()); // --epoch-hex / --day-hex: the chain's byte seeds (Epoch::from_seed_bytes), as the worker protocol carries them - let e = match (&a.epoch_hex, &a.day_hex) { - (Some(eh), Some(dh)) => { - let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage()); - let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage()); - Epoch::from_seed_bytes(&eb, &db, "cli") - } - _ => Epoch::new(&a.seed, &a.day, mode, a.dataset_log2), - }; + let (e, _) = epoch_of(&a, mode); let init = igneum_pow::bind::block_init_words(&prehash, a.nonce); println!("init words {}", init.iter().map(|w| format!("{w:08x}")).collect::>().join(" ")); println!("{:016x}", e.hash_bound(&prehash, a.nonce)); @@ -108,6 +186,49 @@ fn main() { } } +/// The epoch every command works on, and the day label for packs. `--epoch-hex`/`--day-hex` give the chain's byte +/// seeds (the day label then names the day bytes); else the string seed and day. `--program-class v3` draws the +/// program through the seam (generator 3 on `V3_CLASS`, the era bytes of `--era-hex` recorded) and sizes the +/// dataset for `--days` through `Epoch::chain_dataset_day`; `--class` is ignored under a program class (the class +/// names the load class). Closed-form mode is only for string seeds under the default class. +fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) { + let (mut e, label) = epoch_of_class(a, mode); + stamp_era(&mut e, a); + (e, label) +} + +fn epoch_of_class(a: &Args, mode: DatasetMode) -> (Epoch, String) { + let era = a.era_hex.as_ref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage())); + match (&a.epoch_hex, &a.day_hex) { + (Some(eh), Some(dh)) => { + let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage()); + let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage()); + let label = format!("igneum-epoch/{eh}/day/{dh}"); + let e = match a.program_class { + Some(pc) => Epoch { + program: Epoch::chain_program(&eb, era.as_deref(), pc, &label), + dataset: Epoch::chain_dataset_day(&db, pc, a.days, a.dataset_log2), + }, + None => Epoch::from_seed_bytes_day(&eb, &db, &label, a.class, a.days, a.dataset_log2), + }; + (e, format!("bytes:{dh}")) + } + _ => { + let e = match a.program_class { + Some(pc) => { + let program = igneum_pow::generator::generate_from_seed_bytes_program_class(&a.seed, a.seed.as_bytes(), pc, era.as_deref()); + let lc = pc.load_class(); + let shape = Shape::for_class_day(&lc, a.days); + let log2 = if lc.growth { igneum_pow::memhard::dataset_log2_words(a.dataset_log2, a.days) } else { a.dataset_log2 }; + Epoch { program, dataset: DatasetSource::new_shape(&a.day, mode, log2, shape) } + } + None => Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days), + }; + (e, a.day.clone()) + } + } +} + fn bench(a: &Args, mode: DatasetMode) { println!( "igneum-pow bench: seed \"{}\", day \"{}\", dataset 2^{} words ({})", @@ -116,21 +237,41 @@ fn bench(a: &Args, mode: DatasetMode) { a.dataset_log2, mode.name() ); + let shape = Shape::for_class_day(&a.program_class.map(|pc| pc.load_class()).unwrap_or(a.class), a.days); if mode == DatasetMode::MemoryHard { // Time the cache fill on its own first (one core), then build the epoch (which fills it again). let t0 = Instant::now(); - let c = Cache::fill(day_key(&a.day)); + let c = Cache::fill_log2(day_key(&a.day), shape.cache_log2_words); let fill_ms = t0.elapsed().as_secs_f64() * 1e3; - println!("cache: fill {fill_ms:.1} ms on one core (2^26 words, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", c.fnv1a64()); + println!( + "cache: fill {fill_ms:.1} ms on one core (2^{} words, {} MiB, {} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", + shape.cache_log2_words, + shape.cache_words() * 4 / (1 << 20), + c.segments(), + c.fnv1a64() + ); drop(c); } + if let Some(h) = a.class.hot { + // the hot table of the epoch on its own first (one core), then the epoch (which fills it again) + let t0 = Instant::now(); + let t = igneum_pow::memhard::HotTable::for_seed_bytes(a.seed.as_bytes(), h.mb as u32); + let fill_ms = t0.elapsed().as_secs_f64() * 1e3; + println!("hot table: {} MiB filled in {fill_ms:.1} ms on one core ({} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", h.mb, igneum_pow::memhard::hot_segments(h.mb as u32), t.fnv1a64()); + } let t0 = Instant::now(); - let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2); + let (e, _) = epoch_of(a, mode); let build_ms = t0.elapsed().as_secs_f64() * 1e3; println!( - "program: {} loads/hash, {} items/warp, op mix {}; epoch built in {build_ms:.1} ms", + "program: class {}, {} loads/hash, {} bytes/hash, widths (1,4,16 words) {:?}, {} items/warp, mixer x{} ({} mixers/item), cache 2^{} words, op mix {}; epoch built in {build_ms:.1} ms", + e.program.class.name(), e.program.loads_per_hash(), + e.program.bytes_per_hash(), + e.program.width_counts(), e.program.items_per_warp(), + shape.mixer_mult, + shape.mixers_per_item(), + shape.cache_log2_words, e.program.op_mix() ); let bases = [0u32, 4096, 1_000_000]; @@ -158,14 +299,7 @@ fn export(a: &Args, mode: DatasetMode) { let out = a.out.clone().unwrap_or_else(|| usage()); let t0 = Instant::now(); // --epoch-hex / --day-hex: the chain's byte seeds; the day label then names the day bytes - let (e, day_label) = match (&a.epoch_hex, &a.day_hex) { - (Some(eh), Some(dh)) => { - let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage()); - let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage()); - (Epoch::from_seed_bytes(&eb, &db, &format!("igneum-epoch/{eh}/day/{dh}")), format!("bytes:{dh}")) - } - _ => (Epoch::new(&a.seed, &a.day, mode, a.dataset_log2), a.day.clone()), - }; + let (e, day_label) = epoch_of(a, mode); let build_ms = t0.elapsed().as_secs_f64() * 1e3; println!("igneum-pow export {out}"); println!( @@ -179,7 +313,7 @@ fn export(a: &Args, mode: DatasetMode) { e.program.program_id(), e.program.loads_per_hash() ); - println!("op mix: {}", e.program.op_mix()); + println!("op mix: {}; class {}, {} bytes/hash, widths (1,4,16 words) {:?}", e.program.op_mix(), e.program.class.name(), e.program.bytes_per_hash(), e.program.width_counts()); let source = format!("igneum-pow (Rust) CPU interpreter, generator v{}, {} dataset", e.program.generator, e.dataset.mode().name()); let pack = export_pack(&e, &day_label, &source); let dir = std::path::Path::new(&out); @@ -209,7 +343,7 @@ fn seed_bytes_of(a: &Args) -> (String, Vec) { fn accept(a: &Args) { let (label, bytes) = seed_bytes_of(a); let t0 = Instant::now(); - let tries = igneum_pow::generator::attempts(&label, &bytes); + let tries = igneum_pow::generator::attempts_class(&label, &bytes, a.class); let ms = t0.elapsed().as_secs_f64() * 1e3; for (p, verdict) in &tries { match verdict { @@ -233,19 +367,42 @@ fn accept(a: &Args) { fn show(a: &Args) { let (label, bytes) = seed_bytes_of(a); - let p = igneum_pow::generator::generate_from_seed_bytes(&label, &bytes); + let mut p = igneum_pow::generator::generate_from_seed_bytes_class(&label, &bytes, a.class); + if let Some((_, eb, _)) = &a.era { + p.generator = GENERATOR_VERSION_V3; + p.era_bytes = Some(eb.clone()); + } println!( - "seed \"{}\" generator v{} attempt {} program id {:016x} seed words {}", + "seed \"{}\" generator v{} class {} attempt {} program id {:016x} seed words {}", p.seed_string, p.generator, + p.class.name(), p.attempt, p.program_id(), p.seed.iter().map(|w| format!("{w:08x}")).collect::>().join(" ") ); - println!("op mix {} loads/hash {}", p.op_mix(), p.loads_per_hash()); + println!("op mix {} loads/hash {} bytes/hash {}", p.op_mix(), p.loads_per_hash(), p.bytes_per_hash()); + if let Some(e) = p.class.era { + println!( + "era {} ({}): width {} B, stride mul {:#010x} rot {}, interleave {:?}, windows (site:shrink:offset) {}", + e.label(), + a.era.as_ref().map(|x| x.2.as_str()).unwrap_or("?"), + e.width_words as u32 * 4, + e.stride_mul, + e.stride_rot, + e.pos, + p.instrs + .iter() + .enumerate() + .filter(|(_, i)| i.op == igneum_pow::generator::Op::Load) + .map(|(k, i)| format!("{k}:{}:{}", i.win, i.off)) + .collect::>() + .join(" ") + ); + } for (k, i) in p.instrs.iter().enumerate() { println!( - "{k:2}: {:5} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={} bit={} mask={}", + "{k:2}: {:5} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={} bit={} mask={}{}", i.op.name(), i.dst, i.src, @@ -254,7 +411,8 @@ fn show(a: &Args) { i.imm2, i.rot, i.bit, - i.mask + i.mask, + if i.op == igneum_pow::generator::Op::Load && i.width > 1 { format!(" width={}B", i.width as u32 * 4) } else { String::new() } ); } } diff --git a/igneum-pow/src/memhard.rs b/igneum-pow/src/memhard.rs index 3141f7d04..4458ab865 100644 --- a/igneum-pow/src/memhard.rs +++ b/igneum-pow/src/memhard.rs @@ -3,12 +3,18 @@ //! ARX-multiply mixer. The verifier holds the cache and never the dataset. //! //! All arithmetic is on u32 modulo 2^32. Rotations are by 1..31 at every call site. +//! +//! Counter ASIC 2.0 (5 October 2026, `docs/plans/mixer-x4.md`, behind the program class): the construction has a +//! [`Shape`], the mixer multiplier `m` and the cache size. Under `m` every mixer application of an item becomes +//! `m` applications with distinct round keys, the 8 dependent cache reads unchanged; the cache doubles when the +//! dataset doubles ([`growth_doublings`]). [`Shape::V2`] (`m = 1`, 2^26 words) is version 2 bit for bit. +use crate::generator::LoadClass; use crate::seed::{day_key, fnv1a64_words, SplitMix64}; pub const CACHE_LOG2_WORDS: usize = 26; pub const CACHE_SEGMENT_LOG2_LINES: usize = 6; -/// 2^26 words = 256 MiB. +/// 2^26 words = 256 MiB (the version 2 cache, and the v3 cache until the first dataset doubling). pub const CACHE_WORDS: usize = 1 << CACHE_LOG2_WORDS; /// 2^22 lines of 16 words. pub const CACHE_LINES: usize = CACHE_WORDS >> 4; @@ -23,6 +29,102 @@ pub const CHACHA_ROUNDS: usize = 12; pub const CHACHA_SIGMA: [u32; 4] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574]; /// "Igne", "umMH". pub const CACHE_TAG: [u32; 2] = [0x49676e65, 0x756d4d48]; +/// "Igne", "umHT": the chain tag of the hot table (hot-table experiment, `docs/plans/hot-table.md`). +pub const HOT_TAG: [u32; 2] = [0x49676e65, 0x756d4854]; +/// Domain tag of the hot key: `KH = seed_words_from_bytes("igneum-hot/" || epoch seed bytes)`. +pub const HOT_KEY_TAG: &[u8] = b"igneum-hot/"; +/// Words per MiB of hot table. +pub const HOT_WORDS_PER_MIB: u32 = 1 << 18; +/// Segments (64 chained lines of 16 words, 4 KiB) per MiB of hot table. +pub const HOT_SEGMENTS_PER_MIB: u32 = 256; + +/// The shape of the item derivation and of the cache: the mixer multiplier and the cache size. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct Shape { + /// Mixer applications per round (and after the last read): 1 under version 2, 4 under class v3. + pub mixer_mult: u32, + /// The cache is 2^cache_log2_words words (26 at genesis; 27 and 28 after the dataset doublings of 1.13.3). + pub cache_log2_words: u32, +} + +impl Shape { + /// Version 2: one mixer application per round, a 2^26-word cache. + pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32 }; + + /// The shape of a load class on day 0 of the chain (and on every day for a class without the growth rule). + pub fn for_class(class: &LoadClass) -> Shape { + Shape::for_class_day(class, 0) + } + + /// The shape of a load class on day `days_since_genesis` of the chain: the class's multiplier, and the cache + /// of [`cache_log2_words`] when the class has the growth rule, else 2^26 words. + pub fn for_class_day(class: &LoadClass, days_since_genesis: u64) -> Shape { + Shape { + mixer_mult: class.mixer_mult(), + cache_log2_words: if class.growth { cache_log2_words(days_since_genesis) } else { CACHE_LOG2_WORDS as u32 }, + } + } + + pub fn is_v2(&self) -> bool { + *self == Shape::V2 + } + pub fn cache_words(&self) -> usize { + 1usize << self.cache_log2_words + } + pub fn cache_lines(&self) -> usize { + self.cache_words() >> 4 + } + pub fn cache_line_mask(&self) -> u32 { + (self.cache_lines() - 1) as u32 + } + pub fn cache_segments(&self) -> usize { + self.cache_lines() >> CACHE_SEGMENT_LOG2_LINES + } + pub fn log2_segments(&self) -> u32 { + self.cache_log2_words - 4 - CACHE_SEGMENT_LOG2_LINES as u32 + } + /// Mixer applications per item: `(ITEM_ROUNDS + 1) x m`. + pub fn mixers_per_item(&self) -> u32 { + (ITEM_ROUNDS as u32 + 1) * self.mixer_mult + } +} + +// -------------------------------------------------------------------------------------------------------------- +// Dataset growth, option C (spec 01 section 1.13.3 option (b) with the cache tied to the dataset's doublings) +// -------------------------------------------------------------------------------------------------------------- + +/// Days per year of the growth schedule: one year = 31,536,000 DAA seconds of 86,400 (spec 01 section 1.13.3). +pub const GROWTH_DAYS_PER_YEAR: u64 = 365; +/// The linear schedule of 1.13.3, 2 GiB at genesis plus 0.5 GiB per year, is `G x (1 + d / 1460)` for the genesis +/// size `G` and the day `d`: it doubles at day 1,460 (year 4), quadruples at day 4,380 (year 12), reaches 8x at +/// day 10,220 (year 28) and 16x at day 21,900 (year 60). +pub const GROWTH_DOUBLING_DAYS: u64 = 4 * GROWTH_DAYS_PER_YEAR; + +/// The number of dataset doublings reached by day `days_since_genesis` of the chain: `floor(log2(1 + d / 1460))`, +/// in integers (`1 + d / 1460` rounded down, then its integer log2, which equals the real log2's floor because a +/// power of two is an integer). 0 until day 1,459; 1 from day 1,460 (year 4); 2 from day 4,380 (year 12). +pub fn growth_doublings(days_since_genesis: u64) -> u32 { + (1 + days_since_genesis / GROWTH_DOUBLING_DAYS).ilog2() +} + +/// The cache size on day `d` under option C: 2^26 words doubled once per dataset doubling (256 MiB, 512 MiB from +/// year 4, 1 GiB from year 12). +pub fn cache_log2_words(days_since_genesis: u64) -> u32 { + CACHE_LOG2_WORDS as u32 + growth_doublings(days_since_genesis) +} + +/// The dataset size on day `d` under option (b) of 1.13.3: the genesis size (2^`genesis_log2_words` words: 28 for +/// the 1 GiB packs and the devnet, 29 for the designed 2 GiB) doubled once per doubling of the linear schedule. The +/// result is capped at 32 (the item index is 32 bits, spec 1.13.3). +pub fn dataset_log2_words(genesis_log2_words: u32, days_since_genesis: u64) -> u32 { + (genesis_log2_words + growth_doublings(days_since_genesis)).min(32) +} + +/// Days since genesis from two day indices of `bind::day_index` (the header's `timestamp_ms / 86,400,000`): the +/// day of the block and the day of the genesis header. A block before the genesis day (clock skew) is day 0. +pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 { + day_index.saturating_sub(genesis_day_index) +} #[inline(always)] fn rotl(x: u32, n: u32) -> u32 { @@ -66,17 +168,23 @@ pub fn chacha_block(x: &[u32; 16]) -> [u32; 16] { y } -/// Mixer parameters drawn from the day key. Draw order: ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15]. +/// Mixer parameters drawn from the day key, plus the [`Shape`] the mixer is applied under. Draw order: +/// ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15]. The shape is not drawn: it is the class's. #[derive(Clone, Debug, PartialEq, Eq)] pub struct MixParams { pub key: [u32; 8], pub rot: [u32; 8], pub mul: [u32; 16], pub rc: [u32; 16], + pub shape: Shape, } impl MixParams { + /// Version 2 shape. pub fn new(key: [u32; 8]) -> Self { + Self::with_shape(key, Shape::V2) + } + pub fn with_shape(key: [u32; 8], shape: Shape) -> Self { let mut rng = SplitMix64::new(key[0] as u64 | ((key[1] as u64) << 32)); let mut rot = [0u32; 8]; let mut mul = [0u32; 16]; @@ -90,7 +198,7 @@ impl MixParams { for c in rc.iter_mut() { *c = rng.next() as u32; } - Self { key, rot, mul, rc } + Self { key, rot, mul, rc, shape } } /// Parameters for a day string: the key is `seed_words("day/" + day)`. pub fn for_day(day: &str) -> Self { @@ -98,12 +206,84 @@ impl MixParams { } } +/// The dataset layout (era layout, `docs/plans/era-layout.md` section 1.2): word `w` of the dataset holds word +/// `j(w)` of item `t(w)`, where `j(w)` gathers the four bits of `w` at the ascending positions `pos` and `t(w)` is +/// `w` with those bits removed. [`Layout::LINEAR`] (`pos = [0, 1, 2, 3]`) is `dataset[w] = item(w >> 4)[w & 15]`, +/// the lottery hash's mapping. Every position is below 16, so the mapping is the same at every dataset size of +/// at least 2^16 words and an item keeps its value at every size. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct Layout { + pub pos: [u8; 4], +} + +impl Layout { + pub const LINEAR: Layout = Layout { pos: [0, 1, 2, 3] }; + + pub fn is_linear(&self) -> bool { + self.pos == [0, 1, 2, 3] + } + + /// Positions ascending, distinct, below 16. + pub fn is_valid(&self) -> bool { + self.pos.iter().all(|&p| p < 16) && (1..4).all(|i| self.pos[i] > self.pos[i - 1]) + } + + /// `(t, j)` of word index `w`. + #[inline(always)] + pub fn split(&self, w: u32) -> (u32, u32) { + if self.is_linear() { + return (w >> 4, w & 15); + } + let mut j = 0u32; + for (i, &p) in self.pos.iter().enumerate() { + j |= ((w >> p) & 1) << i; + } + // remove the highest position first so the lower ones stay where they are + let mut t = w; + for &p in self.pos.iter().rev() { + let p = p as u32; + let low = (1u32 << p) - 1; + t = (t & low) | ((t >> (p + 1)) << p); + } + (t, j) + } + + /// The word index of word `j` of item `t`: the inverse of [`Layout::split`]. + #[inline(always)] + pub fn join(&self, t: u32, j: u32) -> u32 { + if self.is_linear() { + return (t << 4) | (j & 15); + } + // insert the lowest position first: every later position counts the bit just inserted + let mut w = t; + for (i, &p) in self.pos.iter().enumerate() { + let p = p as u32; + let low = (1u32 << p) - 1; + w = ((w >> p) << (p + 1)) | (w & low) | (((j >> i) & 1) << p); + } + w + } +} + +impl Default for Layout { + fn default() -> Self { + Layout::LINEAR + } +} + /// Round key `(r + 1) * 0x9E3779B9` mod 2^32. #[inline(always)] pub fn round_key(r: usize) -> u32 { ((r + 1) as u32).wrapping_mul(0x9E3779B9) } +/// The round key of application `j` (0 <= j < m) of round `r` under multiplier `m`: `round_key(r * m + j)`. For +/// `m = 1` this is `round_key(r)`, version 2's key. +#[inline(always)] +pub fn round_key_mult(r: usize, j: usize, m: usize) -> u32 { + round_key(r * m + j) +} + /// `M_r` on 16 words in place: per word `(s ^ (RC + rk)) * MUL`, then one ChaCha-shaped double round with /// the four column rotations `ROT[0..3]` and the four diagonal rotations `ROT[4..7]`. #[inline(always)] @@ -122,9 +302,11 @@ pub fn mixer(s: &mut [u32; 16], rk: u32, mp: &MixParams) { qr(s, 3, 4, 9, 14, r[4], r[5], r[6], r[7]); } -/// The 256 MiB cache for one day key. +/// The cache for one day key: 2^log2_words words (256 MiB under version 2). pub struct Cache { pub key: [u32; 8], + pub log2_words: u32, + line_mask: u32, words: Vec, } @@ -132,6 +314,12 @@ impl Cache { /// One segment: 64 chained lines written at `cache[seg * 1024 ..]`. /// `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = B(in_j)`, `prev_0 = 0`. pub fn fill_segment(words: &mut [u32], seg: usize, key: &[u32; 8]) { + Self::fill_segment_tagged(words, seg, key, &CACHE_TAG) + } + + /// [`Cache::fill_segment`] with an explicit chain tag: [`CACHE_TAG`] for the cache, [`HOT_TAG`] for the hot + /// table of `docs/plans/hot-table.md` (the same chain, another key and tag). + pub fn fill_segment_tagged(words: &mut [u32], seg: usize, key: &[u32; 8], tag: &[u32; 2]) { let base = (seg << CACHE_SEGMENT_LOG2_LINES) * 16; let seg_words = &mut words[base..base + CACHE_LINES_PER_SEGMENT * 16]; let mut prev = [0u32; 16]; @@ -141,8 +329,8 @@ impl Cache { x[4..12].copy_from_slice(key); x[12] = seg as u32; x[13] = j as u32; - x[14] = CACHE_TAG[0]; - x[15] = CACHE_TAG[1]; + x[14] = tag[0]; + x[15] = tag[1]; for i in 0..16 { x[i] ^= prev[i]; } @@ -152,13 +340,22 @@ impl Cache { } } - /// The whole cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order. + /// The version 2 cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order. pub fn fill(key: [u32; 8]) -> Cache { - let mut words = vec![0u32; CACHE_WORDS]; - for seg in 0..CACHE_SEGMENTS { + Self::fill_log2(key, CACHE_LOG2_WORDS as u32) + } + + /// A cache of 2^`log2_words` words (26, 27 or 28 under the growth rule; smaller sizes for tests): 2^(log2 - 10) + /// independent chains of 64 lines, the same chain function at every size, so a larger cache's first segments + /// are the smaller cache's segments word for word. + pub fn fill_log2(key: [u32; 8], log2_words: u32) -> Cache { + assert!((10..=30).contains(&log2_words), "cache log2 words must be in 10..=30"); + let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words }; + let mut words = vec![0u32; shape.cache_words()]; + for seg in 0..shape.cache_segments() { Self::fill_segment(&mut words, seg, &key); } - Cache { key, words } + Cache { key, log2_words, line_mask: shape.cache_line_mask(), words } } pub fn for_day(day: &str) -> Cache { @@ -170,10 +367,27 @@ impl Cache { &self.words } - /// Cache line `a` (0 <= a < 2^22) as 16 words. + pub fn lines(&self) -> usize { + self.words.len() >> 4 + } + pub fn line_mask(&self) -> u32 { + self.line_mask + } + pub fn segments(&self) -> usize { + self.lines() >> CACHE_SEGMENT_LOG2_LINES + } + + /// Cache line `a` (masked to the cache's lines) as 16 words. #[inline(always)] pub fn line(&self, a: u32) -> &[u32] { - let o = (a & CACHE_LINE_MASK) as usize * 16; + let o = (a & self.line_mask) as usize * 16; + &self.words[o..o + 16] + } + + /// [`Cache::line`] with the mask as a constant (the verifier's hot path, see [`derive_items`]). + #[inline(always)] + pub fn line_const(&self, a: u32) -> &[u32] { + let o = (a & LINE_MASK) as usize * 16; &self.words[o..o + 16] } @@ -183,11 +397,114 @@ impl Cache { } } +/// The hot key of an epoch: `seed_words_from_bytes("igneum-hot/" || seed_bytes)`, `seed_bytes` the program seed +/// bytes before any attempt suffix, so every attempt of one epoch shares one table. +pub fn hot_key(seed_bytes: &[u8]) -> [u32; 8] { + let mut b = Vec::with_capacity(HOT_KEY_TAG.len() + seed_bytes.len()); + b.extend_from_slice(HOT_KEY_TAG); + b.extend_from_slice(seed_bytes); + crate::seed::seed_words_from_bytes(&b) +} + +/// Words of a hot table of `mb` MiB. +pub fn hot_words(mb: u32) -> u32 { + mb * HOT_WORDS_PER_MIB +} + +/// Segments of a hot table of `mb` MiB. +pub fn hot_segments(mb: u32) -> u32 { + mb * HOT_SEGMENTS_PER_MIB +} + +/// The hot index of a source word: `mulhi(src, words)`, the high 32 bits of the 64-bit product, in `[0, words)` +/// for any table size (the multiply-shift range reduction of spec 01 section 1.13.3). +#[inline(always)] +pub fn hot_index(src: u32, words: u32) -> u32 { + ((src as u64 * words as u64) >> 32) as u32 +} + +/// The hot table `H` of one epoch (hot-table experiment): `mb` MiB of chained ChaCha12 lines under the hot key, +/// read by the hot load slots as `dst ^= H[hot_index(src, words)]`. The verifier holds it beside the cache. +pub struct HotTable { + pub key: [u32; 8], + pub mb: u32, + words: Vec, +} + +impl HotTable { + /// Fill `mb` MiB under `key` on the calling thread. + pub fn fill(key: [u32; 8], mb: u32) -> HotTable { + assert!(mb >= 1 && mb <= 4096, "hot table size in MiB out of range"); + let n = hot_words(mb) as usize; + let mut words = vec![0u32; n]; + for seg in 0..hot_segments(mb) as usize { + Cache::fill_segment_tagged(&mut words, seg, &key, &HOT_TAG); + } + HotTable { key, mb, words } + } + + /// The table of the epoch whose program seed bytes are `seed_bytes`. + pub fn for_seed_bytes(seed_bytes: &[u8], mb: u32) -> HotTable { + Self::fill(hot_key(seed_bytes), mb) + } + + #[inline(always)] + pub fn n_words(&self) -> u32 { + self.words.len() as u32 + } + + /// `H[i]`. + #[inline(always)] + pub fn at(&self, i: u32) -> u32 { + self.words[i as usize] + } + + /// `H[hot_index(src, words)]`: what a hot load reads for source word `src`. + #[inline(always)] + pub fn word(&self, src: u32) -> u32 { + self.words[hot_index(src, self.n_words()) as usize] + } + + #[inline(always)] + pub fn words(&self) -> &[u32] { + &self.words + } + + /// FNV-1a 64 over the table as little-endian bytes (what `vectors.h` carries as `IGNEUM_HOT_FNV64`). + pub fn fnv1a64(&self) -> u64 { + fnv1a64_words(&self.words) + } +} + /// Derive `ts.len()` items into `out`, all chains interleaved round by round so the cache-line misses of -/// independent items overlap in the memory system (`deriveItems` in the Swift). +/// independent items overlap in the memory system (`deriveItems` in the Swift). Under multiplier `m` +/// (`mp.shape.mixer_mult`) round `r` applies `M` with keys `round_key(r m + j)` for `j = 0 .. m - 1` before its +/// one cache read; the final mixer applies `M` with keys `round_key(8 m + j)`. `m = 1` is version 2. pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) { + // The item loop lives in its own function, one instance per cache size the growth rule can reach with the line + // mask a constant, never inlined into the callers. Inlined into `MemhardCpu::fetch` it ran at 1.33 ms per unit + // against 0.61 out of line (the version 2 verifier, bisected on one core under the measure lock, 5 October 2026, + // `docs/plans/mixer-x4.md` section 6.6: the constant mask alone, or the mask hoisted into a local, or the + // constant with the loop still inlined, all stayed at 1.33; the out-of-line instances read 0.60 to 0.62). Any + // other cache size (tests) takes the instance with the run-time mask. + match cache.log2_words { + 26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, out), + 27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, out), + 28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, out), + 29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, out), + 30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, out), + _ => derive_items_mask::<0>(ts, mp, cache, out), + } +} + +/// [`derive_items`] with the cache line mask as a constant (`LINE_MASK = 0`: the cache's own run-time mask). Kept +/// out of line on purpose (see [`derive_items`]). +#[inline(never)] +fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) { let n = ts.len(); debug_assert!(out.len() >= n); + debug_assert!(LINE_MASK == 0 || LINE_MASK == cache.line_mask); + let m = mp.shape.mixer_mult as usize; for k in 0..n { let s = &mut out[k]; let t = ts[k]; @@ -197,20 +514,24 @@ pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; } } for r in 0..ITEM_ROUNDS { - let rk = round_key(r); - for s in out[..n].iter_mut() { - mixer(s, rk, mp); + for j in 0..m { + let rk = round_key_mult(r, j, m); + for s in out[..n].iter_mut() { + mixer(s, rk, mp); + } } for s in out[..n].iter_mut() { - let line = cache.line(s[0]); + let line = if LINE_MASK != 0 { cache.line_const::(s[0]) } else { cache.line(s[0]) }; for i in 0..16 { s[i] ^= line[i]; } } } - let rk = round_key(ITEM_ROUNDS); - for s in out[..n].iter_mut() { - mixer(s, rk, mp); + for j in 0..m { + let rk = round_key_mult(ITEM_ROUNDS, j, m); + for s in out[..n].iter_mut() { + mixer(s, rk, mp); + } } } @@ -221,7 +542,7 @@ pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] { out[0] } -/// The CPU verifier's view of the memory-hard dataset: the mixer parameters and the 256 MiB cache. +/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape) and the cache. pub struct MemhardCpu { pub params: MixParams, pub cache: Cache, @@ -231,26 +552,41 @@ pub struct MemhardCpu { pub const FETCH_MAX: usize = 64; impl MemhardCpu { + /// Version 2 shape. pub fn new(key: [u32; 8]) -> Self { - Self { params: MixParams::new(key), cache: Cache::fill(key) } + Self::with_shape(key, Shape::V2) + } + pub fn with_shape(key: [u32; 8], shape: Shape) -> Self { + Self { params: MixParams::with_shape(key, shape), cache: Cache::fill_log2(key, shape.cache_log2_words) } } pub fn for_day(day: &str) -> Self { Self::new(day_key(day)) } - /// `dataset[w] = item(w >> 4)[w & 15]`. + pub fn shape(&self) -> Shape { + self.params.shape + } + /// `dataset[w] = item(w >> 4)[w & 15]` (the linear layout). pub fn word(&self, w: u32) -> u32 { - derive_item(w >> 4, &self.params, &self.cache)[(w & 15) as usize] + self.word_at(Layout::LINEAR, w) + } + /// `dataset[w] = item(t(w))[j(w)]` under `layout` (era layout; the layout is the program's, the cache the + /// day's, so one cache serves every era of a day). + pub fn word_at(&self, layout: Layout, w: u32) -> u32 { + let (t, j) = layout.split(w); + derive_item(t, &self.params, &self.cache)[j as usize] } /// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once. /// Returns the number of distinct items derived. - pub fn fetch(&self, idx: &[u32], out: &mut [u32]) -> usize { + pub fn fetch(&self, idx: &[u32], out: &mut [u32], layout: Layout) -> usize { let n = idx.len(); assert!(n <= FETCH_MAX && out.len() >= n); let mut uniq = [0u32; FETCH_MAX]; let mut slot = [0u8; FETCH_MAX]; + let mut word = [0u8; FETCH_MAX]; let mut u = 0usize; for k in 0..n { - let t = idx[k] >> 4; + let (t, j) = layout.split(idx[k]); + word[k] = j as u8; let found = uniq[..u].iter().position(|&x| x == t); let j = match found { Some(j) => j, @@ -265,7 +601,39 @@ impl MemhardCpu { let mut items = [[0u32; 16]; FETCH_MAX]; derive_items(&uniq[..u], &self.params, &self.cache, &mut items); for k in 0..n { - out[k] = items[slot[k] as usize][(idx[k] & 15) as usize]; + out[k] = items[slot[k] as usize][word[k] as usize]; + } + u + } + /// `out[k][j] = dataset[base[k] + j]` for `j < width` (read-width experiment): `base[k]` is aligned to `width` + /// words and the layout's low `log2(width)` positions are the identity, so every lane's words lie in one item + /// at consecutive word offsets, derived once per distinct item. Returns the distinct items. + pub fn fetch_wide(&self, base: &[u32], width: usize, out: &mut [[u32; 16]], layout: Layout) -> usize { + let n = base.len(); + assert!(n <= FETCH_MAX && out.len() >= n && width <= 16); + debug_assert!((0..width.trailing_zeros() as usize).all(|i| layout.pos[i] == i as u8), "a wide load needs the identity on its low positions"); + let mut uniq = [0u32; FETCH_MAX]; + let mut slot = [0u8; FETCH_MAX]; + let mut word = [0u8; FETCH_MAX]; + let mut u = 0usize; + for k in 0..n { + let (t, j0) = layout.split(base[k]); + word[k] = j0 as u8; + let j = match uniq[..u].iter().position(|&x| x == t) { + Some(j) => j, + None => { + uniq[u] = t; + u += 1; + u - 1 + } + }; + slot[k] = j as u8; + } + let mut items = [[0u32; 16]; FETCH_MAX]; + derive_items(&uniq[..u], &self.params, &self.cache, &mut items); + for k in 0..n { + let o = word[k] as usize; + out[k][..width].copy_from_slice(&items[slot[k] as usize][o..o + width]); } u } @@ -285,6 +653,40 @@ mod tests { assert_eq!(mp.rc[0], 0xbab68293); assert_eq!(mp.rc[15], 0x31b49ee2); assert!(mp.mul.iter().all(|m| m & 1 == 1)); + assert_eq!(mp.shape, Shape::V2); + } + + /// Era layout: split and join are inverse, the linear layout is today's mapping, and an interleaved layout + /// keeps every position below 16 so the mapping is the same at every size of at least 2^16 words. + #[test] + fn layout_split_join() { + let lin = Layout::LINEAR; + assert!(lin.is_linear() && lin.is_valid()); + for w in [0u32, 1, 15, 16, 17, 0x0fff_ffff, 0xffff_ffff] { + assert_eq!(lin.split(w), (w >> 4, w & 15)); + assert_eq!(lin.join(w >> 4, w & 15), w); + } + let l = Layout { pos: [0, 1, 7, 12] }; + assert!(!l.is_linear() && l.is_valid()); + for w in [0u32, 1, 2, 3, 4, 127, 128, 129, 4095, 4096, 0x0fff_ffff, 0x1234_5678, 0xffff_ffff] { + let (t, j) = l.split(w); + assert!(j < 16); + assert_eq!(l.join(t, j), w, "w {w:#x}"); + } + // bits: j0 = bit 0, j1 = bit 1, j2 = bit 7, j3 = bit 12; t = the other 28 bits in order + assert_eq!(l.split(0b1_0000_0000_0000), (0, 8)); + assert_eq!(l.split(1 << 7), (0, 4)); + assert_eq!(l.split(0b100), (1, 0)); + // every t in 0..2^(D-4) appears exactly once among w < 2^D (D = 16), with every j + let mut seen = vec![0u32; 1 << 12]; + for w in 0..(1u32 << 16) { + let (t, j) = l.split(w); + seen[t as usize] |= 1 << j; + } + assert!(seen.iter().all(|&s| s == 0xffff)); + assert!(!Layout { pos: [0, 1, 1, 5] }.is_valid()); + assert!(!Layout { pos: [0, 1, 2, 16] }.is_valid()); + assert!(!Layout { pos: [1, 0, 2, 3] }.is_valid()); } #[test] @@ -296,6 +698,40 @@ mod tests { assert_eq!(y, z); } + /// Hot-table experiment: the genesis epoch's table (seed bytes "igneum-genesis") as the hot packs carry it + /// (`proto-cuda/packs-ca2-hot/hot32k4/vectors.json`: hot_head, hot_fnv1a64; the head is the same at every size, + /// a larger table is more segments). The index mapping stays inside the table for any size. + #[test] + fn hot_table_fill_vector_and_index() { + assert_ne!(HOT_TAG, CACHE_TAG); + let h = HotTable::for_seed_bytes(b"igneum-genesis", 32); + assert_eq!(h.n_words(), 1 << 23); + assert_eq!(hot_segments(32), 8192); + assert_eq!( + &h.words()[..16], + &[ + 0x8068cc73, 0x6036ebf9, 0xb604cd25, 0x8ffb840e, 0xc54074a2, 0x285c0695, 0x77512425, 0xc26a58a7, + 0x72c88757, 0xc10fca78, 0x513825dd, 0x30d6ccc8, 0x9a05e7cf, 0xb9533f50, 0x4bac3ba0, 0xa5c19528 + ] + ); + assert_eq!(h.fnv1a64(), 0xc1767ba3ef02719f, "hot32k4 pack, hot_fnv1a64"); + assert_eq!(h.key, hot_key(b"igneum-genesis")); + assert_ne!(h.key, day_key("2026-10-03")); + // a different seed, a different table; the same seed under the cache tag is not the hot table + assert_ne!(HotTable::for_seed_bytes(b"igneum-genesis\x01\x00\x00\x00", 1).words()[..16], h.words()[..16]); + let mut under_cache_tag = vec![0u32; 1024]; + Cache::fill_segment(&mut under_cache_tag, 0, &h.key); + assert_ne!(&under_cache_tag[..16], &h.words()[..16]); + for words in [hot_words(32), hot_words(64), hot_words(96)] { + assert_eq!(hot_index(0, words), 0); + assert!(hot_index(u32::MAX, words) < words); + assert_eq!(hot_index(u32::MAX, words), words - 1); + assert!(hot_index(0x8000_0000, words) == words / 2); + } + assert_eq!(hot_index(0x1234_5678, 1 << 24), 0x1234_5678 >> 8); + assert_eq!(h.word(0x8000_0000), h.at(1 << 22)); + } + #[test] fn first_cache_line_matches_pack() { // vectors.json cache_head for day 2026-10-03: segment 0, line 0, with prev = 0. @@ -310,4 +746,96 @@ mod tests { ] ); } + + /// Option C: the schedule table of `docs/plans/mixer-x4.md` (day -> doublings, cache words, dataset words at a + /// 2^28 genesis). The doublings fall at years 4 and 12 exactly, never a day early. + #[test] + fn growth_schedule_table() { + let table: [(u64, u32, u32, u32); 12] = [ + (0, 0, 26, 28), + (1, 0, 26, 28), + (365, 0, 26, 28), + (1_459, 0, 26, 28), + (1_460, 1, 27, 29), + (2_920, 1, 27, 29), + (4_379, 1, 27, 29), + (4_380, 2, 28, 30), + (10_219, 2, 28, 30), + (10_220, 3, 29, 31), + (21_900, 4, 30, 32), + (100_000, 6, 32, 32), + ]; + for (d, k, c, s) in table { + assert_eq!(growth_doublings(d), k, "day {d}"); + assert_eq!(cache_log2_words(d), c, "day {d}"); + assert_eq!(dataset_log2_words(28, d), s, "day {d}"); + } + // the designed 2 GiB genesis: 2^29 words, 2^30 at year 4, 2^31 at year 12 + assert_eq!(dataset_log2_words(29, 0), 29); + assert_eq!(dataset_log2_words(29, 1_460), 30); + assert_eq!(dataset_log2_words(29, 4_380), 31); + // the linear schedule itself: 2 GiB x (1 + d / 1460) crosses 4 GiB at day 1,460 and 8 GiB at day 4,380 + for d in [1_459u64, 1_460, 4_379, 4_380] { + let bytes = 2u64 * (1 << 30) + (1u64 << 29) * d / 365; + let k = (bytes / (2u64 << 30)).ilog2(); + assert_eq!(growth_doublings(d), k, "day {d}: linear {bytes} bytes"); + } + assert_eq!(days_since_genesis(20_730, 20_729), 1); + assert_eq!(days_since_genesis(20_729, 20_729), 0); + assert_eq!(days_since_genesis(20_000, 20_729), 0); + let v2 = Shape::for_class_day(&LoadClass::V2, 100_000); + assert_eq!(v2, Shape::V2); + let v3 = Shape::for_class_day(&LoadClass::MX4, 0); + assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26 }); + assert_eq!(Shape::for_class_day(&LoadClass::MX4, 1_460).cache_log2_words, 27); + assert_eq!(v3.mixers_per_item(), 36); + assert_eq!(Shape::V2.mixers_per_item(), 9); + assert_eq!(Shape::V2.cache_segments(), CACHE_SEGMENTS); + assert_eq!(Shape::V2.cache_line_mask(), CACHE_LINE_MASK); + assert_eq!(Shape::V2.log2_segments(), 16); + } + + /// The multiplied mixer, restated by hand on a small cache: `m` applications with keys `round_key(r m + j)` + /// before every read, the same 8 reads; `m = 1` is `derive_item` of version 2 word for word; a larger cache's + /// first segments equal the smaller cache's. + #[test] + fn mixer_mult_by_hand() { + let key = day_key("2026-10-03"); + let small = Cache::fill_log2(key, 16); + let big = Cache::fill_log2(key, 18); + assert_eq!(&big.words()[..small.words().len()], small.words()); + assert_eq!(small.segments(), 64); + assert_eq!(small.line_mask(), 4095); + for m in [1u32, 2, 4] { + let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16 }); + for t in [0u32, 1, 12_345, u32::MAX] { + let got = derive_item(t, &mp, &small); + let mut s = [0u32; 16]; + s[..8].copy_from_slice(&key); + for i in 0..8 { + s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); + } + for r in 0..8usize { + for j in 0..m as usize { + mixer(&mut s, round_key(r * m as usize + j), &mp); + } + let line = small.line(s[0]); + for i in 0..16 { + s[i] ^= line[i]; + } + } + for j in 0..m as usize { + mixer(&mut s, round_key(8 * m as usize + j), &mp); + } + assert_eq!(got, s, "m {m} t {t}"); + } + } + let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16 }); + let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16 }); + assert_ne!(derive_item(0, &v2, &small), derive_item(0, &v3, &small)); + assert_eq!(round_key_mult(0, 0, 1), round_key(0)); + assert_eq!(round_key_mult(8, 0, 1), round_key(8)); + assert_eq!(round_key_mult(2, 3, 4), round_key(11)); + assert_eq!(round_key_mult(8, 3, 4), round_key(35)); + } } diff --git a/igneum-pow/src/packcheck.rs b/igneum-pow/src/packcheck.rs new file mode 100644 index 000000000..5bd4240c2 --- /dev/null +++ b/igneum-pow/src/packcheck.rs @@ -0,0 +1,397 @@ +//! A program pack on disk, read the way the one-click workers read it (`proto-cuda/nvrtc/packfile.h`, `pf_load`), +//! and checked against the seeds the node is on. +//! +//! The rule (5 October 2026, the epoch 34 incident on both PCs): a pack's `IGNEUM_SEEDW_INIT` is the seed words of +//! the program's ATTEMPT, `attempt_words(epoch_seed, IGNEUM_PROGRAM_ATTEMPT)`, not the words of the bare seed. The +//! generator retries a rejected candidate with `seed || k_le32` (spec 01 section 1.4.6), so from attempt 1 on the +//! bare-seed words and the pack's words differ. The workers derived the expected words from the bare seed, refused +//! every pack of a retried program ("the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT") and the miner +//! and the app restarted them forever. Epoch 34 (seed `009858237e11...`) was the first live epoch whose program is +//! a later attempt. This module is the one place that rule is written in Rust; the miner checks every pack it +//! writes with it before a worker sees the pack, and the tests pin the attempt vectors the C side also pins. + +use crate::generator::{attempt_words, ProgramClass}; +use crate::seed::seed_words_from_bytes; +use std::fmt; +use std::path::Path; + +/// What a pack says about itself. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PackIdentity { + pub epoch_hex: String, + pub day_hex: String, + pub attempt: u32, + pub seedw: [u32; 8], + pub keyw: [u32; 8], + /// `IGNEUM_GENERATOR` (2 or 3; a pack without the line is generator 1, which no worker runs). + pub generator: u32, + /// The program class the generator version names (Counter ASIC 2.0). + pub class: ProgramClass, + /// `IGNEUM_ERA_SEED_HEX` when the pack carries one (class v3 chain packs). + pub era_hex: Option, +} + +/// Why a pack is not the one a worker should mine with. `Display` is the plain-words line the logs carry. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PackFault { + /// program.h or seeds.txt is missing or does not parse. + Unreadable(String), + /// A well-formed pack for other seeds than the node's: the pack is stale (or the node moved on). + OutOfDate { pack_epoch: String, pack_day: String, want_epoch: String, want_day: String }, + /// The files of one pack contradict each other (seeds.txt against program.h, or the init words against the + /// seeds and the attempt): a half-written or hand-edited pack, or a worker and an exporter on different rules. + Disagree(String), + /// The pack is of another program class than the one the chain is on (spec 01 section 1.4.5: an implementation + /// refuses a pack whose generator version is not its own), or its era seed is not the era the job names. + WrongClass(String), +} + +impl fmt::Display for PackFault { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + PackFault::Unreadable(w) => write!(f, "program pack unreadable: {w}"), + PackFault::OutOfDate { pack_epoch, pack_day, want_epoch, want_day } => write!( + f, + "program pack out of date: the pack is for epoch {} day {}, the node is on epoch {} day {}", + short(pack_epoch), + day_label(pack_day), + short(want_epoch), + day_label(want_day) + ), + PackFault::Disagree(w) => write!(f, "program pack and its seeds disagree: {w}"), + PackFault::WrongClass(w) => write!(f, "program pack of the wrong class: {w}"), + } + } +} + +impl std::error::Error for PackFault {} + +fn short(hex: &str) -> &str { + if hex.len() >= 16 { + &hex[..16] + } else { + hex + } +} + +/// The day bytes are `igneum-day/` followed by the little-endian day index (`bind::day_bytes`); print the index +/// when the hex has that shape, else the hex. +fn day_label(hex: &str) -> String { + const PREFIX: &str = "69676e65756d2d6461792f"; // "igneum-day/" + if let Some(rest) = hex.strip_prefix(PREFIX) { + if let Some(bytes) = unhex(rest) { + let mut v = 0u64; + for (i, b) in bytes.iter().enumerate().take(8) { + v |= (*b as u64) << (8 * i); + } + return v.to_string(); + } + } + hex.to_string() +} + +pub fn hex(bytes: &[u8]) -> String { + bytes.iter().map(|b| format!("{b:02x}")).collect() +} + +fn unhex(s: &str) -> Option> { + if s.len() % 2 != 0 { + return None; + } + (0..s.len()).step_by(2).map(|i| u8::from_str_radix(&s[i..i + 2], 16).ok()).collect() +} + +/// `#define NAME ` in a header; the value as text, trimmed, with a trailing `//` comment removed. +fn define(text: &str, name: &str) -> Option { + for line in text.lines() { + let t = line.trim_start(); + let Some(rest) = t.strip_prefix("#define ") else { continue }; + let rest = rest.trim_start(); + let Some(after) = rest.strip_prefix(name) else { continue }; + if !after.starts_with(|c: char| c.is_whitespace()) { + continue; + } + let v = after.trim(); + let v = v.split("//").next().unwrap_or("").trim(); + return Some(v.to_string()); + } + None +} + +fn define_str(text: &str, name: &str) -> Option { + let v = define(text, name)?; + let v = v.strip_prefix('"')?.strip_suffix('"')?; + Some(v.to_string()) +} + +fn define_u32(text: &str, name: &str) -> Option { + let v = define(text, name)?; + let v = v.trim_end_matches('u'); + if let Some(h) = v.strip_prefix("0x") { + u32::from_str_radix(h, 16).ok() + } else { + v.parse().ok() + } +} + +fn define_words(text: &str, name: &str) -> Option<[u32; 8]> { + let v = define(text, name)?; + let inner = v.trim().strip_prefix('{')?.strip_suffix('}')?; + let mut out = [0u32; 8]; + let mut n = 0; + for part in inner.split(',') { + let p = part.trim().trim_end_matches('u'); + if p.is_empty() { + continue; + } + if n >= 8 { + return None; + } + out[n] = if let Some(h) = p.strip_prefix("0x") { u32::from_str_radix(h, 16).ok()? } else { p.parse().ok()? }; + n += 1; + } + (n == 8).then_some(out) +} + +/// One `key value` line of seeds.txt. +fn seeds_line(text: &str, key: &str) -> Option { + text.lines().find_map(|l| l.strip_prefix(key).and_then(|r| r.strip_prefix(' ')).map(|v| v.trim().to_string())) +} + +/// Checks the texts of a pack (program.h, and seeds.txt when it exists) against the seeds a worker will be asked +/// to mine with. Pure: the miner and the tests call it with file contents. +pub fn verify_pack_texts(program_h: &str, seeds_txt: Option<&str>, want_epoch: &[u8], want_day: &[u8]) -> Result { + verify_pack_texts_chain(program_h, seeds_txt, want_epoch, want_day, None, None) +} + +/// [`verify_pack_texts`] that also demands a program class and, for class v3, the era seed the chain is on +/// (Counter ASIC 2.0, 5 October 2026). `want_class` `None` accepts either class; `want_era` `None` skips the era. +/// A pack whose `IGNEUM_GENERATOR` is neither 2 nor 3 is refused whatever is wanted. +pub fn verify_pack_texts_chain( + program_h: &str, + seeds_txt: Option<&str>, + want_epoch: &[u8], + want_day: &[u8], + want_class: Option, + want_era: Option<&[u8]>, +) -> Result { + let generator = define_u32(program_h, "IGNEUM_GENERATOR").unwrap_or(1); + let Some(class) = ProgramClass::from_generator(generator) else { + return Err(PackFault::WrongClass(format!("IGNEUM_GENERATOR {generator} is not a generator version this software runs (2 or 3)"))); + }; + // IGNEUM_PROGRAM_CLASS, when present, must name the class the generator version names + if let Some(named) = define_str(program_h, "IGNEUM_PROGRAM_CLASS") { + if ProgramClass::parse(&named) != Some(class) { + return Err(PackFault::Disagree(format!("IGNEUM_PROGRAM_CLASS {named:?} does not match IGNEUM_GENERATOR {generator}"))); + } + } + let era_hex = define_str(program_h, "IGNEUM_ERA_SEED_HEX").map(|h| h.to_ascii_lowercase()); + if let Some(want) = want_class { + if want != class { + return Err(PackFault::WrongClass(format!("the pack is program class {} (generator {generator}), the chain is on class {}", class.name(), want.name()))); + } + } + if let (Some(want), ProgramClass::V3) = (want_era, class) { + let want_hex = hex(want); + match &era_hex { + Some(h) if *h == want_hex => {} + Some(h) => return Err(PackFault::WrongClass(format!("the pack's era seed {} is not the era seed {} the job names", short(h), short(&want_hex)))), + None => return Err(PackFault::WrongClass("a class v3 pack without IGNEUM_ERA_SEED_HEX; the job names an era seed".into())), + } + } + let seedw = define_words(program_h, "IGNEUM_SEEDW_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_SEEDW_INIT with 8 words".into()))?; + let keyw = define_words(program_h, "IGNEUM_KEY_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_KEY_INIT with 8 words".into()))?; + let attempt = define_u32(program_h, "IGNEUM_PROGRAM_ATTEMPT").unwrap_or(0); + let mut epoch_hex = define_str(program_h, "IGNEUM_SEED_BYTES_HEX").unwrap_or_default(); + let mut day_hex = define_str(program_h, "IGNEUM_DAY_BYTES_HEX").unwrap_or_default(); + if let Some(s) = seeds_txt { + let e = seeds_line(s, "epoch_seed_hex").ok_or_else(|| PackFault::Unreadable("seeds.txt has no epoch_seed_hex line".into()))?; + let d = seeds_line(s, "day_seed_hex").ok_or_else(|| PackFault::Unreadable("seeds.txt has no day_seed_hex line".into()))?; + if !epoch_hex.is_empty() && !epoch_hex.eq_ignore_ascii_case(&e) { + return Err(PackFault::Disagree(format!("seeds.txt names epoch {} but program.h was generated for epoch {} (a pack half rewritten?)", short(&e), short(&epoch_hex)))); + } + if !day_hex.is_empty() && !day_hex.eq_ignore_ascii_case(&d) { + return Err(PackFault::Disagree(format!("seeds.txt names day {} but program.h was generated for day {}", day_label(&d), day_label(&day_hex)))); + } + epoch_hex = e.to_ascii_lowercase(); + day_hex = d.to_ascii_lowercase(); + } + if epoch_hex.is_empty() || day_hex.is_empty() { + return Err(PackFault::Unreadable("no seeds: neither seeds.txt nor IGNEUM_SEED_BYTES_HEX / IGNEUM_DAY_BYTES_HEX in program.h".into())); + } + let epoch_bytes = unhex(&epoch_hex).filter(|b| b.len() == 32).ok_or_else(|| PackFault::Unreadable("the epoch seed is not 32 bytes of hex".into()))?; + let day_bytes = unhex(&day_hex).ok_or_else(|| PackFault::Unreadable("the day seed hex is malformed".into()))?; + // The pack's own consistency first: a pack that contradicts itself is never "out of date", it is broken + let want_w = attempt_words(&epoch_bytes, attempt); + if want_w != seedw { + return Err(PackFault::Disagree(format!( + "IGNEUM_SEEDW_INIT is not attempt {attempt} of the epoch seed {} (the words of attempt {attempt} are {:08x} {:08x} ..., the pack has {:08x} {:08x} ...)", + short(&epoch_hex), + want_w[0], + want_w[1], + seedw[0], + seedw[1] + ))); + } + let want_k = seed_words_from_bytes(&day_bytes); + if want_k != keyw { + return Err(PackFault::Disagree(format!("IGNEUM_KEY_INIT is not the key of the day seed {} ", day_label(&day_hex)))); + } + let want_epoch_hex = hex(want_epoch); + let want_day_hex = hex(want_day); + if epoch_hex != want_epoch_hex || day_hex != want_day_hex { + return Err(PackFault::OutOfDate { pack_epoch: epoch_hex, pack_day: day_hex, want_epoch: want_epoch_hex, want_day: want_day_hex }); + } + Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex }) +} + +/// [`verify_pack_texts`] over a pack directory. +pub fn verify_pack_dir(dir: &Path, want_epoch: &[u8], want_day: &[u8]) -> Result { + verify_pack_dir_chain(dir, want_epoch, want_day, None, None) +} + +/// [`verify_pack_texts_chain`] over a pack directory. +pub fn verify_pack_dir_chain(dir: &Path, want_epoch: &[u8], want_day: &[u8], want_class: Option, want_era: Option<&[u8]>) -> Result { + let program_h = std::fs::read_to_string(dir.join("program.h")).map_err(|e| PackFault::Unreadable(format!("cannot read {}/program.h: {e}", dir.display())))?; + let seeds = std::fs::read_to_string(dir.join("seeds.txt")).ok(); + verify_pack_texts_chain(&program_h, seeds.as_deref(), want_epoch, want_day, want_class, want_era) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::emit::program_header; + use crate::generator::generate_from_seed_bytes; + use crate::verify::Epoch; + + // The two live devnet epochs of 5 October 2026 (epoch 33 mined, epoch 34 refused by the one-click workers) + const EPOCH_33: &str = "bed7ab62cbece66cf791485336d81d90fa1452ffed28ecd8a7416960ef64164c"; + const EPOCH_34: &str = "009858237e118f69abc8d096e9b1af21c24539eaecdfd1b896588825660a69ec"; + // "igneum-day/" || le64(20731) + const DAY_20731: &str = "69676e65756d2d6461792ffb50000000000000"; + + fn bytes(h: &str) -> Vec { + unhex(h).unwrap() + } + + /// The attempt vectors the C side pins too (proto-cuda/nvrtc/emu/packfile-test.c): a change to either + /// derivation fails on one side first. + #[test] + fn attempt_words_vectors_shared_with_the_workers() { + let e = bytes(EPOCH_34); + assert_eq!(attempt_words(&e, 0), [0x06af2a61, 0x4d67274e, 0x4ebda738, 0xad1dea73, 0x6233cd8c, 0x50371601, 0x39d0b873, 0x6af024a2]); + assert_eq!(attempt_words(&e, 1), [0x0dcff56b, 0x6b1beb0d, 0x234dc70c, 0xe4016fa9, 0x72397152, 0xb558aa79, 0x3ffb3299, 0x72b9962e]); + // epoch 33's bare words, as the CUDA worker printed them on 5 October ("seed words be5983a6 f750dab7 ...") + assert_eq!(attempt_words(&bytes(EPOCH_33), 0)[..2], [0xbe5983a6, 0xf750dab7]); + } + + /// The incident: epoch 34's program is a later attempt, epoch 33's is the bare seed. A worker that derives the + /// words from the bare seed accepts 33 and refuses 34. + #[test] + fn epoch_34_program_is_a_later_attempt() { + let p34 = generate_from_seed_bytes("epoch 34", &bytes(EPOCH_34)); + assert!(p34.attempt >= 1, "epoch 34 must be a retried program for the incident to reproduce; attempt {}", p34.attempt); + assert_eq!(p34.seed, attempt_words(&bytes(EPOCH_34), p34.attempt)); + assert_ne!(p34.seed, attempt_words(&bytes(EPOCH_34), 0)); + let p33 = generate_from_seed_bytes("epoch 33", &bytes(EPOCH_33)); + assert_eq!(p33.attempt, 0); + } + + fn pack_texts(epoch_hex: &str, day_hex: &str) -> (String, String, u32) { + let (e, d) = (bytes(epoch_hex), bytes(day_hex)); + let epoch = Epoch::from_seed_bytes(&e, &d, "test"); + let h = program_header(&epoch.program, "test day", &epoch.dataset); + let s = format!("epoch_seed_hex {epoch_hex}\nday_seed_hex {day_hex}\nday_index 20731\n"); + (h, s, epoch.program.attempt) + } + + /// Known-good: the pack of a retried program verifies against its own seeds, with its attempt. + #[test] + fn known_good_pack_of_a_later_attempt_verifies() { + let (h, s, attempt) = pack_texts(EPOCH_34, DAY_20731); + assert!(attempt >= 1); + let id = verify_pack_texts(&h, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).expect("the pack verifies"); + assert_eq!(id.attempt, attempt); + assert_eq!(id.epoch_hex, EPOCH_34); + assert_eq!(id.seedw, attempt_words(&bytes(EPOCH_34), attempt)); + // without seeds.txt program.h's own bytes carry the pack + assert!(verify_pack_texts(&h, None, &bytes(EPOCH_34), &bytes(DAY_20731)).is_ok()); + } + + /// Known-mismatched: a well-formed pack for the previous epoch is "out of date" against the new one, in plain + /// words with both epochs named. + #[test] + fn known_mismatched_pack_is_out_of_date() { + let (h, s, _) = pack_texts(EPOCH_33, DAY_20731); + let err = verify_pack_texts(&h, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err(); + assert!(matches!(err, PackFault::OutOfDate { .. }), "{err}"); + assert_eq!(err.to_string(), "program pack out of date: the pack is for epoch bed7ab62cbece66c day 20731, the node is on epoch 009858237e118f69 day 20731"); + } + + /// A pack that contradicts itself is "disagree", never "out of date": seeds.txt of one epoch with program.h of + /// another (a half rewritten directory), or init words that are not the attempt's words (a worker on the old + /// rule would have produced this verdict for every retried program). + #[test] + fn inconsistent_pack_disagrees() { + let (h33, _, _) = pack_texts(EPOCH_33, DAY_20731); + let s34 = format!("epoch_seed_hex {EPOCH_34}\nday_seed_hex {DAY_20731}\n"); + let err = verify_pack_texts(&h33, Some(&s34), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err(); + assert!(matches!(err, PackFault::Disagree(_)), "{err}"); + assert!(err.to_string().starts_with("program pack and its seeds disagree: seeds.txt names epoch 009858237e118f69"), "{err}"); + + let (h34, s, _) = pack_texts(EPOCH_34, DAY_20731); + let bare = attempt_words(&bytes(EPOCH_34), 0); + let edited = h34.lines().map(|l| if l.starts_with("#define IGNEUM_SEEDW_INIT") { format!("#define IGNEUM_SEEDW_INIT {{ {} }}", bare.iter().map(|w| format!("0x{w:08x}")).collect::>().join(", ")) } else { l.to_string() }).collect::>().join("\n"); + let err = verify_pack_texts(&edited, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err(); + assert!(err.to_string().contains("IGNEUM_SEEDW_INIT is not attempt"), "{err}"); + // and a pack with no attempt line at all is read as attempt 0 (the packs before generator version 2) + let no_attempt = h34.lines().filter(|l| !l.starts_with("#define IGNEUM_PROGRAM_ATTEMPT")).collect::>().join("\n"); + assert!(verify_pack_texts(&no_attempt, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).is_err()); + } + + #[test] + fn day_label_reads_the_index() { + assert_eq!(day_label(DAY_20731), "20731"); + assert_eq!(day_label("abcd"), "abcd"); + } + + /// Counter ASIC 2.0: a class v3 chain pack carries generator 3, the class line and the era seed; it is refused + /// when the chain wants class v2, when the era differs, and a v2 pack is refused when the chain wants v3; a + /// generator this software does not run is refused whatever is wanted. + #[test] + fn program_class_and_era_are_checked() { + let e = bytes(EPOCH_34); + let d = bytes(DAY_20731); + let era = [0x5au8; 32]; + let v3 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V3, "class test"); + let h3 = program_header(&v3.program, "test day", &v3.dataset); + assert!(h3.contains("#define IGNEUM_GENERATOR 3\n")); + assert!(h3.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n")); + assert!(h3.contains(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era)))); + let id = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&era)).unwrap(); + assert_eq!((id.generator, id.class, id.era_hex.as_deref()), (3, ProgramClass::V3, Some(hex(&era).as_str()))); + assert_eq!(id.attempt, v3.program.attempt); + assert!(verify_pack_texts(&h3, None, &e, &d).is_ok(), "no class wanted: either class passes"); + let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V2), None).unwrap_err(); + assert!(matches!(err, PackFault::WrongClass(_)), "{err}"); + assert!(err.to_string().contains("program pack of the wrong class"), "{err}"); + let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&[1u8; 32])).unwrap_err(); + assert!(err.to_string().contains("era seed"), "{err}"); + // the v2 pack of the same seeds: generator 2, no class line, no era line, refused when v3 is wanted + let v2 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V2, "class test"); + let h2 = program_header(&v2.program, "test day", &v2.dataset); + assert!(h2.contains("#define IGNEUM_GENERATOR 2\n")); + assert!(!h2.contains("IGNEUM_PROGRAM_CLASS") && !h2.contains("IGNEUM_ERA_SEED_HEX")); + let plain = Epoch::from_seed_bytes(&e, &d, "class test"); + assert_eq!(program_header(&plain.program, "test day", &plain.dataset), h2, "class v2 from the chain is the v2 export byte for byte"); + let id = verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V2), Some(&era)).unwrap(); + assert_eq!((id.generator, id.class, id.era_hex), (2, ProgramClass::V2, None)); + assert!(matches!(verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V3), None), Err(PackFault::WrongClass(_)))); + // a generator nobody runs + let h9 = h2.replace("#define IGNEUM_GENERATOR 2\n", "#define IGNEUM_GENERATOR 9\n"); + assert!(matches!(verify_pack_texts(&h9, None, &e, &d), Err(PackFault::WrongClass(_)))); + // a class line that contradicts the generator + let bad = h3.replace("#define IGNEUM_PROGRAM_CLASS \"v3\"\n", "#define IGNEUM_PROGRAM_CLASS \"v2\"\n"); + assert!(matches!(verify_pack_texts(&bad, None, &e, &d), Err(PackFault::Disagree(_)))); + } +} diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index 3472165ef..91f9ee31f 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -1,10 +1,137 @@ //! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node //! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs). -use crate::generator::{generate, Instr, Op, Program, ITERATIONS, LANES}; -use crate::memhard::MemhardCpu; +use crate::generator::{generate, generate_class, EraParams, Instr, LoadClass, Op, Program, ProgramClass, ITERATIONS, LANES}; +use crate::memhard::{hot_index, HotTable, Layout, MemhardCpu, Shape}; use crate::seed::day_key; +/// The load address of an era program (`docs/plans/era-layout.md` section 1.3): `y = rotl(x * M, R)`, then the +/// window of the load site, `k = min(win, D - 26)` (0 when `D <= 26`), `idx = ((y & (MASK >> k)) | ((off & +/// (2^k - 1)) << (D - k))) & MASK`. For every other class `idx = x & MASK`, the lottery hash's address. `mask` is +/// `2^D - 1`. The acceptance mirror calls this at the rule's constant `D = 28`. +#[inline(always)] +pub fn load_index(era: Option<&EraParams>, ins: &Instr, x: u32, mask: u32, log2: u32) -> u32 { + match era { + None => x & mask, + Some(e) => { + let (wm, off) = window(ins, mask, log2); + let y = x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot); + ((y & wm) | off) & mask + } + } +} + +/// The window of a load site at a dataset of `2^log2` words: `(window mask, offset)` such that +/// `idx = (y & window mask) | offset` lies in the site's aligned window of `2^(log2 - k)` words. +#[inline(always)] +pub fn window(ins: &Instr, mask: u32, log2: u32) -> (u32, u32) { + let k = (ins.win as u32).min(log2.saturating_sub(26)); + let wm = mask >> k; + let off = ((ins.off as u32) & ((1u32 << k) - 1)) << (log2 - k); + (wm, off) +} + +/// Read-width experiment (5 October 2026): a `load` of `W` words folds every word into `dst`: +/// `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, FOLD_ROT) * FOLD_MUL) XOR w[j]; dst = x`. For `W = 1` this is the +/// lottery hash's `dst XOR dataset[...]`. The fold is state-dependent (the rotate-multiply sits between the words), +/// so no function of the line alone replaces it: two different lines give two different maps of `dst`, and a +/// dataset of folded lines cannot be stored in place of the dataset (see `docs/plans/read-width.md`). +pub const FOLD_ROT: u32 = 11; +pub const FOLD_MUL: u32 = 0x9E3779B1; + +/// The fold of `words` into `dst` (at least one word). +#[inline(always)] +pub fn fold_words(dst: u32, words: &[u32]) -> u32 { + let mut x = dst ^ words[0]; + for &w in &words[1..] { + x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w; + } + x +} + +/// Variant 5 (scratch): the fill value of word `j` (0..2) of slot `slot` of lane `lane` of the unit at base nonce +/// `base`, under program seed words `seed`. The scratch of a unit starts as these values; a slot written during +/// the unit's hash holds what was written. Mirrored as `scr_fill` in every emitted kernel. +#[inline(always)] +pub fn scratch_fill(seed: &[u32; 8], base: u32, lane: u32, slot: u32, j: u32) -> u32 { + splitmix32( + (base.wrapping_add(lane) ^ seed[j as usize]) + .wrapping_add(slot.wrapping_mul(0x9E3779B1)) + .wrapping_add((j + 1).wrapping_mul(0x85EBCA77)), + ) +} + +/// Variant 5: the 16-byte slot after a read-modify-write that read `w` and folded to `x`: `(x ^ w1, rotl(x, 7) ^ w2, +/// x + w0)` behind the slot's tag. +#[inline(always)] +pub fn scratch_rewrite(x: u32, w: &[u32; 3]) -> [u32; 3] { + [x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])] +} + +/// The CPU model of one unit's scratch (variant 5): per lane, the written slots and their words. Unwritten slots +/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots; a GPU keeps the real scratch per +/// resident warp with a per-unit tag per slot. +pub struct ScratchModel { + slots: usize, + written: Vec, + data: Vec<[u32; 3]>, + pub reads: usize, + pub writes: usize, + /// Soundness tests (`tests/scratch.rs`, `docs/analysis/scratch-soundness.md`): when `Some`, every + /// read-modify-write is appended as it happened. `None` on every verification path. + pub trace: Option>, +} + +/// One scratch read-modify-write as the interpreter saw it (variant 5 soundness tests). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ScratchEvent { + pub lane: u8, + pub slot: u32, + /// The slot had been written earlier in this unit (a re-hit): the words read were a rewrite, not the fill. + pub hit: bool, + pub read: [u32; 3], + /// The fold result, the new value of `dst`. + pub x: u32, + pub written: [u32; 3], +} + +impl ScratchModel { + pub fn new(slots_per_lane: usize) -> Self { + Self { + slots: slots_per_lane, + written: vec![false; LANES * slots_per_lane], + data: vec![[0; 3]; LANES * slots_per_lane], + reads: 0, + writes: 0, + trace: None, + } + } + /// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read. + #[inline] + pub fn rmw(&mut self, seed: &[u32; 8], base: u32, lane: usize, slot: u32, dst: u32) -> u32 { + let i = lane * self.slots + slot as usize; + let w = if self.written[i] { + self.data[i] + } else { + [ + scratch_fill(seed, base, lane as u32, slot, 0), + scratch_fill(seed, base, lane as u32, slot, 1), + scratch_fill(seed, base, lane as u32, slot, 2), + ] + }; + let x = fold_words(dst, &w); + let out = scratch_rewrite(x, &w); + if let Some(t) = self.trace.as_mut() { + t.push(ScratchEvent { lane: lane as u8, slot, hit: self.written[i], read: w, x, written: out }); + } + self.data[i] = out; + self.written[i] = true; + self.reads += 1; + self.writes += 1; + x + } +} + /// Dataset element, closed form of (day words, index). The original prototype's six-operation element. #[inline(always)] pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 { @@ -65,24 +192,55 @@ pub struct DatasetSource { /// in packs so any implementation can rebuild the key. Empty when the key was given directly. pub key_bytes: Vec, pub dataset: Dataset, + /// The hot table of the epoch (hot-table experiment, `docs/plans/hot-table.md`): `Some` when the program's + /// class has one; filled by [`Epoch::new_class`] and [`Epoch::from_seed_bytes_class`] from the program's seed + /// bytes. A hot load reads `hot[hot_index(src, words)]`. + pub hot: Option, } impl DatasetSource { /// Build the source for a day. Memory-hard mode fills the 256 MiB cache on the calling thread. pub fn new(day: &str, mode: DatasetMode, log2_words: u32) -> Self { - let mut ds = Self::from_key(day_key(day), mode, log2_words); + Self::new_shape(day, mode, log2_words, Shape::V2) + } + + /// [`DatasetSource::new`] with the construction's shape (mixer multiplier, cache size; Counter ASIC 2.0). + pub fn new_shape(day: &str, mode: DatasetMode, log2_words: u32, shape: Shape) -> Self { + let mut ds = Self::from_key_shape(day_key(day), mode, log2_words, shape); ds.key_bytes = format!("day/{day}").into_bytes(); ds } pub fn from_key(key: [u32; 8], mode: DatasetMode, log2_words: u32) -> Self { + Self::from_key_shape(key, mode, log2_words, Shape::V2) + } + + /// [`DatasetSource::from_key`] with the construction's shape. Memory-hard mode fills a cache of + /// `2^shape.cache_log2_words` words on the calling thread. + pub fn from_key_shape(key: [u32; 8], mode: DatasetMode, log2_words: u32, shape: Shape) -> Self { assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32"); let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 }; let dataset = match mode { DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] }, - DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::new(key)), + DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)), }; - Self { log2_words, mask, key, key_bytes: Vec::new(), dataset } + Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None } + } + + /// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB). + pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self { + self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb)); + self + } + + /// The hot table of a program's class, filled from its seed bytes (none for a class without one). + pub fn attach_hot_for(&mut self, program: &Program) { + self.hot = program.class.hot.map(|h| HotTable::for_seed_bytes(&program.seed_bytes, h.mb as u32)); + } + + /// The shape of the memory-hard construction ([`Shape::V2`] for the closed form, which has none). + pub fn shape(&self) -> Shape { + self.memhard().map(|m| m.shape()).unwrap_or(Shape::V2) } pub fn mode(&self) -> DatasetMode { @@ -99,18 +257,23 @@ impl DatasetSource { } } - /// `dataset[w & mask]`. + /// `dataset[w & mask]` under the linear layout (the lottery hash). pub fn word(&self, w: u32) -> u32 { + self.word_at(Layout::LINEAR, w) + } + + /// `dataset[w & mask]` under a program's layout (era layout). The closed form has no items and ignores it. + pub fn word_at(&self, layout: Layout, w: u32) -> u32 { let w = w & self.mask; match &self.dataset { Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1), - Dataset::MemoryHard(m) => m.word(w), + Dataset::MemoryHard(m) => m.word_at(layout, w), } } /// `out[k] = dataset[idx[k]]`; indices are already masked. Returns items derived (0 for the closed form). #[inline] - fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES]) -> usize { + fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES], layout: Layout) -> usize { match &self.dataset { Dataset::ClosedForm { d0, d1 } => { for k in 0..LANES { @@ -118,7 +281,24 @@ impl DatasetSource { } 0 } - Dataset::MemoryHard(m) => m.fetch(idx, out), + Dataset::MemoryHard(m) => m.fetch(idx, out, layout), + } + } + + /// `out[k][j] = dataset[base[k] + j]` for `j < width`; bases are masked and aligned to `width` words + /// (`width` 4 or 16, so a lane's words lie in one item). Returns items derived (0 for the closed form). + #[inline] + fn fetch_wide(&self, base: &[u32; LANES], width: usize, out: &mut [[u32; 16]; LANES], layout: Layout) -> usize { + match &self.dataset { + Dataset::ClosedForm { d0, d1 } => { + for k in 0..LANES { + for j in 0..width { + out[k][j] = dataset_elem(base[k] + j as u32, *d0, *d1); + } + } + 0 + } + Dataset::MemoryHard(m) => m.fetch_wide(base, width, out, layout), } } } @@ -146,7 +326,23 @@ pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> /// [`interpret_warp`] with explicit init words `I` (section 1.6 of the spec). The packs use `I = program.seed`; /// a block uses `I = bind::block_init_words(H, nonce)`. pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> WarpResult { + interpret_warp_scratch(program, seed, base_nonce, ds, false).0 +} + +/// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order +/// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for +/// a class without a scratch. For the soundness tests of variant 5 only. +pub fn interpret_warp_scratch( + program: &Program, + seed: &[u32; 8], + base_nonce: u32, + ds: &DatasetSource, + trace: bool, +) -> (WarpResult, Vec) { let mask = ds.mask; + let log2 = ds.log2_words; + let era = program.class.era; + let layout = program.class.layout(); let mut r = [[0u32; LANES]; 8]; for lane in 0..LANES { let nonce = base_nonce.wrapping_add(lane as u32); @@ -160,10 +356,29 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, let mut items_derived = 0usize; let mut idx = [0u32; LANES]; let mut val = [0u32; LANES]; + let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None }; + if trace { + if let Some(m) = scratch.as_mut() { + m.trace = Some(Vec::new()); + } + } + let slot_mask = program.class.scratch_slot_mask(); + if program.has_hot() { + let h = ds.hot.as_ref().expect("a hot-table program needs the epoch's hot table on the dataset source"); + assert_eq!(h.n_words(), program.hot_words(), "the hot table's size is the class's"); + } for _ in 0..ITERATIONS { let sel = r[0]; for ins in &program.instrs { - step(ins, &mut r, &sel, mask, ds, &mut idx, &mut val, &mut items_derived); + step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); + if ins.op == Op::Scratch { + let m = scratch.as_mut().expect("a scratch op needs a scratch class"); + let (d, a) = (ins.dst as usize, ins.src as usize); + for lane in 0..LANES { + let slot = r[a][lane] & slot_mask; + r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]); + } + } } } let mut hashes = [0u64; LANES]; @@ -172,7 +387,8 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); hashes[lane] = ((hi as u64) << 32) | lo as u64; } - WarpResult { hashes, items_derived } + let events = scratch.and_then(|m| m.trace).unwrap_or_default(); + (WarpResult { hashes, items_derived }, events) } #[inline(always)] @@ -182,6 +398,9 @@ fn step( r: &mut [[u32; LANES]; 8], sel: &[u32; LANES], mask: u32, + log2: u32, + era: Option<&EraParams>, + layout: Layout, ds: &DatasetSource, idx: &mut [u32; LANES], val: &mut [u32; LANES], @@ -255,22 +474,46 @@ fn step( r[d][lane] ^= src[lane ^ m]; } } - Op::Load => { + Op::Load if ins.width == 1 => { for lane in 0..LANES { - idx[lane] = r[a][lane] & mask; + idx[lane] = load_index(era, ins, r[a][lane], mask, log2); } - *items_derived += ds.fetch(idx, val); + *items_derived += ds.fetch(idx, val, layout); for lane in 0..LANES { r[d][lane] ^= val[lane]; } } + Op::Load => { + // Read-width experiment: `width` words from the aligned address, every word folded into dst. + let width = ins.width as usize; + let align = !(ins.width as u32 - 1); + for lane in 0..LANES { + idx[lane] = load_index(era, ins, r[a][lane], mask, log2) & align; + } + let mut vals = [[0u32; 16]; LANES]; + *items_derived += ds.fetch_wide(idx, width, &mut vals, layout); + for lane in 0..LANES { + r[d][lane] = fold_words(r[d][lane], &vals[lane][..width]); + } + } + Op::Scratch => { + // handled by the caller (interpret_warp_init), which owns the unit's scratch model + } + Op::Hot => { + // Hot-table experiment: one word of the epoch table at the multiply-shift index, plain xor fold. + let h = ds.hot.as_ref().expect("a hot load needs the hot table"); + let n = h.n_words(); + for lane in 0..LANES { + r[d][lane] ^= h.at(hot_index(r[a][lane], n)); + } + } Op::WLoad => { // Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l. let base = (r[a][0] & mask) & !31; for lane in 0..LANES { idx[lane] = base + lane as u32; } - *items_derived += ds.fetch(idx, val); + *items_derived += ds.fetch(idx, val, Layout::LINEAR); for lane in 0..LANES { r[d][lane] ^= val[lane]; } @@ -294,11 +537,39 @@ pub struct Epoch { /// Default dataset size: 2^28 words = 1 GiB. pub const DEFAULT_DATASET_LOG2: u32 = 28; +/// Days a day index lies after the network's genesis day (0 for the genesis day and any day before it). The node's +/// entry; the same function as `memhard::days_since_genesis`. +pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 { + crate::memhard::days_since_genesis(day_index, genesis_day_index) +} + impl Epoch { pub fn new(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32) -> Self { Self { program: generate(seed), dataset: DatasetSource::new(day, mode, dataset_log2) } } + /// [`Epoch::new`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer multiplier + /// shapes the dataset, the cache is the genesis size since a string day has no day index). + pub fn new_class(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass) -> Self { + Self::new_class_day(seed, day, mode, dataset_log2, class, 0) + } + + /// [`Epoch::new_class`] on day `days_since_genesis` of the growth schedule (the cache of + /// `memhard::cache_log2_words` for a class with the growth rule; the dataset size is the caller's). + pub fn new_class_day(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass, days_since_genesis: u64) -> Self { + let shape = Shape::for_class_day(&class, days_since_genesis); + let program = generate_class(seed, class); + let mut dataset = DatasetSource::new_shape(day, mode, dataset_log2, shape); + // hot-table experiment: a hot class fills its table from the seed bytes + dataset.attach_hot_for(&program); + Self { program, dataset } + } + + /// `dataset[w]` as this epoch's program reads it: under the program's layout (era layout; linear for v2). + pub fn dataset_word(&self, w: u32) -> u32 { + self.dataset.word_at(self.program.class.layout(), w) + } + /// The production shape: memory-hard, 1 GiB dataset. pub fn memory_hard(seed: &str, day: &str) -> Self { Self::new(seed, day, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2) @@ -309,13 +580,71 @@ impl Epoch { /// `seed_words_from_bytes(day_bytes)` (`bind::day_bytes`). Memory-hard, 1 GiB dataset. `label` is only /// recorded in emitted packs. pub fn from_seed_bytes(epoch_seed: &[u8], day_bytes: &[u8], label: &str) -> Self { - let program = crate::generator::generate_from_seed_bytes(label, epoch_seed); + Self::from_seed_bytes_class(epoch_seed, day_bytes, label, LoadClass::V2) + } + + /// [`Epoch::from_seed_bytes`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer + /// multiplier shapes the dataset). Day 0 of the growth schedule: the 2^26-word cache and the 2^28-word dataset, + /// which is every devnet pack and vector. A node past the first doubling calls [`Epoch::from_seed_bytes_day`]. + pub fn from_seed_bytes_class(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass) -> Self { + Self::from_seed_bytes_day(epoch_seed, day_bytes, label, class, 0, DEFAULT_DATASET_LOG2) + } + + /// The chain's shape on day `days_since_genesis` (`memhard::days_since_genesis(day_index(header), day_index(genesis))`, + /// the node's two day indices): the program of the class, and under the class's growth rule the cache of + /// `memhard::cache_log2_words(d)` and the dataset of `memhard::dataset_log2_words(genesis_dataset_log2, d)` + /// (the genesis size is 28 for the 1 GiB devnet, 29 for the designed 2 GiB). Without the growth rule the cache + /// is 2^26 words and the dataset `2^genesis_dataset_log2` on every day. + pub fn from_seed_bytes_day(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> Self { + let program = crate::generator::generate_from_seed_bytes_class(label, epoch_seed, class); let key = crate::seed::seed_words_from_bytes(day_bytes); - let mut dataset = DatasetSource::from_key(key, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2); + let shape = Shape::for_class_day(&class, days_since_genesis); + let dataset_log2 = if class.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 }; + let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape); dataset.key_bytes = day_bytes.to_vec(); + dataset.attach_hot_for(&program); Self { program, dataset } } + /// The chain's shape with the program class (Counter ASIC 2.0, 5 October 2026): what the node's engine and the + /// miner's pack export build from the seeds a block template carries. Class v2 is [`Epoch::from_seed_bytes`] + /// exactly (the era bytes are ignored and not recorded); class v3 draws from [`crate::generator::V3_CLASS`] + /// with generator version 3 and records the era seed bytes (`E_n`) in the program for the pack. + pub fn from_chain_seeds(epoch_seed: &[u8], day_bytes: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Self { + Self { program: Self::chain_program(epoch_seed, era_bytes, class, label), dataset: Self::chain_dataset(day_bytes, class) } + } + + /// The program alone of [`Epoch::from_chain_seeds`] (no cache fill): for an engine that shares the day's cache. + pub fn chain_program(epoch_seed: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Program { + crate::generator::generate_from_seed_bytes_program_class(label, epoch_seed, class, era_bytes) + } + + /// The day's cache and dataset of [`Epoch::from_chain_seeds`], the one entry the node's engine builds a day + /// cache through. The class is an argument because the Counter ASIC 2.0 integration gives class v3 its own item + /// construction (the mixer multiplier) and cache size schedule (ca2-mixer); today both classes build the day of + /// [`Epoch::from_seed_bytes`], and the engine keys its day caches on `(day, class)` so the two never share one. + pub fn chain_dataset(day_bytes: &[u8], class: ProgramClass) -> DatasetSource { + Self::chain_dataset_day(day_bytes, class, 0, DEFAULT_DATASET_LOG2) + } + + /// [`Epoch::chain_dataset`] with the day's position since genesis and the network's genesis dataset size: the + /// entry the node's engine and the miner's export build every day cache through, so the cache growth schedule + /// of spec 01 section 1.13.3 has one place to act (ca2-mixer, 5 October 2026, `docs/plans/mixer-x4.md`): the + /// class's load class gives the mixer multiplier and whether the growth rule applies (`Shape::for_class_day`); + /// under the rule the cache is `2^memhard::cache_log2_words(d)` words and the dataset + /// `2^memhard::dataset_log2_words(genesis_dataset_log2, d)`; without it (class v2) the cache is 2^26 words and + /// the dataset the genesis size on every day. `days_since_genesis` is [`days_since_genesis`] of the block's and + /// the genesis header's day indices. + pub fn chain_dataset_day(day_bytes: &[u8], class: ProgramClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> DatasetSource { + let lc = class.load_class(); + let shape = Shape::for_class_day(&lc, days_since_genesis); + let dataset_log2 = if lc.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 }; + let key = crate::seed::seed_words_from_bytes(day_bytes); + let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape); + dataset.key_bytes = day_bytes.to_vec(); + dataset + } + /// The 32 hashes of the warp starting at `base_nonce`. pub fn hash_warp(&self, base_nonce: u32) -> [u64; LANES] { hash_warp(&self.program, base_nonce, &self.dataset) @@ -356,6 +685,190 @@ mod tests { assert_eq!(ds.word(0x0fffffff), 0xf78c84a4); } + /// Read-width experiment: the fold for one word is a plain xor; a wide fetch hands each lane the words the + /// scalar path would; two distinct lines give two distinct maps of dst (one point suffices as a smoke check). + #[test] + fn fold_and_wide_fetch() { + assert_eq!(fold_words(0x1234_5678, &[0xdead_beef]), 0x1234_5678 ^ 0xdead_beef); + let w = [1u32, 2, 3, 4]; + let x = fold_words(7, &w); + let mut y: u32 = 7 ^ 1; + for &v in &w[1..] { + y = y.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ v; + } + assert_eq!(x, y); + assert_ne!(fold_words(7, &[1, 2, 3, 4]), fold_words(7, &[1, 2, 3, 5])); + let ds = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20); + let mut base = [0u32; LANES]; + for (k, b) in base.iter_mut().enumerate() { + *b = ((k as u32).wrapping_mul(0x9E37_79B1) & ds.mask) & !15; + } + let mut out = [[0u32; 16]; LANES]; + let items = ds.fetch_wide(&base, 16, &mut out, Layout::LINEAR); + assert!(items >= 1 && items <= LANES); + for k in 0..LANES { + for j in 0..16 { + assert_eq!(out[k][j], ds.word(base[k] + j as u32), "lane {k} word {j}"); + } + } + let mut base4 = base; + for b in base4.iter_mut() { + *b += 8; + } + let items4 = ds.fetch_wide(&base4, 4, &mut out, Layout::LINEAR); + assert_eq!(items4, items); + for k in 0..LANES { + for j in 0..4 { + assert_eq!(out[k][j], ds.word(base4[k] + j as u32)); + } + } + } + + /// A wide-load program interprets identically on the closed form and through the memory-hard path's fold + /// (the same fold code), and a mixed-class epoch builds and hashes. + /// Variant 5: a fill word is deterministic, a rewrite changes the slot, and a second read of a written slot + /// returns the rewrite, not the fill. + #[test] + fn scratch_model() { + let seed = [1u32, 2, 3, 4, 5, 6, 7, 8]; + assert_eq!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 1)); + assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)); + assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 64, 3, 100, 1)); + let mut m = ScratchModel::new(256); + let w = [scratch_fill(&seed, 32, 3, 100, 0), scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)]; + let x = m.rmw(&seed, 32, 3, 100, 0xabcd); + assert_eq!(x, fold_words(0xabcd, &w)); + let x2 = m.rmw(&seed, 32, 3, 100, 0xabcd); + assert_eq!(x2, fold_words(0xabcd, &scratch_rewrite(x, &w))); + assert_eq!(m.reads, 2); + let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4, 128)); + assert_eq!(e.program.scratch_ops_per_hash(), 32); + assert_eq!(e.hash_warp(0), e.hash_warp(0)); + } + + /// Era layout: the load address stays inside the site's window and below the mask at every dataset size (the + /// window floor of 2^26 words clamps the shrink), the interleaved memory-hard dataset reads item(t(w))[j(w)] and + /// is the same prefix at 2^20 and 2^22 words, the wide fetch agrees word for word, and an era epoch hashes + /// deterministically through the interpreter and the single-nonce API. + #[test] + fn era_windows_layout_and_epochs() { + let eb = EraParams::test_era_bytes("igneum-era-test/1"); + let c = LoadClass::era(LoadClass::V2, &eb, &[1]); + let e = c.era.unwrap(); + let mut ins = Instr { op: Op::Load, dst: 0, src: 1, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 2, off: 3 }; + let mut s = crate::seed::SplitMix64::new(7); + for log2 in [20u32, 26, 27, 28, 29] { + let mask = (1u64 << log2) as u32 - 1; + let k = ins.win.min(log2.saturating_sub(26) as u8) as u32; + for _ in 0..1000 { + let x = s.next() as u32; + let idx = load_index(Some(&e), &ins, x, mask, log2); + assert!(idx <= mask); + let (wm, off) = window(&ins, mask, log2); + assert_eq!(idx & !wm, off, "log2 {log2}"); + assert_eq!(wm, mask >> k); + assert_eq!(idx, ((x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot) & wm) | off) & mask); + } + } + ins.win = 0; + assert_eq!(load_index(None, &ins, 0xdead_beef, 0x0fff_ffff, 28), 0xdead_beef & 0x0fff_ffff); + // the interleaved dataset: one day cache, the layout per program + let l = e.layout(); + assert_eq!(l.pos, [1, 3, 8, 13]); + let small = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20); + let big = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 22); + let m = small.memhard().unwrap(); + for w in [0u32, 1, 4, 5, 255, 256, 4095, 8192, 0x0f_ffff] { + let (t, j) = l.split(w); + assert_eq!(small.word_at(l, w), crate::memhard::derive_item(t, &m.params, &m.cache)[j as usize], "w {w}"); + assert_eq!(small.word_at(l, w), big.word_at(l, w), "prefix at w {w}"); + assert_eq!(small.word(w), small.word_at(Layout::LINEAR, w)); + } + let mut idx = [0u32; LANES]; + for (k, i) in idx.iter_mut().enumerate() { + *i = (k as u32).wrapping_mul(0x9E37_79B1) & small.mask; + } + let mut out = [0u32; LANES]; + small.fetch(&idx, &mut out, l); + for k in 0..LANES { + assert_eq!(out[k], small.word_at(l, idx[k])); + } + // a 16-byte era: the wide fetch keeps a lane's four words in one item + let eb3 = EraParams::test_era_bytes("igneum-era-test/3"); + let c3 = LoadClass::era(LoadClass::fixed(4, 16), &eb3, &[4]); + let l3 = c3.layout(); + assert_eq!(l3.pos[..2], [0, 1]); + let mut base = [0u32; LANES]; + for (k, b) in base.iter_mut().enumerate() { + *b = ((k as u32).wrapping_mul(0x9E37_79B1) & small.mask) & !3; + } + let mut wide = [[0u32; 16]; LANES]; + small.fetch_wide(&base, 4, &mut wide, l3); + for k in 0..LANES { + for j in 0..4 { + assert_eq!(wide[k][j], small.word_at(l3, base[k] + j as u32), "lane {k} word {j}"); + } + } + // era epochs hash deterministically, differ per era, and the single-nonce API agrees with the warp + let mut seen = std::collections::HashSet::new(); + for (n, c) in [(1u64, c), (3, c3)] { + let ep = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::MemoryHard, 20, c); + assert_eq!(ep.dataset_word(5), ep.dataset.word_at(c.layout(), 5)); + let a = ep.hash_warp(64); + assert_eq!(a, ep.hash_warp(64)); + assert_eq!(ep.hash(64 + 5), a[5]); + assert!(seen.insert(a[0]), "era {n}"); + let closed = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c); + assert_ne!(closed.hash_warp(64), a); + } + } + + /// Hot-table experiment: an epoch of a hot class carries the table, hashes deterministically and differs from + /// version 2; the reference interpreter agrees with a hand-stepped hot load; a hot program without its table is + /// refused. + #[test] + fn hot_epochs_hash() { + let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(32, 4)); + let h = e.dataset.hot.as_ref().expect("the epoch fills the hot table"); + assert_eq!(h.n_words(), 1 << 23); + assert_eq!(h.key, crate::memhard::hot_key(b"igneum-genesis")); + let v2 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::V2); + let a = e.hash_warp(0); + assert_eq!(a, e.hash_warp(0)); + assert_ne!(a, v2.hash_warp(0)); + assert_ne!(a[0], a[1]); + // the same program under a 64 MiB table reads other words + let e64 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(64, 4)); + assert_eq!(e64.program.instrs, e.program.instrs); + assert_ne!(e64.hash_warp(0), a); + // from seed bytes, the chain's shape, with a hot class + let genesis = crate::bind::unhex("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07").unwrap(); + let ec = Epoch::from_seed_bytes_class(&genesis, &crate::bind::day_bytes(20_730), "devnet", LoadClass::hot(32, 2)); + assert_eq!(ec.dataset.hot.as_ref().unwrap().key, crate::memhard::hot_key(&genesis)); + assert_eq!(ec.hash_warp(0), ec.hash_warp(0)); + } + + #[test] + #[should_panic(expected = "needs the epoch's hot table")] + fn hot_program_without_a_table_is_refused() { + let p = generate_class("igneum-genesis", LoadClass::hot(32, 4)); + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 20); + let _ = hash_warp(&p, 0, &ds); + } + + #[test] + fn wide_class_epochs_hash() { + for name in ["w16", "w64x4", "50,35,15"] { + let c = LoadClass::parse(name).unwrap(); + let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c); + assert_eq!(e.program.class, c); + let a = e.hash_warp(0); + let b = e.hash_warp(0); + assert_eq!(a, b); + assert_ne!(a[0], a[1]); + } + } + #[test] fn closed_form_genesis_vector_lane0() { // Generator v2 vectors (4 October 2026), proto-cuda/packs/igneum-genesis/vectors.json. diff --git a/igneum-pow/tests/mixer.rs b/igneum-pow/tests/mixer.rs new file mode 100644 index 000000000..e3f090b87 --- /dev/null +++ b/igneum-pow/tests/mixer.rs @@ -0,0 +1,247 @@ +//! The class v3 dataset construction (mixer x4, cache growth option C; `docs/plans/mixer-x4.md`): the soundness +//! runs the brief asks for, on the CPU, with the packs for the GPU runs written on request. +//! +//! 1. Fuzz: `IGNEUM_MIXER_FUZZ` (default 200) programs through the seam (`ProgramClass::V3`), the contract on every +//! instruction (the v2 program of the seed, instruction for instruction), 4 units each across the 32-bit range +//! including the wrap, interpreted twice on the CPU; with `IGNEUM_MIXER_PACKS_OUT=` every program is written +//! as a pack with its 4 bases in vectors.json for `packbench` and the OpenCL host (the Metal fuzz). +//! 2. Stats: bit balance and single-bit-flip avalanche of the v3 hash against v2 on the same programs and nonces. +//! 3. Edge: the dataset at word 0, word MASK and the item boundary, derived through the interpreter's fetch path and +//! by hand at every multiplier 1, 2, 4, 8, on a small cache. +//! 4. Determinism: two independent epochs of the same seed and day agree on every vector and every emitted file. + +use igneum_pow::emit::{export_pack, vectors_json}; +use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION_V3, INSTR_COUNT, V3_CLASS}; +use igneum_pow::memhard::{derive_item, mixer, round_key, Cache, MixParams, Shape}; +use igneum_pow::seed::{day_key, SplitMix64}; +use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch}; +use std::collections::HashMap; +use std::path::PathBuf; + +const DAY: &str = "2026-10-03"; + +fn contract(p: &Program, seed: &str, class: LoadClass) { + if class == V3_CLASS { + assert_eq!(p.generator, GENERATOR_VERSION_V3); + } + assert_eq!(p.class, class); + assert_eq!(p.instrs.len(), INSTR_COUNT); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16); + for (k, i) in p.instrs.iter().enumerate() { + assert!(i.src != i.dst, "#{k}: src == dst"); + assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot); + assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask); + assert!(i.dst < 8 && i.src < 8 && i.src2 < 8); + assert_eq!(i.width, 1, "#{k}: a v3 load reads one word"); + } + assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program"); + let v2 = generate_from_seed_bytes(seed, seed.as_bytes()); + assert_eq!(p.instrs, v2.instrs, "the v2 program of the seed under the v3 construction"); + assert_eq!(p.attempt, v2.attempt); +} + +/// Write a pack whose vectors.json carries `bases` instead of the three standard bases (packbench and the OpenCL +/// host check every unit standalone and the ones inside the batch window). +fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) { + let mut pack = export_pack(e, day, source); + let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect(); + let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true); + for f in pack.files.iter_mut() { + if f.0 == "vectors.json" { + f.1 = vj.clone(); + } + } + pack.write_to(dir).unwrap(); +} + +#[test] +fn fuzz_v3_programs_cpu() { + let n: usize = std::env::var("IGNEUM_MIXER_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200); + let out = std::env::var("IGNEUM_MIXER_PACKS_OUT").ok().map(PathBuf::from); + // IGNEUM_MIXER_CLASS=mx8 fuzzes the x8 candidate as a load class (generator 2 with the class in the id); the + // default is V3_CLASS through the seam + let class = std::env::var("IGNEUM_MIXER_CLASS").ok().map(|s| LoadClass::parse(&s).expect("a load class")).unwrap_or(V3_CLASS); + let mut rng = SplitMix64::new(0x6967_6e65_756d_2d6d); // "igneum-m" + let shape = Shape::for_class(&class); + assert_eq!(shape.cache_log2_words, 26); + assert!(shape.mixer_mult > 1); + // one memory-hard source per dataset size (the 256 MiB cache fill is 0.2 s each) + let mut mh: HashMap = HashMap::new(); + let mut manifest = String::from("pack\tlog2\tprogram_id\tbases\n"); + let mut units = 0usize; + let mut wraps = 0usize; + for i in 0..n { + let seed = format!("igneum-mixer-fuzz/{i}"); + let p = if class == V3_CLASS { + generate_from_seed_bytes_program_class(&seed, seed.as_bytes(), ProgramClass::V3, None) + } else { + generate_from_seed_bytes_class(&seed, seed.as_bytes(), class) + }; + contract(&p, &seed, class); + let b0 = (rng.below(8) as u32) * 32; + let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32); + let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32); + let b3 = (rng.next() as u32) & !31; + let bases = [b0, b1, b2, b3]; + wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count(); + let log2 = [24u32, 26, 28][rng.below(3) as usize]; + let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, log2, shape)); + let e = Epoch { program: p, dataset: ds }; + for &b in &bases { + let r1 = e.interpret_warp(b); + let r2 = e.interpret_warp(b); + assert_eq!(r1.hashes, r2.hashes); + assert!(r1.items_derived >= 120 * 32 / 32 && r1.items_derived <= 4_096, "{seed}: {} items", r1.items_derived); + units += 1; + } + if let Some(dir) = &out { + let pack_name = format!("fuzz-{i:03}-{}-l{log2}", class.name()); + write_pack_with_bases(&dir.join(&pack_name), &e, DAY, &bases, "igneum-pow tests/mixer.rs fuzz"); + manifest.push_str(&format!( + "{pack_name}\t{log2}\t{:016x}\t{}\n", + e.program.program_id(), + bases.iter().map(|b| format!("{b}")).collect::>().join(",") + )); + } + mh.insert(log2, e.dataset); + } + println!("fuzz: {n} {} programs, {units} units on the CPU, {wraps} units in the top 256 nonces", class.name()); + assert_eq!(units, 4 * n); + assert_eq!(wraps, n); + if let Some(dir) = &out { + std::fs::create_dir_all(dir).unwrap(); + std::fs::write(dir.join("manifest.tsv"), manifest).unwrap(); + println!("packs written to {}", dir.display()); + } +} + +/// Bit balance and avalanche of the v3 hash beside v2 on the same program (the TESTS.md section 3 shape, on the +/// CPU, 2^13 nonces per seed): every output bit within 5 sigma of half ones; a single nonce-bit flip moves 50 percent +/// of the output bits within 2 points; no duplicate among the outputs. +#[test] +fn stats_v3_against_v2() { + let n_warps = 256usize; // 8,192 nonces + for seed in ["igneum-genesis", "igneum-genesis/stats1"] { + let v3 = Epoch { + program: generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, None), + dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 24, Shape::for_class(&V3_CLASS)), + }; + let v2 = Epoch::new(seed, DAY, DatasetMode::MemoryHard, 24); + for (name, e) in [("v3", &v3), ("v2", &v2)] { + let mut ones = [0u64; 64]; + let mut outs = Vec::with_capacity(n_warps * 32); + for w in 0..n_warps { + let h = e.hash_warp(w as u32 * 32); + for &x in &h { + outs.push(x); + for b in 0..64 { + ones[b] += (x >> b) & 1; + } + } + } + let total = (n_warps * 32) as f64; + let sigma = (total / 4.0).sqrt(); + for (b, &c) in ones.iter().enumerate() { + let z = (c as f64 - total / 2.0).abs() / sigma; + assert!(z < 5.0, "{seed} {name}: bit {b} ones {c} of {total}, z {z:.2}"); + } + // avalanche: flip one bit of the nonce within the unit (lanes 0..31 differ in the low 5 bits) and across + // units (bit 5 and up): compare lane l of unit u with lane l ^ (1 << k) and with unit u ^ (1 << k) + let mut flips = 0u64; + let mut moved = 0u64; + for w in 0..64usize { + let h = e.hash_warp(w as u32 * 32); + for k in 0..5 { + for l in 0..32usize { + moved += (h[l] ^ h[l ^ (1 << k)]).count_ones() as u64; + flips += 1; + } + } + let h2 = e.hash_warp((w ^ 1) as u32 * 32); + for l in 0..32usize { + moved += (h[l] ^ h2[l]).count_ones() as u64; + flips += 1; + } + } + let avg = moved as f64 / flips as f64 / 64.0 * 100.0; + assert!((avg - 50.0).abs() < 2.0, "{seed} {name}: avalanche {avg:.2} percent"); + outs.sort_unstable(); + let dups = outs.windows(2).filter(|p| p[0] == p[1]).count(); + assert_eq!(dups, 0, "{seed} {name}: duplicate outputs"); + println!("{seed} {name}: {} outputs, avalanche {avg:.2} percent, worst bit z {:.2}", outs.len(), ones.iter().map(|&c| (c as f64 - total / 2.0).abs() / sigma).fold(0.0, f64::max)); + } + assert_ne!(v3.hash_warp(0), v2.hash_warp(0)); + } +} + +/// The dataset edges under every multiplier on a small cache: word 0, word MASK, the last word of item 0 and the +/// first of item 1, through `DatasetSource::word` and by hand. +#[test] +fn edge_items_every_multiplier() { + let key = day_key(DAY); + let cache = Cache::fill_log2(key, 14); + for m in [1u32, 2, 4, 8] { + let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14 }); + let by_hand = |t: u32| -> [u32; 16] { + let mut s = [0u32; 16]; + s[..8].copy_from_slice(&key); + for i in 0..8 { + s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); + } + for r in 0..8usize { + for j in 0..m as usize { + mixer(&mut s, round_key(r * m as usize + j), &mp); + } + let line = cache.line(s[0]); + for i in 0..16 { + s[i] ^= line[i]; + } + } + for j in 0..m as usize { + mixer(&mut s, round_key(8 * m as usize + j), &mp); + } + s + }; + for t in [0u32, 1, 0x0fff_ffff, 0xffff_ffff] { + assert_eq!(derive_item(t, &mp, &cache), by_hand(t), "m {m} item {t}"); + } + } + // the interpreter's fetch path at the genesis cache: words 0, 15, 16 and MASK of a 2^20-word dataset agree with + // the item derivation, under v3 + let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape::for_class(&V3_CLASS)); + let m = ds.memhard().unwrap(); + for w in [0u32, 15, 16, 17, ds.mask - 1, ds.mask] { + assert_eq!(ds.word(w), derive_item(w >> 4, &m.params, &m.cache)[(w & 15) as usize]); + assert_eq!(ds.word(w), m.word(w)); + } + // a load at an out-of-range register masks to the dataset: the word at mask + 1 is the word at 0 + assert_eq!(ds.word(ds.mask.wrapping_add(1)), ds.word(0)); +} + +/// Two independent epochs of the same seed and day: every vector and every emitted file identical; the pinned v3 +/// pack is what a third export writes. +#[test] +fn determinism_v3() { + let build = || Epoch { + program: generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, None), + dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 28, Shape::for_class(&V3_CLASS)), + }; + let a = build(); + let b = build(); + let pa = export_pack(&a, DAY, "a"); + let pb = export_pack(&b, DAY, "a"); + assert_eq!(pa.outs, pb.outs); + assert_eq!(pa.vectors, pb.vectors); + assert_eq!(pa.files, pb.files); + for (w, warp) in [(0u32, 0usize), (4096, 1), (1_000_000, 2)] { + assert_eq!(a.hash_warp(w), pa.outs[warp]); + } + let dir = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer/mx8-genesis"); + for (name, text) in &pa.files { + if name == "vectors.json" || name == "vectors.h" { + continue; // the source string differs ("a" here) + } + let on_disk = std::fs::read_to_string(dir.join(name)).unwrap(); + assert_eq!(&on_disk, text, "{name}"); + } +} diff --git a/igneum-pow/tests/packs.rs b/igneum-pow/tests/packs.rs index 0bab3d9c1..d461bae42 100644 --- a/igneum-pow/tests/packs.rs +++ b/igneum-pow/tests/packs.rs @@ -5,28 +5,52 @@ //! //! Packs: igneum-genesis-mh and igneum-devnet-v4-epoch0 (memory-hard; the latter from the devnet genesis hash as //! the epoch seed and the day bytes of 2026-10-04), igneum-genesis and igneum-hourly (closed-form dataset, -//! interpreter regression only). +//! interpreter regression only); and, under `proto-cuda/packs-ca2-mixer/`, the class v3 packs mx8-genesis and +//! mx8-devnet-epoch0 (Counter ASIC 2.0, 5 October 2026: generator 3 on `V3_CLASS` = mixer x8 with the cache growth +//! rule, decided 22:05 UTC under the delegated rule; the same seeds and days as the two memory-hard v2 packs, so the +//! v2 program and cache carry over and only the dataset words and the hashes change) and the x4 candidate's packs +//! mx4-genesis and mx4-devnet-epoch0 (generator 2 with the load class in the id, the record of the x4 rows). use igneum_pow::accept; use igneum_pow::emit::{ - cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_program, + cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_memhard_for, metal_program, metal_program_bound, opencl_kernel, opencl_kernel_bound, program_header, program_json, LoadSource, }; -use igneum_pow::generator::{generate_from_seed_bytes, Op, GENERATOR_VERSION, LOAD_SLOTS}; -use igneum_pow::memhard::CACHE_WORDS; +use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, LOAD_SLOTS, V3_CLASS}; +use igneum_pow::memhard::{Shape, CACHE_WORDS}; use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch}; use serde_json::Value; use std::path::PathBuf; use std::sync::OnceLock; -const PACKS: [&str; 4] = ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "igneum-genesis", "igneum-hourly"]; +const PACKS: [&str; 8] = [ + "igneum-genesis-mh", + "igneum-devnet-v4-epoch0", + "igneum-genesis", + "igneum-hourly", + "mx8-genesis", + "mx8-devnet-epoch0", + "mx4-genesis", + "mx4-devnet-epoch0", +]; +/// The class v3 packs (generator 3 through the seam). +const PACKS_V3: [&str; 2] = ["mx8-genesis", "mx8-devnet-epoch0"]; fn packs_dir() -> PathBuf { PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs") } +/// The directory a pack lives in: the class v3 packs under packs-ca2-mixer, the rest under packs. +fn pack_dir(pack: &str) -> PathBuf { + if pack.starts_with("mx4-") || pack.starts_with("mx8-") { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer").join(pack) + } else { + packs_dir().join(pack) + } +} + fn read(pack: &str, file: &str) -> String { - let p = packs_dir().join(pack).join(file); + let p = pack_dir(pack).join(file); std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) } @@ -62,9 +86,22 @@ fn epoch(pack: &str) -> &'static Epoch { _ => DatasetMode::ClosedForm, }; let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; - let program = generate_from_seed_bytes(seed, &seed_bytes); + // a class v3 pack: generator 3 on V3_CLASS through the seam, the era bytes it records, the dataset + // in the class's shape on day 0 (the growth rule's genesis cache: every pinned pack is a day-0 size) + let program = match j["generator"].as_u64().unwrap() as u32 { + GENERATOR_VERSION_V3 => { + let era = j.get("era_seed_bytes").map(unhex); + generate_from_seed_bytes_program_class(seed, &seed_bytes, ProgramClass::V3, era.as_deref()) + } + // a generator 2 pack with a load class (the x4 candidate's packs): the class from program.json + _ => match j.get("load_class").and_then(|c| c.as_str()) { + Some(c) => generate_from_seed_bytes_class(seed, &seed_bytes, LoadClass::parse(c).expect("a load class name")), + None => generate_from_seed_bytes(seed, &seed_bytes), + }, + }; + let shape = Shape::for_class(&program.class); let mut dataset = - DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2); + DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2, shape); dataset.key_bytes = day_bytes; (p.to_string(), Epoch { program, dataset }) }) @@ -81,7 +118,8 @@ fn check_program_json(pack: &str) { let j = json(pack, "program.json"); let p = &epoch(pack).program; assert_eq!(j["format"].as_str().unwrap(), "igneum-program-pack-3"); - assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION, "{pack}: generator version"); + assert_eq!(j["generator"].as_u64().unwrap() as u32, p.generator, "{pack}: generator version"); + assert_eq!(p.generator, if PACKS_V3.contains(&pack) { GENERATOR_VERSION_V3 } else { GENERATOR_VERSION }); assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt"); assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id"); let sw: Vec = j["seed_words"].as_array().unwrap().iter().map(hex32).collect(); @@ -129,7 +167,7 @@ fn genesis_program_shape() { #[test] fn mixer_params_match_pack() { - for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0"] { + for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "mx8-genesis", "mx8-devnet-epoch0", "mx4-genesis", "mx4-devnet-epoch0"] { let j = json(pack, "program.json"); let mp = &epoch(pack).dataset.memhard().unwrap().params; let key: Vec = j["dataset"]["key"].as_array().unwrap().iter().map(hex32).collect(); @@ -166,19 +204,21 @@ fn cache_matches_vectors() { fn check_dataset_words(pack: &str) { let v = json(pack, "vectors.json"); - let ds = &epoch(pack).dataset; + let e = epoch(pack); + let ds = &e.dataset; + // a pack's self-test words are read under its program's layout (the era layout; linear for every v2 pack) let head: Vec = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect(); for (i, h) in head.iter().enumerate() { - assert_eq!(ds.word(i as u32), *h, "{pack}: dataset[{i}]"); + assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]"); } let last_index = v["dataset_last_index"].as_u64().unwrap() as u32; assert_eq!(last_index, ds.mask); - assert_eq!(ds.word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]"); + assert_eq!(e.dataset_word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]"); let samples = v["dataset_samples"].as_array().unwrap(); assert_eq!(samples.len(), 64); for s in samples { let idx = s["index"].as_u64().unwrap() as u32; - assert_eq!(ds.word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]"); + assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]"); } } @@ -261,12 +301,15 @@ fn check_sources(pack: &str) { assert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset)); if let Some(mp) = mp { assert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp)); - assert_same_text(pack, "memhard.metal", &metal_memhard(mp)); + assert_same_text(pack, "memhard.metal", &igneum_pow::emit::metal_memhard_layout(mp, p.class.layout())); } let got = program_json(p, &day, &e.dataset); assert_same_text(pack, "program.json", &got); let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON"); - // Every load in every emitted hash kernel has the masked form, and there are exactly 16 of them. + // Every load in every emitted hash kernel has the masked form, and there are exactly 16 of them (a class with + // the era layout inside has the era form instead: `((rotl_imm(rN * M, R) & WM) | OFF) & mask`, checked by + // era_emitted_sources_match_and_loads_have_the_era_form over the era packs, and here by the same count). + let era_load = if p.class.era.is_some() { "((rotl_imm(r" } else { "" }; for (file, load, masked) in [ ("kernel.cu", "ds[r", " & mask]"), ("kernel_bound.cu", "ds[r", " & mask]"), @@ -274,6 +317,7 @@ fn check_sources(pack: &str) { ("program_bound.metal", "dataset[r", " & MASK]"), ] { let text = read(pack, file); + let load = if era_load.is_empty() { load } else { era_load }; assert_eq!(text.matches(load).count(), LOAD_SLOTS, "{pack}/{file}: 16 loads"); assert_eq!(text.matches(masked).count(), LOAD_SLOTS, "{pack}/{file}: 16 masked loads"); } @@ -287,6 +331,85 @@ fn emitted_sources_match_all_packs() { } /// The whole pack as `export` writes it: vectors.json and vectors.h match, and the file list is the full set. +/// The class v3 packs (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): generator 3 on V3_CLASS = mx8; the program of +/// each is the v2 program of the same seed instruction for instruction (v2 loads take no width roll); the cache is +/// the v2 cache (day 0 of the growth rule: 2^26 words, the same FNV-1a 64); the dataset words differ from v2's; +/// program.json, program.h and the emitted memhard core say so; the id carries generator 3. +#[test] +fn v3_packs_are_the_v2_seeds_under_mixer_x8() { + assert_eq!(V3_CLASS.name(), "mx8"); + assert_eq!(V3_CLASS.mixer_mult, 8); + assert!(V3_CLASS.growth); + assert_eq!(LoadClass { mixer_mult: 4, ..V3_CLASS }, LoadClass::MX4, "the x4 candidate differs from v3 in the multiplier alone"); + for (v3, v2) in [("mx8-genesis", "igneum-genesis-mh"), ("mx8-devnet-epoch0", "igneum-devnet-v4-epoch0")] { + let e3 = epoch(v3); + let e2 = epoch(v2); + let j = json(v3, "program.json"); + assert_eq!(j["program_class"].as_str().unwrap(), "v3"); + assert!(j["load_class"].as_str().unwrap().starts_with("mx8"), "{v3}: mx8, or mx8 with the era inside"); + assert_eq!(j["mixer_mult"].as_u64().unwrap(), 8); + assert_eq!(j["cache_growth"].as_bool().unwrap(), true); + assert_eq!(j["dataset"]["mixer_mult"].as_u64().unwrap(), 8); + assert_eq!(j["dataset"]["cache"]["log2_words"].as_u64().unwrap(), 26); + assert_eq!(e3.program.generator, GENERATOR_VERSION_V3); + if e3.program.era_bytes.is_some() { + // the era layout composed into class v3 (docs/plans/era-layout.md, 5 October 2026): a chain pack carries an + // era, so its class is V3_CLASS with the era drawn inside and its stream takes two window draws per + // instruction; the v2 program carries over only in the seed, the attempt and the day + assert_eq!(LoadClass { era: None, ..e3.program.class }, V3_CLASS, "{v3}: the composed class"); + assert!(e3.program.class.era.is_some()); + assert_ne!(e3.program.instrs, e2.program.instrs, "{v3}: the era windows change the stream"); + } else { + assert_eq!(e3.program.class, V3_CLASS); + assert_eq!(e3.program.instrs, e2.program.instrs, "{v3}: the v2 program under the v3 construction"); + } + assert_eq!(e3.program.seed, e2.program.seed); + assert_eq!(e3.program.attempt, e2.program.attempt); + assert_ne!(e3.program.program_id(), e2.program.program_id()); + assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt)); + let m3 = e3.dataset.memhard().unwrap(); + let m2 = e2.dataset.memhard().unwrap(); + assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26 }); + assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0"); + assert_eq!(m3.params.rot, m2.params.rot); + assert_eq!(e3.dataset.log2_words, 28); + assert_ne!(e3.dataset.word(0), e2.dataset.word(0), "{v3}: the dataset words differ"); + assert_ne!(e3.hash_warp(0), e2.hash_warp(0)); + let h = read(v3, "program.h"); + assert!(h.contains("#define IGNEUM_GENERATOR 3\n")); + assert!(h.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n")); + assert!(h.contains("#define IGNEUM_MIXER_MULT 8")); + assert!(h.contains("#define IGNEUM_CACHE_GROWTH 1")); + assert!(h.contains("#define IGNEUM_CACHE_LOG2_WORDS 26\n")); + assert!(h.contains("#define IGNEUM_LOAD_CLASS \"mx8"), "mx8, or mx8 with the era inside"); + for file in ["memhard.h", "memhard.metal", "kernel.cl"] { + let text = read(v3, file); + assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u))").count(), 1, "{v3}/{file}"); + assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u))").count(), 1, "{v3}/{file}"); + } + for file in ["memhard.h", "memhard.metal", "kernel.cl"] { + let text = read(v2, file); + assert_eq!(text.matches("j < 8u").count(), 0, "{v2}/{file}: the v2 text has no multiplier loop"); + } + } + // the devnet v3 pack records era 0's stand-in, the devnet genesis hash + let j = json("mx8-devnet-epoch0", "program.json"); + assert_eq!(unhex(&j["era_seed_bytes"]), unhex(&j["seed_bytes"])); + assert!(read("mx8-devnet-epoch0", "program.h").contains("#define IGNEUM_ERA_SEED_HEX \"edc4fa844da9dc98")); + assert!(json("mx8-genesis", "program.json").get("era_seed_bytes").is_none()); + // the x4 candidate's packs: generator 2, the class in the id, the v2 program of the seed, mixer x4 + for (x4, v2) in [("mx4-genesis", "igneum-genesis-mh"), ("mx4-devnet-epoch0", "igneum-devnet-v4-epoch0")] { + let e4 = epoch(x4); + let j = json(x4, "program.json"); + assert_eq!(e4.program.generator, GENERATOR_VERSION); + assert_eq!(j["load_class"].as_str().unwrap(), "mx4"); + assert_eq!(e4.program.class, LoadClass::MX4); + assert_eq!(e4.program.instrs, epoch(v2).program.instrs); + assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26 }); + assert!(read(x4, "memhard.h").contains("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))")); + } +} + fn check_export(pack: &str) { let e = epoch(pack); let v = json(pack, "vectors.json"); @@ -311,7 +434,7 @@ fn check_export(pack: &str) { expected.extend(["memhard.h", "memhard.metal"]); } assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::>(), expected); - let mut on_disk: Vec = std::fs::read_dir(packs_dir().join(pack)) + let mut on_disk: Vec = std::fs::read_dir(pack_dir(pack)) .unwrap() .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) .filter(|n| !n.starts_with('.')) @@ -350,3 +473,429 @@ fn devnet_pack_is_the_chain_derivation() { assert_eq!(e.program.instrs, epoch("igneum-devnet-v4-epoch0").program.instrs); assert_eq!(e.hash_warp(0), epoch("igneum-devnet-v4-epoch0").hash_warp(0)); } + +// --------------------------------------------------------------------------------------------------------- +// Era layout packs (5 October 2026, docs/plans/era-layout.md): proto-cuda/packs-ca2-era/era-, n in 0..5, the +// devnet epoch seed and day bytes under the era class of test era seed igneum-era-test/, the width pinned at +// 4 bytes (allowed_widths in program.json). Checked like the pinned packs, plus the one load form of 1.3 by text search. +// --------------------------------------------------------------------------------------------------------- + +use igneum_pow::generator::{generate_era, EraParams, V3_ALLOWED}; + +const ERA_PACKS: [&str; 6] = ["era-0", "era-1", "era-2", "era-3", "era-4", "era-5"]; + +fn era_packs_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-era") +} + +fn era_read(pack: &str, file: &str) -> String { + let p = era_packs_dir().join(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +fn era_json(pack: &str, file: &str) -> Value { + serde_json::from_str(&era_read(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")) +} + +/// The era class of a pack: the era bytes from program.json (`era_seed_bytes`, the chain's `E_n`; the test seed +/// `igneum-era-test/` of the pack's number gives the same bytes), the allowed set from `era.allowed_widths` +/// (class v3's `V3_ALLOWED`); the pack's recorded stream words and draw must be the class's. +fn era_class(pack: &str) -> LoadClass { + let j = era_json(pack, "program.json"); + let n: u64 = pack.trim_start_matches("era-").parse().unwrap(); + let eb = unhex(&j["era_seed_bytes"]); + assert_eq!(eb, EraParams::test_era_bytes(&format!("igneum-era-test/{n}")).to_vec(), "{pack}: the era bytes of test seed {n}"); + let allowed: Vec = j["era"]["allowed_widths"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect(); + assert_eq!(allowed, V3_ALLOWED.to_vec(), "{pack}: class v3's width set"); + // the measurement packs of 5 October 2026: the era layout over version 2's construction (mixer x1, the genesis + // cache), generator 3 and the era bytes recorded; the chain's class v3 composes the same draw over LoadClass::MX4 + // (generator tests era_programs_are_accepted and program_classes), and the integration re-exports these packs + let c = LoadClass::era(LoadClass::V2, &eb, &allowed); + let e = c.era.unwrap(); + let words: Vec = j["era"]["seed_words"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(e.words.to_vec(), words, "{pack}: era seed words"); + assert_eq!(e.width_words as u64, j["era"]["width_words"].as_u64().unwrap(), "{pack}: width"); + assert_eq!(e.stride_mul, hex32(&j["era"]["stride_mul"]), "{pack}: stride mul"); + assert_eq!(e.stride_rot as u64, j["era"]["stride_rot"].as_u64().unwrap(), "{pack}: stride rot"); + let pos: Vec = j["era"]["interleave"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect(); + assert_eq!(e.pos.to_vec(), pos, "{pack}: interleave"); + c +} + +fn era_epoch(pack: &str) -> &'static Epoch { + static E: OnceLock> = OnceLock::new(); + let all = E.get_or_init(|| { + ERA_PACKS + .iter() + .map(|p| { + let j = era_json(p, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = unhex(&j["seed_bytes"]); + let day_bytes = unhex(&j["dataset"]["day_bytes"]); + assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let class = era_class(p); + assert_eq!(log2, igneum_pow::verify::DEFAULT_DATASET_LOG2); + let eb = unhex(&j["era_seed_bytes"]); + let program = generate_era(seed, &seed_bytes, LoadClass::V2, &eb, &V3_ALLOWED); + assert_eq!(program.class, class, "{p}: the pack's era class"); + assert_eq!(program.generator, GENERATOR_VERSION_V3); + assert_eq!(program.era_bytes.as_deref(), Some(&eb[..])); + let mut dataset = DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); + dataset.key_bytes = day_bytes; + (p.to_string(), Epoch { program, dataset }) + }) + .collect() + }); + &all.iter().find(|(n, _)| n == pack).unwrap().1 +} + +/// program.json of an era pack: generator, attempt, id, class, every instruction with width, win and off, and the +/// program passes the acceptance rule. +#[test] +fn era_program_json_matches() { + for pack in ERA_PACKS { + let j = era_json(pack, "program.json"); + let p = &era_epoch(pack).program; + assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION_V3, "{pack}: a class v3 pack"); + assert_eq!(j["program_class"].as_str().unwrap(), "v3"); + assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt"); + assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id"); + assert_eq!(j["load_class"].as_str().unwrap(), p.class.name(), "{pack}: class"); + assert_eq!(j["bytes_per_hash"].as_u64().unwrap() as usize, p.bytes_per_hash()); + assert_eq!(p.loads_per_hash(), 8 * LOAD_SLOTS); + assert!(accept::check(p).is_ok(), "{pack}: acceptance"); + let instrs = j["instructions"].as_array().unwrap(); + assert_eq!(instrs.len(), p.instrs.len()); + for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() { + assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op"); + assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64); + assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64); + assert_eq!(hex32(&ji["imm"]), ins.imm); + assert_eq!(ji["width"].as_u64().unwrap(), ins.width as u64, "{pack} #{k} width"); + assert_eq!(ji["win"].as_u64().unwrap(), ins.win as u64, "{pack} #{k} win"); + assert_eq!(ji["off"].as_u64().unwrap(), ins.off as u64, "{pack} #{k} off"); + if ins.op == Op::Load { + assert_eq!(ins.width, p.class.era.unwrap().width_words); + assert!(ins.win <= 2 && (ins.off as u32) < (1u32 << ins.win)); + } + } + } +} + +/// The six era packs are the same program seed under six draws: the instruction lists agree, the widths and layouts +/// follow the draw, and the dataset words differ from the linear layout exactly when the interleave is not linear. +#[test] +fn era_packs_share_the_program_and_differ_in_layout() { + let linear = &epoch("igneum-devnet-v4-epoch0").dataset; + for pack in ERA_PACKS { + let e = era_epoch(pack); + assert_eq!(e.program.seed_bytes, epoch("igneum-devnet-v4-epoch0").program.seed_bytes, "{pack}: the devnet seed"); + let strip = |p: &igneum_pow::generator::Program| { + p.instrs.iter().map(|i| (i.op, i.dst, i.src, i.src2, i.imm, i.imm2, i.rot, i.bit, i.mask, i.win, i.off)).collect::>() + }; + assert_eq!(strip(&e.program), strip(&era_epoch("era-0").program), "{pack}: same stream as era-0"); + let l = e.program.class.layout(); + let same_at_1 = (0..64u32).all(|w| e.dataset_word(w * 977 + 1) == linear.word(w * 977 + 1)); + assert_eq!(same_at_1, l.is_linear(), "{pack}: layout {:?}", l.pos); + assert_eq!(e.dataset_word(0), linear.word(0), "{pack}: word 0 is item 0 word 0 in every layout"); + // the chain's shared day cache serves every era: the day's dataset source is the pinned pack's, bit for bit + assert_eq!(e.dataset.key, linear.key); + assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), linear.memhard().unwrap().cache.fnv1a64()); + } +} + +#[test] +fn era_dataset_words_and_vectors_match() { + for pack in ERA_PACKS { + let v = era_json(pack, "vectors.json"); + let e = era_epoch(pack); + let ds = &e.dataset; + let head: Vec = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect(); + for (i, h) in head.iter().enumerate() { + assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]"); + } + assert_eq!(e.dataset_word(ds.mask), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]"); + for s in v["dataset_samples"].as_array().unwrap() { + let idx = s["index"].as_u64().unwrap() as u32; + assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]"); + } + assert_eq!(ds.memhard().unwrap().cache.fnv1a64(), hex64(&v["cache_fnv1a64"])); + let mut n = 0; + for w in v["warps"].as_array().unwrap() { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let expected: Vec = w["expected"].as_array().unwrap().iter().map(hex64).collect(); + let got = e.hash_warp(base); + for lane in 0..32 { + assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}"); + n += 1; + } + assert_eq!(e.hash(base + 7), expected[7]); + } + assert_eq!(n, 96, "{pack}"); + } +} + +/// Every emitted file of every era pack matches the emitters byte for byte, the export reproduces vectors.json and +/// vectors.h, and every dataset load in every hash kernel has the one era form (no plain `ds[rN & mask]` remains). +#[test] +fn era_emitted_sources_match_and_loads_have_the_era_form() { + for pack in ERA_PACKS { + let e = era_epoch(pack); + let day = era_json(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string(); + let source = era_json(pack, "vectors.json")["source"].as_str().unwrap().to_string(); + let out = export_pack(e, &day, &source); + for (name, text) in &out.files { + let want = era_read(pack, name); + assert!(text == &want, "{pack}/{name} differs from the emitter"); + } + let mut on_disk: Vec = std::fs::read_dir(era_packs_dir().join(pack)) + .unwrap() + .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) + .filter(|n| !n.starts_with('.') && n != "seeds.txt") + .collect(); + on_disk.sort(); + let mut want: Vec = out.files.iter().map(|(n, _)| n.clone()).collect(); + want.sort(); + assert_eq!(on_disk, want, "{pack}: the pack holds the export's files and seeds.txt only"); + let era = e.program.class.era.unwrap(); + let mul = format!("0x{:08x}u", era.stride_mul); + for (file, mask) in [ + ("kernel.cu", "mask"), + ("kernel_bound.cu", "mask"), + ("kernel.cl", "mask"), + ("kernel_bound.cl", "mask"), + ("program.metal", "MASK"), + ("program_bound.metal", "MASK"), + ] { + let text = era_read(pack, file); + let kernels = if file.starts_with("kernel_bound") || file == "kernel.cu" || file == "kernel.cl" || file.starts_with("program") { 1 } else { 1 }; + // kernel_bound.cl carries igneum_hash and igneum_hash_bound: two kernels + let kernels = if file == "kernel_bound.cl" { 2 } else { kernels }; + let era_form: usize = text + .lines() + .filter(|l| l.contains("rotl_imm(r") && l.contains(&format!(" * {mul}, {}u) & ", era.stride_rot)) && l.contains(&format!(") & {mask}"))) + .filter(|l| l.contains("ds[") || l.contains("dataset[") || l.contains("b_ = ")) + .count(); + assert_eq!(era_form, LOAD_SLOTS * kernels, "{pack}/{file}: {} loads of the era form", LOAD_SLOTS * kernels); + let plain = text.lines().filter(|l| l.contains("ds[r") || l.contains("dataset[r")).count(); + assert_eq!(plain, 0, "{pack}/{file}: a load without the era form"); + } + // the layout helpers appear exactly when the layout is not linear + let mh = era_read(pack, "memhard.h"); + assert_eq!(mh.contains("mh_addr("), !era.layout().is_linear(), "{pack}: memhard.h layout helpers"); + } +} + +/// An era pack's dataset is a prefix at every size of at least 2^16 words: the 2^20-word source gives the pack's +/// words below 2^20. +#[test] +fn era_dataset_is_a_prefix_at_smaller_sizes() { + for pack in ["era-1", "era-3"] { + let e = era_epoch(pack); + let small = DatasetSource::from_key(e.dataset.key, DatasetMode::MemoryHard, 20); + let l = e.program.class.layout(); + for w in [0u32, 1, 2, 3, 16, 255, 4096, 65_535, 65_536, 0x000f_ffff] { + assert_eq!(small.word_at(l, w), e.dataset_word(w), "{pack}: w {w}"); + } + } +} + +// --------------------------------------------------------------------------------------------------------- +// Hot-table experiment (5 October 2026, docs/plans/hot-table.md): the five packs under proto-cuda/packs-ca2-hot/ are +// pinned the same way (program, vectors, every emitted file byte for byte), plus the hot table's fingerprint and the +// one-form load check: exactly 16 - k masked dataset loads and exactly k hot loads in every hash kernel. +// --------------------------------------------------------------------------------------------------------- + +const HOT_PACKS: [&str; 8] = ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "hot32k4a", "hot64k4a", "hot96k4a"]; + +fn hot_packs_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-hot") +} + +fn hread(pack: &str, file: &str) -> String { + let p = hot_packs_dir().join(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +fn hjson(pack: &str, file: &str) -> Value { + serde_json::from_str(&hread(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")) +} + +/// The epoch of a hot pack from program.json alone: the class from `load_class`, the program from the seed bytes, +/// the dataset from the day bytes, the hot table from the seed bytes (what `Epoch::from_seed_bytes_class` does). +fn hepoch(pack: &str) -> &'static Epoch { + static E: OnceLock> = OnceLock::new(); + let all = E.get_or_init(|| { + HOT_PACKS + .iter() + .map(|p| { + let j = hjson(p, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = unhex(&j["seed_bytes"]); + let day_bytes = unhex(&j["dataset"]["day_bytes"]); + assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); + let class = LoadClass::parse(j["load_class"].as_str().unwrap()).unwrap(); + assert_eq!(class.name(), *p, "the pack directory is the class name"); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let program = generate_from_seed_bytes_class(seed, &seed_bytes, class); + let mut dataset = + DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); + dataset.key_bytes = day_bytes; + dataset.attach_hot_for(&program); + (p.to_string(), Epoch { program, dataset }) + }) + .collect() + }); + &all.iter().find(|(n, _)| n == pack).unwrap().1 +} + +fn hassert_same_text(pack: &str, file: &str, got: &str) { + let want = hread(pack, file); + if got != want { + let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect()); + for i in 0..gl.len().max(wl.len()) { + let g = gl.get(i).copied().unwrap_or(""); + let w = wl.get(i).copied().unwrap_or(""); + if g != w { + panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1); + } + } + panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len()); + } +} + +#[test] +fn hot_packs_program_and_vectors() { + for pack in HOT_PACKS { + let e = hepoch(pack); + let p = &e.program; + let j = hjson(pack, "program.json"); + let h = p.class.hot.unwrap(); + assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION); + assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt); + assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id"); + let dataset_slots = if h.added { 16 } else { 16 - h.k as usize }; + assert_eq!(j["loads_per_hash"].as_u64().unwrap() as usize, (dataset_slots + h.k as usize) * 8); + assert_eq!(j["hot_table"]["mb"].as_u64().unwrap(), h.mb as u64); + assert_eq!(j["hot_table"]["slots"].as_u64().unwrap(), h.k as u64); + assert_eq!(j["hot_table"]["dataset_slots"].as_u64().unwrap() as usize, dataset_slots); + assert_eq!(j["hot_table"]["words"].as_u64().unwrap() as u32, p.hot_words()); + assert_eq!(j["op_mix"]["hot"].as_u64().unwrap(), h.k as u64, "{pack}: k hot instructions"); + assert_eq!(j["op_mix"]["load"].as_u64().unwrap() as usize, dataset_slots); + assert_eq!(p.items_per_warp(), dataset_slots * 8 * 32); + assert!(accept::check(p).is_ok(), "{pack}: passes the acceptance rule"); + let v2 = &epoch("igneum-genesis-mh").program; + if !h.added { + // replaced form: the version 2 genesis program with k loads redirected (attempt 0 on both) + assert_eq!(p.attempt, v2.attempt); + for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) { + if a.op == Op::Hot { + assert_eq!(b.op, Op::Load); + } else { + assert_eq!(a, b); + } + } + } else { + // added form: 16 + k load slots, so another slot draw and another program; 16 dataset loads stay + assert_ne!(p.instrs, v2.instrs); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16); + assert!(p.instrs.iter().all(|i| i.width == 1)); + } + // the hot table: the pack's head, last line and fingerprint + let v = hjson(pack, "vectors.json"); + let t = e.dataset.hot.as_ref().unwrap(); + assert_eq!(t.n_words(), p.hot_words()); + let head: Vec = v["hot_head"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(&t.words()[..16], &head[..]); + let last: Vec = v["hot_last_line"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(&t.words()[t.words().len() - 16..], &last[..]); + assert_eq!(t.fnv1a64(), hex64(&v["hot_fnv1a64"]), "{pack}: hot_fnv1a64"); + assert_eq!(t.key, igneum_pow::memhard::hot_key(&p.seed_bytes)); + // the cache is the day's, unchanged by the class + assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), 0x48c4f5bf24166b2e); + // 96 vectors + let warps = v["warps"].as_array().unwrap(); + assert_eq!(warps.len(), 3); + for w in warps { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let expected: Vec = w["expected"].as_array().unwrap().iter().map(hex64).collect(); + let got = e.hash_warp(base); + for lane in 0..32 { + assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}"); + } + assert_eq!(e.hash(base + 5), expected[5]); + } + // the dataset words are the day's + let head: Vec = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect(); + for (i, hd) in head.iter().enumerate() { + assert_eq!(e.dataset.word(i as u32), *hd); + } + } + // the same k at three sizes: identical programs, three fingerprints, three vector sets + let a = hepoch("hot32k4"); + let b = hepoch("hot64k4"); + let c = hepoch("hot96k4"); + assert_eq!(a.program.instrs, b.program.instrs); + assert_eq!(b.program.instrs, c.program.instrs); + assert_ne!(a.hash_warp(0), b.hash_warp(0)); + assert_ne!(b.hash_warp(0), c.hash_warp(0)); +} + +#[test] +fn hot_packs_emitted_sources_and_load_forms() { + for pack in HOT_PACKS { + let e = hepoch(pack); + let p = &e.program; + let k = p.class.hot.unwrap().k as usize; + let dataset_loads = p.class.dataset_slots(); + let day = hjson(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string(); + let mp = &e.dataset.memhard().unwrap().params; + hassert_same_text(pack, "kernel.cu", &cuda_kernel(p, Some(mp))); + hassert_same_text(pack, "kernel_bound.cu", &cuda_kernel_bound(p, Some(mp))); + hassert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored)); + hassert_same_text(pack, "program_bound.metal", &metal_program_bound(p, e.dataset.log2_words)); + hassert_same_text(pack, "kernel.cl", &opencl_kernel(p, Some(mp))); + hassert_same_text(pack, "kernel_bound.cl", &opencl_kernel_bound(p, Some(mp))); + hassert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset)); + hassert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp)); + hassert_same_text(pack, "memhard.metal", &metal_memhard_for(p, mp)); + assert_ne!(metal_memhard_for(p, mp), metal_memhard(mp), "{pack}: the hot fill kernel is in memhard.metal"); + let got = program_json(p, &day, &e.dataset); + hassert_same_text(pack, "program.json", &got); + let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON"); + let v = hjson(pack, "vectors.json"); + let out = export_pack(e, &day, v["source"].as_str().unwrap()); + let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 }; + hassert_same_text(pack, "vectors.json", file("vectors.json")); + hassert_same_text(pack, "vectors.h", file("vectors.h")); + assert_eq!(out.files.len(), 12); + // One form per dialect, exactly 16 - k masked dataset loads and k hot loads in every hash kernel; the fill + // kernel is present once per source that builds the table. + for (file, load, masked, hot) in [ + ("kernel.cu", "ds[r", " & mask]", "hot[__umulhi(r"), + ("kernel_bound.cu", "ds[r", " & mask]", "hot[__umulhi(r"), + ("program.metal", "dataset[r", " & MASK]", "hot[mulhi(r"), + ("program_bound.metal", "dataset[r", " & MASK]", "hot[mulhi(r"), + ("kernel.cl", "ds[r", " & mask]", "hot[mul_hi(r"), + ] { + let text = hread(pack, file); + assert_eq!(text.matches(load).count(), dataset_loads, "{pack}/{file}: {dataset_loads} dataset loads"); + assert_eq!(text.matches(masked).count(), dataset_loads, "{pack}/{file}: masked loads"); + assert_eq!(text.matches(hot).count(), k, "{pack}/{file}: {k} hot loads"); + assert!(text.contains(&format!("#define HOT_WORDS 0x{:08x}u", p.hot_words())), "{pack}/{file}: HOT_WORDS literal"); + } + // kernel_bound.cl carries both kernels + let text = hread(pack, "kernel_bound.cl"); + assert_eq!(text.matches("hot[mul_hi(r").count(), 2 * k); + assert_eq!(text.matches("ds[r").count(), 2 * dataset_loads); + for file in ["kernel.cu", "kernel.cl", "kernel_bound.cl", "memhard.metal"] { + assert_eq!(hread(pack, file).matches("igneum_hot_fill(").count(), 1, "{pack}/{file}: one hot fill kernel"); + } + assert_eq!(hread(pack, "memhard.h").matches("void ht_segment(").count(), 1); + let ph = hread(pack, "program.h"); + assert!(ph.contains(&format!("#define IGNEUM_HOT_MB {}", p.class.hot.unwrap().mb))); + assert!(ph.contains(&format!("#define IGNEUM_HOT_SLOTS {k}"))); + assert!(ph.contains("igneum_launch_hot_fill(")); + } +} diff --git a/igneum-pow/tests/scratch.rs b/igneum-pow/tests/scratch.rs new file mode 100644 index 000000000..acf376d14 --- /dev/null +++ b/igneum-pow/tests/scratch.rs @@ -0,0 +1,766 @@ +//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes +//! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results: +//! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count +//! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks. +//! +//! What runs under plain `cargo test`: +//! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as +//! functions (question 1). +//! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class +//! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2). +//! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to +//! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16 +//! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the +//! hand model shown to have teeth (question 3, CPU half). +//! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under +//! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check; +//! the check is shown to fail on four deliberate breaks (question 4). +//! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every +//! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with +//! `IGNEUM_SCRATCH_PACKS_OUT=` it also writes the packs (and the edge packs) for the Metal runs of +//! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape). + +use igneum_pow::emit::{ + cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel, + opencl_kernel_bound, vectors_json, LoadSource, +}; +use igneum_pow::generator::{ + generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT, + ITERATIONS, LANES, +}; +use igneum_pow::seed::{seed_words_from_bytes, SplitMix64}; +use igneum_pow::verify::{ + fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource, + Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT, +}; +use serde_json::Value; +use std::collections::HashMap; +use std::path::PathBuf; + +/// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the +/// RMW shares the readwidth branch measures. +const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"]; + +fn class(name: &str) -> LoadClass { + LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}")) +} + +// --------------------------------------------------------------------------------------------------------------- +// 1. The written words as functions (question 1) +// --------------------------------------------------------------------------------------------------------------- + +/// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x` +/// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform +/// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`. +#[test] +fn rewrite_is_a_bijection_of_the_fold_value() { + let mut rng = SplitMix64::new(0x7363_7261_7463_6801); + for _ in 0..16 { + let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32]; + let x0 = rng.next() as u32; + let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]]; + for i in 0..(1u32 << 16) { + let x = x0.wrapping_add(i); + let out = scratch_rewrite(x, &w); + for j in 0..3 { + // a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are + // distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16 + // bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits + let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff }; + assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x"); + seen[j][key as usize] = true; + } + } + } + // The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a + // rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic). + let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9]; + let x = 0xdead_beef; + let out = scratch_rewrite(x, &w); + assert_eq!(out[0] ^ w[1], x); + assert_eq!((out[1] ^ w[2]).rotate_right(7), x); + assert_eq!(out[2].wrapping_sub(w[0]), x); +} + +/// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its +/// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive +/// nonces no fill word repeats, for 8 slots x 3 words. +#[test] +fn fill_is_a_bijection_of_the_nonce() { + let seed = seed_words_from_bytes(b"igneum-genesis"); + for slot in [0u32, 1, 63, 64, 255, 1023, 2047] { + for j in 0..3u32 { + let mut words: Vec = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect(); + words.sort_unstable(); + words.dedup(); + assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct"); + } + } + // base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l + assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2)); + // and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0 + assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2)); + // the three word positions of one slot and nonce are three different permutation outputs + let f: Vec = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect(); + assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]); +} + +// --------------------------------------------------------------------------------------------------------------- +// 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2) +// --------------------------------------------------------------------------------------------------------------- + +/// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots. +fn expected_distinct(s: usize, n: usize) -> f64 { + let s = s as f64; + s * (1.0 - (1.0 - 1.0 / s).powi(n as i32)) +} + +struct ClassStats { + units: usize, + events: usize, + hits: usize, + /// ones count per bit of the written words, 3 x 32 + ones: [[u64; 32]; 3], + /// ones count per bit of written XOR read (the change the rewrite makes to the slot) + delta_ones: [[u64; 32]; 3], + /// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch) + depth: Vec, + max_depth: usize, + /// how often each slot index was addressed (the slot comes from a register's low bits) + slot_hist: Vec, +} + +fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats { + let c = class(name); + let mut st = ClassStats { + units: 0, + events: 0, + hits: 0, + ones: [[0; 32]; 3], + delta_ones: [[0; 32]; 3], + depth: vec![0; 256], + max_depth: 0, + slot_hist: vec![0; c.scratch_slots_per_lane()], + }; + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28); + for seed in seeds { + let p = generate_class(seed, c); + assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS); + for u in 0..units_per_seed { + let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000); + let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true); + assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES); + let mut count: HashMap<(u8, u32), usize> = HashMap::new(); + for e in &ev { + assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch"); + let d = count.entry((e.lane, e.slot)).or_insert(0); + assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history"); + assert_eq!(e.written, scratch_rewrite(e.x, &e.read)); + if !e.hit { + let fill = [ + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0), + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1), + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2), + ]; + assert_eq!(e.read, fill, "a first touch reads the fill"); + } + st.depth[(*d).min(255)] += 1; + st.max_depth = st.max_depth.max(*d); + st.slot_hist[e.slot as usize] += 1; + *d += 1; + st.events += 1; + st.hits += e.hit as usize; + for j in 0..3 { + for b in 0..32 { + st.ones[j][b] += ((e.written[j] >> b) & 1) as u64; + st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64; + } + } + } + st.units += 1; + } + } + st +} + +/// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over +/// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday +/// formula within 3 percent relative. The table printed here is the one in the analysis. +#[test] +fn written_words_unbiased_and_rehit_rates() { + let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"]; + let units = 1usize << 11; + println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma"); + for name in CLASSES { + let c = class(name); + let st = class_stats(name, &seeds, units); + let n = st.events as f64; + let sigma = (n / 4.0).sqrt(); + let mut worst = 0.0f64; + let mut worst_delta = 0.0f64; + for j in 0..3 { + for b in 0..32 { + let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma; + let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma; + assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma"); + assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma"); + worst = worst.max(z); + worst_delta = worst_delta.max(zd); + } + } + let per_lane_hash = c.scratch_slots() * ITERATIONS; + let s = c.scratch_slots_per_lane(); + let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash); + let exp_pct = 100.0 * exp_hits / per_lane_hash as f64; + let got_pct = 100.0 * st.hits as f64 / st.events as f64; + // chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and + // the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2) + let expect_per_slot = n / s as f64; + let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum(); + let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt(); + let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot; + let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot; + println!( + "{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}", + st.events, st.hits, st.max_depth + ); + // a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it + assert!( + got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct, + "{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%" + ); + // depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit + let shown: Vec = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect(); + println!(" depth histogram {}", shown.join(" ")); + } +} + +// --------------------------------------------------------------------------------------------------------------- +// 3. Hand-built edge programs against an independent hand model (question 3, CPU half) +// --------------------------------------------------------------------------------------------------------------- + +fn ins(op: Op, dst: u8, src: u8) -> Instr { + Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 } +} +fn add_imm(dst: u8, src: u8, imm: u32) -> Instr { + Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 } +} + +/// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are +/// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a +/// register as the Swift edge set does. +fn edge(name: &str, c: LoadClass, instrs: Vec) -> Program { + let seed_string = format!("igneum-scratch-edge/{name}"); + let seed_bytes = seed_string.as_bytes().to_vec(); + let k = instrs.iter().filter(|i| i.op == Op::Scratch).count(); + assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count"); + Program { + seed: seed_words_from_bytes(&seed_bytes), + seed_string, + seed_bytes, + generator: GENERATOR_VERSION, + attempt: 0, + class: c, + era_bytes: None, + instrs, + } +} + +/// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program). +fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> { + let m = LoadClass::scratch(1, kb).scratch_slot_mask(); + let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3]; + let scr = |n: usize, src: u8| -> Vec { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() }; + let mut v = Vec::new(); + // every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane + let mut p = vec![ins(Op::Sub, 1, 1)]; + p.extend(scr(8, 1)); + v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot MASK through the in-range register MASK + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)]; + p.extend(scr(8, 1)); + v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot MASK through the out-of-range register 0xffffffff + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)]; + p.extend(scr(8, 1)); + v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot 0 through the out-of-range register MASK + 1 + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))]; + p.extend(scr(8, 1)); + v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p))); + // 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash + let mut p = vec![ins(Op::Sub, 1, 1)]; + p.extend(scr(16, 1)); + v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p))); + // one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing + // (r5 is the slot register and is never a destination here) + let p: Vec = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect(); + v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p))); + // alternating slot 0 and slot MASK inside one iteration + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)]; + for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() { + // r1 and r2 hold the two slots and are never destinations + p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 })); + } + v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p))); + v +} + +/// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its +/// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth. +fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] { + let seed = &p.seed; + let m = p.class.scratch_slot_mask(); + let mut r = [[0u32; LANES]; 8]; + for lane in 0..LANES { + let nonce = base.wrapping_add(lane as u32); + for i in 0..8 { + let mut x = nonce ^ seed[i]; + x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1)); + x = splitmix32(x); + r[i][lane] = x ^ seed[(i + 1) & 7]; + } + } + let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new(); + for _ in 0..ITERATIONS { + let sel = r[0]; + for ins in &p.instrs { + let (d, a) = (ins.dst as usize, ins.src as usize); + match ins.op { + Op::Sub => { + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]); + } + } + Op::Add => { + for lane in 0..LANES { + let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm }; + r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c); + } + } + Op::Scratch => { + for lane in 0..LANES { + let slot = r[a][lane] & m; + let w = *store.entry((lane, slot)).or_insert_with(|| { + let mut f = [0u32; 3]; + for j in 0..3u32 { + // the fill, written out in full rather than through verify::scratch_fill + let n = base.wrapping_add(lane as u32); + f[j as usize] = splitmix32( + (n ^ seed[j as usize]) + .wrapping_add(slot.wrapping_mul(0x9E37_79B1)) + .wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)), + ); + } + f + }); + let mut x = r[d][lane] ^ w[0]; + x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1]; + x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2]; + r[d][lane] = x; + let out = if mutate { + [x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])] + } else { + [x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])] + }; + store.insert((lane, slot), out); + } + } + other => panic!("the hand model does not implement {other:?}"), + } + } + } + let mut out = [0u64; 32]; + for lane in 0..LANES { + let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21); + let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); + out[lane] = ((hi as u64) << 32) | lo as u64; + } + out +} + +/// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent +/// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32. +const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0]; + +#[test] +fn edge_programs_match_the_hand_model() { + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24); + let mut cases = 0; + for kb in [32u8, 128] { + for (name, what, p) in edge_programs(kb) { + let slots = p.class.scratch_slots_per_lane(); + for base in EDGE_BASES { + let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true); + let hand = hand_model(&p, base, false); + assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})"); + assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth"); + // the slots the trace saw are the ones the program was built to drive + let slot_set: std::collections::BTreeSet = ev.iter().map(|e| e.slot).collect(); + let m = (slots - 1) as u32; + match name.as_str() { + "slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::>(), vec![0]), + "slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::>(), vec![m]), + "twoslots" => assert_eq!(slot_set.into_iter().collect::>(), vec![0, m]), + "lanevar" => { + for e in &ev { + assert!(e.slot <= m); + } + } + _ => unreachable!(), + } + // the chain depth on the driven slot: every RMW after the first per lane is a re-hit + let per_lane = p.scratch_ops_per_hash(); + let hits = ev.iter().filter(|e| e.hit).count(); + let expected_hits = match name.as_str() { + "twoslots" => (per_lane - 2) * LANES, + _ => (per_lane - 1) * LANES, + }; + assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits"); + cases += 1; + } + } + } + assert_eq!(cases, 2 * 7 * 4); +} + +// --------------------------------------------------------------------------------------------------------------- +// 4. The static scratch check over every emitted kernel of every scr pack (question 4) +// --------------------------------------------------------------------------------------------------------------- + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Dialect { + Metal, + Cuda, + OpenCl, +} + +/// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the +/// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else +/// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check: +/// the guarantee is that the emitter has one template and it masks. +pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> { + assert!(kernels >= 1); + // every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound + let k = k * kernels; + assert!(slots.is_power_of_two() && slots >= 1); + let mask = (slots - 1) as u32; + let wpl = slots * 4; + let (u, load, store, ptr) = match dialect { + Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"), + Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"), + Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"), + }; + let count = |needle: &str| text.matches(needle).count(); + let mut errs = Vec::new(); + let mut expect = |what: &str, got: usize, want: usize| { + if got != want { + errs.push(format!("{what}: {got}, expected {want}")); + } + }; + // k slot computations, each masked with exactly the class's mask and immediately followed by the one load form + expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k); + expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k); + expect("stores of the tagged slot", count(store), k); + expect("tag compares", count("(v_.x == tag)"), k); + expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k); + // the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW) + expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels); + expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k); + expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels); + expect("direct scratch indexing", count("scratch["), 0); + expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels); + // no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_` + let any_mask_load = count(&format!("u; {load}")); + expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k); + if errs.is_empty() { + Ok(()) + } else { + Err(errs.join("; ")) + } +} + +fn packs_rw_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth") +} + +fn scr_packs() -> Vec { + let mut v: Vec = std::fs::read_dir(packs_rw_dir()) + .unwrap() + .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) + .filter(|n| n.starts_with("scr")) + .collect(); + v.sort(); + v +} + +fn read_pack(pack: &str, file: &str) -> String { + let p = packs_rw_dir().join(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +/// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel +/// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot +/// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena +/// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first. +#[test] +fn scr_packs_regenerate_and_pass_the_static_scratch_check() { + let packs = scr_packs(); + assert!(packs.len() >= 6, "the scr packs: {packs:?}"); + let mut checked = 0; + let mut sample_metal = String::new(); + let mut sample_cl = String::new(); + let mut sample_cu = String::new(); + let mut sample_k = 0; + let mut sample_slots = 0; + for pack in &packs { + let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap(); + let name = j["load_class"].as_str().unwrap(); + let c = class(name); + assert_eq!(&format!("{name}"), pack, "pack directory named after its class"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap(); + let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap(); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); + let program = generate_from_seed_bytes_class(seed, &seed_bytes, c); + assert_eq!(program.class, c); + assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap()); + let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); + dataset.key_bytes = day_bytes; + let e = Epoch { program, dataset }; + let p = &e.program; + let mp = e.dataset.memhard().map(|m| &m.params); + let k = c.scratch_slots(); + let slots = c.scratch_slots_per_lane(); + assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS); + for (file, text, dialect, kernels) in [ + ("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1), + ("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1), + ("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1), + ("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1), + ("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1), + // the OpenCL bound file carries igneum_hash and igneum_hash_bound + ("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2), + ] { + let on_disk = read_pack(pack, file); + assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text"); + // scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0 + scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")); + checked += 1; + } + // the vectors of the pack are the CPU's + let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap(); + for w in v["warps"].as_array().unwrap() { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let got = e.hash_warp(base); + for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() { + let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap(); + assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}"); + } + } + if k == 4 && slots == 64 { + sample_metal = read_pack(pack, "program.metal"); + sample_cl = read_pack(pack, "kernel.cl"); + sample_cu = read_pack(pack, "kernel.cu"); + sample_k = k; + sample_slots = slots; + } + } + assert_eq!(checked, packs.len() * 6); + println!("static scratch check: {checked} kernels over {} scr packs", packs.len()); + + // The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken + // case). Each must be caught; the message names what. + assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set"); + let mask = format!(" & {}u; uint4 v_", sample_slots - 1); + let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1); + assert_ne!(broken_mask, sample_metal); + let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 1 (one mask dropped, Metal): {e}"); + let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;"); + let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}"); + println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}"); + let wrong_stride = sample_metal.replace("* 256u;", "* 128u;"); + let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("arena definition: 0, expected 1"), "{e}"); + println!("break 3 (arena stride 256 -> 128 words, Metal): {e}"); + let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n"); + let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}"); + println!("break 4 (a stray arena and scratch access, Metal): {e}"); + let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 5 (one mask dropped, OpenCL): {e}"); + let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 6 (one mask dropped, CUDA): {e}"); + // and the unbroken texts pass under the same calls + scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap(); + scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap(); + scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap(); + // a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class) + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err()); + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err()); + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err()); +} + +// --------------------------------------------------------------------------------------------------------------- +// 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit +// range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4) +// --------------------------------------------------------------------------------------------------------------- + +/// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases. +fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> { + let mut pack = export_pack(e, day, source); + let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect(); + let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true); + for f in pack.files.iter_mut() { + if f.0 == "vectors.json" { + f.1 = vj.clone(); + } + } + pack.write_to(dir).unwrap(); + outs +} + +fn contract(p: &Program) { + assert_eq!(p.instrs.len(), INSTR_COUNT); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots()); + assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op"); + for (k, i) in p.instrs.iter().enumerate() { + assert!(i.src != i.dst, "#{k}: src == dst"); + assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot); + assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask); + assert!(i.dst < 8 && i.src < 8 && i.src2 < 8); + assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads"); + } + assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program"); +} + +#[test] +fn fuzz_scr_programs_cpu() { + let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200); + let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from); + let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s" + let day = "2026-10-03"; + let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28); + // memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written + let mut mh: HashMap = HashMap::new(); + let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n"); + let mut per_class: HashMap = HashMap::new(); + let mut units = 0usize; + let mut wraps = 0usize; + if let Some(dir) = &out { + std::fs::create_dir_all(dir).unwrap(); + // the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases + for kb in [32u8, 128] { + for (name, _what, p) in edge_programs(kb) { + let log2 = 24; + let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2)); + let e = Epoch { program: p, dataset: ds }; + let pack_name = format!("edge-{name}-k{kb}"); + write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge"); + manifest.push_str(&format!( + "{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n", + e.program.class.name(), + e.program.program_id(), + e.program.scratch_ops_per_hash(), + EDGE_BASES.iter().map(|b| format!("{b}")).collect::>().join(",") + )); + mh.insert(log2, e.dataset); + } + } + } + for i in 0..n { + let name = CLASSES[rng.below(CLASSES.len() as u64) as usize]; + let c = class(name); + let seed = format!("igneum-scratch-fuzz/{i}"); + let p = generate_class(&seed, c); + contract(&p); + *per_class.entry(name.to_string()).or_insert(0) += 1; + // four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in + // the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform + let b0 = (rng.below(8) as u32) * 32; + let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32); + let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32); + let b3 = (rng.next() as u32) & !31; + let bases = [b0, b1, b2, b3]; + // an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent + // kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base + wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count(); + // the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots + for &b in &bases { + let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true); + let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0; + assert_eq!(r1.hashes, r2.hashes); + assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES); + assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32)); + units += 1; + } + if let Some(dir) = &out { + let log2 = [24u32, 26, 28][rng.below(3) as usize]; + let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2)); + let e = Epoch { program: p, dataset: ds }; + let pack_name = format!("fuzz-{i:03}-{name}-l{log2}"); + write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz"); + manifest.push_str(&format!( + "{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n", + e.program.program_id(), + e.program.scratch_ops_per_hash(), + bases.iter().map(|b| format!("{b}")).collect::>().join(",") + )); + mh.insert(log2, e.dataset); + } else { + let _ = rng.below(3); + } + } + let mut classes: Vec<_> = per_class.iter().collect(); + classes.sort(); + println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}"); + assert_eq!(units, 4 * n); + assert_eq!(wraps, n, "every program has a unit in the top 256 nonces"); + if let Some(dir) = &out { + std::fs::write(dir.join("manifest.tsv"), manifest).unwrap(); + println!("packs written to {}", dir.display()); + } +} + +/// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill +/// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold +/// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the +/// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot. +#[test] +fn slot_is_replayable_from_its_fold_values() { + let seed = seed_words_from_bytes(b"igneum-genesis"); + let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32); + let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)]; + let mut rng = SplitMix64::new(99); + let dsts: Vec = (0..64).map(|_| rng.next() as u32).collect(); + // the honest sequence: read, fold, rewrite, 64 times + let mut w = fill; + let mut xs = Vec::new(); + for &d in &dsts { + let x = fold_words(d, &w); + xs.push(x); + w = scratch_rewrite(x, &w); + } + // the replay: from the fill and the stored fold values alone + let mut w2 = fill; + for &x in &xs { + w2 = scratch_rewrite(x, &w2); + } + assert_eq!(w, w2); + // and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on + // every earlier fold value (drop one and the chain diverges) + let mut w3 = fill; + for (i, &x) in xs.iter().enumerate() { + if i != 10 { + w3 = scratch_rewrite(x, &w3); + } + } + assert_ne!(w, w3); +}